Verbatim-safe typography, ^-punctuation quoting, full-range ^UUUU^, polyglot html
Typographic transforms (---, quote pairs, ~) no longer touch verbatim text: @c/@code/@source_listing content and ^'...'^ spans show exactly the characters written. "^" before any punctuation character quotes it in every target (the apostrophe excepted: ^' opens a literal span), with the new :resolve option on @@@target declaring per-target renderings. The ^UUUU^ code-point form accepts 4-6 hex digits, the full Unicode range. The html output and transform spellings are polyglot (XML-valid), in preparation for an EPUB target. New suites: transform_test, character_test (engine), typography_test (SKS). (from dev 07ce5ea86a0a)
This commit is contained in:
@@ -13,15 +13,39 @@ def unescape_ktesc(s):
|
||||
return result
|
||||
return re.sub(r'KTESC([0-9a-f]+)KTESC', replace, s)
|
||||
|
||||
def escape_ktesc(s):
|
||||
"""A KTESC marker for the characters of s -- the mirror of
|
||||
Target::escape_marker() in mac/target.cpp (4 lowercase hex digits per
|
||||
byte). The marker is inert through every re-read and through the
|
||||
target's typographic transform pass, and decodes back to the characters
|
||||
at final escape resolution (ktesc_resolve)."""
|
||||
return 'KTESC' + ''.join(f'{ord(c):04x}' for c in s) + 'KTESC'
|
||||
|
||||
# The characters the SKS targets' typographic transforms act on: the
|
||||
# hyphen runs (-- and ---), the quote conventions (` ' `` ''), and ~.
|
||||
# Verbatim text (@c, @code, @source_listing) must reach the output exactly
|
||||
# as written, so the code klammers hide each of these as a KTESC marker --
|
||||
# a transform source can then never match -- and the markers decode after
|
||||
# the transform pass has run.
|
||||
# SYNC: the transform tables of the html and txt targets in
|
||||
# sks/target/target.k. A transform built from a new character needs it
|
||||
# added here, or verbatim text will show the transformed form.
|
||||
TYPOGRAPHIC_CHARS = "-'`~"
|
||||
|
||||
def hide_typographic(s):
|
||||
"""Hide the typographically active characters of s as KTESC markers so
|
||||
the target's transform pass cannot change verbatim text."""
|
||||
for c in TYPOGRAPHIC_CHARS:
|
||||
s = s.replace(c, escape_ktesc(c))
|
||||
return s
|
||||
|
||||
class Klammer_base:
|
||||
def __init__(self, K):
|
||||
#args = {k:escape(v) for k, v in K.__dict__.items()
|
||||
args = {k:v for k, v in K.__dict__.items()
|
||||
if not k.startswith('__')}
|
||||
|
||||
for key in args:
|
||||
setattr(self, key, args[key])
|
||||
|
||||
#setattr(self, "_klammer_name", klammer_name)
|
||||
#pprint.pprint(self.__dict__)
|
||||
|
||||
|
||||
@@ -21,10 +21,11 @@ def msg(text=""):
|
||||
|
||||
def escape(s):
|
||||
result = s
|
||||
|
||||
# These were commented out -- has it been replaced?
|
||||
# It's only used in kutil/klammer_base.py and block/block.py
|
||||
# and does nothing now.
|
||||
#result = re.sub(r"\b", r"\b", result)
|
||||
#result = re.sub(r"\t", r"\t", result)
|
||||
|
||||
#result = re.sub("\f", r"\\f", result)
|
||||
#result = re.sub("\v", r"\\v", result)
|
||||
#result = re.compile("\\(.)").sub(r"\\\1", result)
|
||||
|
||||
Reference in New Issue
Block a user