Verbatim-safe typography, ^-punctuation quoting, full-range ^UUUU^, polyglot html

Typographic transforms (---, quote pairs, ~) no longer touch verbatim
text: @c/@code/@source_listing content and ^'...'^ spans show exactly the
characters written. "^" before any punctuation character quotes it in
every target (the apostrophe excepted: ^' opens a literal span), with the
new :resolve option on @@@target declaring per-target renderings. The
^UUUU^ code-point form accepts 4-6 hex digits, the full Unicode range.
The html output and transform spellings are polyglot (XML-valid), in
preparation for an EPUB target. New suites: transform_test, character_test
(engine), typography_test (SKS).

(from dev 07ce5ea86a0a)
This commit is contained in:
2026-08-23 20:48:26 +02:00
parent d982c0d6cc
commit 37b6ba1c4f
77 changed files with 1665 additions and 1608 deletions

View File

@@ -13,15 +13,39 @@ def unescape_ktesc(s):
return result
return re.sub(r'KTESC([0-9a-f]+)KTESC', replace, s)
def escape_ktesc(s):
"""A KTESC marker for the characters of s -- the mirror of
Target::escape_marker() in mac/target.cpp (4 lowercase hex digits per
byte). The marker is inert through every re-read and through the
target's typographic transform pass, and decodes back to the characters
at final escape resolution (ktesc_resolve)."""
return 'KTESC' + ''.join(f'{ord(c):04x}' for c in s) + 'KTESC'
# The characters the SKS targets' typographic transforms act on: the
# hyphen runs (-- and ---), the quote conventions (` ' `` ''), and ~.
# Verbatim text (@c, @code, @source_listing) must reach the output exactly
# as written, so the code klammers hide each of these as a KTESC marker --
# a transform source can then never match -- and the markers decode after
# the transform pass has run.
# SYNC: the transform tables of the html and txt targets in
# sks/target/target.k. A transform built from a new character needs it
# added here, or verbatim text will show the transformed form.
TYPOGRAPHIC_CHARS = "-'`~"
def hide_typographic(s):
"""Hide the typographically active characters of s as KTESC markers so
the target's transform pass cannot change verbatim text."""
for c in TYPOGRAPHIC_CHARS:
s = s.replace(c, escape_ktesc(c))
return s
class Klammer_base:
def __init__(self, K):
#args = {k:escape(v) for k, v in K.__dict__.items()
args = {k:v for k, v in K.__dict__.items()
if not k.startswith('__')}
for key in args:
setattr(self, key, args[key])
#setattr(self, "_klammer_name", klammer_name)
#pprint.pprint(self.__dict__)

View File

@@ -21,10 +21,11 @@ def msg(text=""):
def escape(s):
result = s
# These were commented out -- has it been replaced?
# It's only used in kutil/klammer_base.py and block/block.py
# and does nothing now.
#result = re.sub(r"\b", r"\b", result)
#result = re.sub(r"\t", r"\t", result)
#result = re.sub("\f", r"\\f", result)
#result = re.sub("\v", r"\\v", result)
#result = re.compile("\\(.)").sub(r"\\\1", result)