Verbatim-safe typography, ^-punctuation quoting, full-range ^UUUU^, polyglot html
Typographic transforms (---, quote pairs, ~) no longer touch verbatim text: @c/@code/@source_listing content and ^'...'^ spans show exactly the characters written. "^" before any punctuation character quotes it in every target (the apostrophe excepted: ^' opens a literal span), with the new :resolve option on @@@target declaring per-target renderings. The ^UUUU^ code-point form accepts 4-6 hex digits, the full Unicode range. The html output and transform spellings are polyglot (XML-valid), in preparation for an EPUB target. New suites: transform_test, character_test (engine), typography_test (SKS). (from dev 07ce5ea86a0a)
This commit is contained in:
@@ -1,331 +1,93 @@
|
||||
import sys, os, re, textwrap, pprint
|
||||
|
||||
basedir = f'{os.environ["KLAMMERTEXT_HOME"]}/sks'
|
||||
sys.path = [f"{basedir}/kutil"] + sys.path
|
||||
sys.path = [f"{basedir}/target"] + sys.path
|
||||
|
||||
import kutil
|
||||
import html_util
|
||||
|
||||
def tex_to_pdf(text):
|
||||
print("in tex_to_pdf")
|
||||
|
||||
|
||||
|
||||
def expand_whitespace_markers(text, K=None):
|
||||
def count(count_match):
|
||||
count = int(count_match) if count_match else 1
|
||||
return count
|
||||
def replace_spaces(match):
|
||||
return " " * count(match.group(1))
|
||||
def replace_newlines(match):
|
||||
return "\n" * count(match.group(1))
|
||||
|
||||
result = text
|
||||
|
||||
# result = re.sub(r"\s*#-\s*", "", result)
|
||||
|
||||
# space_pat = re.compile(r" *#\+(\d*) *", re.S)
|
||||
# print("hits:", space_pat.findall(text))
|
||||
|
||||
# for p in space_pat.findall(text):
|
||||
# print("FOUND:", p)
|
||||
|
||||
|
||||
#
|
||||
#result = re.sub(" ", "SPACE", result)
|
||||
#result = re.sub("#/", "\n", result)
|
||||
#result = re.sub("X", " ", result)
|
||||
|
||||
result = re.compile(r" *#- *", re.S).sub("", result)
|
||||
result = re.compile(r" *#\+(\d*) *", re.S).sub(replace_spaces, result)
|
||||
result = re.compile(r"\s*\#\/(\d*)\s*", re.S).sub(replace_newlines, result)
|
||||
|
||||
#print(result)
|
||||
#sys.exit(0)
|
||||
#print("expand_whitespace_markers", text, result)
|
||||
return result
|
||||
|
||||
|
||||
def handle_dashes(html_text, K=None):
|
||||
def replace(match):
|
||||
tag_start, text, tag_end = match.groups()
|
||||
text = re.sub("__MDASH__", "---", text)
|
||||
text = re.sub("__NDASH__", "--", text)
|
||||
result = f"{tag_start}{text}{tag_end}"
|
||||
return result
|
||||
result = html_text
|
||||
result = re.compile('(<span class="monospace">)(.*?)(</span>)', re.S).sub(replace, result)
|
||||
result = re.sub("__MDASH__", "—", result);
|
||||
result = re.sub("__NDASH__", "–", result);
|
||||
return result
|
||||
|
||||
def add_tex_caption_numbers(tex_text, K=None):
|
||||
chapter_pat = re.compile(r"\\section\\{")
|
||||
index = {}
|
||||
result = ""
|
||||
chapter_number = 1
|
||||
for line in tex_text.split("\n"):
|
||||
if chapter_pat.search(line):
|
||||
chapter_number += 1
|
||||
reset_indices(index)
|
||||
if kutil.caption_delimiter() in line:
|
||||
before, caption_type, caption, after = line.split(kutil.caption_delimiter())
|
||||
if index.get(caption_type) is None:
|
||||
index[caption_type] = 1
|
||||
n = index[caption_type]
|
||||
number = f"{chapter_number}.{n}" if chapter_number != 0 else n
|
||||
line = f"{before}{caption_type} {number} {caption}{after}"
|
||||
index[caption_type] += 1
|
||||
result += line + "\n"
|
||||
return result
|
||||
|
||||
|
||||
def remove_redundant_vspace(tex_text, K=None):
|
||||
#print("remove_redundant_vspace")
|
||||
rgx = re.compile(r"(\\vspace\*\{-[^}]+\})\s*\\vspace\*\{-[^}]+\}", re.S)
|
||||
return rgx.sub(r"\1", tex_text)
|
||||
|
||||
def restore_backslash(tex_text, K=None):
|
||||
#print("restore backslash")
|
||||
#return re.compile(r"\^/").sub(r"\\", tex_text)
|
||||
return re.sub("\6", "", tex_text)
|
||||
|
||||
#--------------------------------------------------------------------------------
|
||||
|
||||
def levels_to_section(levels):
|
||||
result = ".".join([str(e) for e in levels])
|
||||
result = re.sub(r"\.0", "", result)
|
||||
return result
|
||||
|
||||
def chapter_title_span():
|
||||
return '<span class="chapter_title">'
|
||||
|
||||
def add_html_section_numbers(html_text, K=None, start=0):
|
||||
depth = 6
|
||||
levels = [start-1] + ([0] * (depth-1))
|
||||
section_pat = re.compile(r"(.*?)<h(\d)(.*?)>(.*?)</h\2>")
|
||||
id_pat = re.compile(r'.*?id="([-\w]+)".*', re.S)
|
||||
result = ""
|
||||
id_number = 1
|
||||
id_map = []
|
||||
for line in html_text.split('\n'):
|
||||
match = section_pat.match(line)
|
||||
if match:
|
||||
pre, level, attr, text = match.groups()
|
||||
id_match = id_pat.fullmatch(attr)
|
||||
if id_match:
|
||||
id = id_match.group(1)
|
||||
else:
|
||||
id = f"id{id_number}"
|
||||
attr = " " + f'id="{id}" {attr}'.strip()
|
||||
id_number += 1
|
||||
level = int(level)
|
||||
levels[level-1] = levels[level-1] + 1
|
||||
for i in range(level, depth):
|
||||
levels[i] = 0
|
||||
section = levels_to_section(levels)
|
||||
line = f'{pre}<h{level}{attr}>{chapter_title_span()}{section}  </span>{text}</h{level}>'
|
||||
id_map.append([level, section, id, text])
|
||||
result += line + "\n"
|
||||
[print(e) for e in id_map]
|
||||
return result, id_map
|
||||
|
||||
def make_html_table_of_contents(titles):
|
||||
result = '<div id="_toc" class="toc-title">Table of contents</div>\n'
|
||||
for level, section, id, text in titles:
|
||||
result += f'<div class="level{level}"><a href="#{id}" class="level">{section} {text}</a></div>\n'
|
||||
return result
|
||||
|
||||
def process_html_sections(html_text, K=None, start=1):
|
||||
html, id_map = add_html_section_numbers(html_text)
|
||||
toc = make_html_table_of_contents(id_map)
|
||||
result = re.sub(r'(<div id="middle">)', rf"\1\n{toc}", html)
|
||||
return result
|
||||
|
||||
|
||||
def reset_indices(indices):
|
||||
for key in indices.keys():
|
||||
indices[key] = 1
|
||||
|
||||
def add_html_caption_numbers(html_text, K=None):
|
||||
chapter_pat = re.compile(rf"{chapter_title_span()}(\d+)</span>")
|
||||
index = {}
|
||||
result = ""
|
||||
chapter_number = ""
|
||||
for line in html_text.split("\n"):
|
||||
if '<span' in line:
|
||||
match = chapter_pat.search(line)
|
||||
print(match.groups())
|
||||
if match and match.group(1) != "0":
|
||||
chapter_number = f"{match.group(1)}."
|
||||
reset_indices(index)
|
||||
if kutil.caption_delimiter() in line:
|
||||
before, caption_type, caption, after = line.split(kutil.caption_delimiter())
|
||||
if index.get(caption_type) is None:
|
||||
index[caption_type] = 1
|
||||
n = index[caption_type]
|
||||
line = f"{before}{caption_type} {chapter_number}{n} {caption}{after}"
|
||||
index[caption_type] += 1
|
||||
result += line + "\n"
|
||||
return result
|
||||
|
||||
|
||||
|
||||
def escape_pre_angle_brackets(code, K=None):
|
||||
def replace(match):
|
||||
code = re.sub(r"<", "<", match.group(1))
|
||||
return f"<pre>{code}</pre>"
|
||||
pat = re.compile("<pre>(.*?)</pre>", re.S)
|
||||
return pat.sub(replace, code)
|
||||
|
||||
def block_elements():
|
||||
return """
|
||||
address article aside blockquote details dialog dd div dl dt fieldset
|
||||
figcaption figure footer form h0 h1 h2 h3 h4 h5 h6 header hgroup hr li
|
||||
main nav ol p section table ul td tr pre
|
||||
""".strip().split()
|
||||
|
||||
|
||||
def not_a_block(par):
|
||||
if par.strip()[0] != "<":
|
||||
return True
|
||||
pat = re.compile(r'^</?([^\s>]+).*?>', re.S|re.M)
|
||||
match = pat.match(par.strip())
|
||||
return match.group(1) not in block_elements()
|
||||
|
||||
def make_paragraphs(html_text, K):
|
||||
def replace(match):
|
||||
before, content, after = match.groups()
|
||||
result = ""
|
||||
for par in [e.strip() for e in
|
||||
re.compile(r"\n *(\n *)+", re.S).split(content)]:
|
||||
if par.strip() and not_a_block(par):
|
||||
#jpar = "\n".join(textwrap.wrap(par, width=80))
|
||||
#result += f"\n<p>\n{jpar}\n</p>\n"
|
||||
jpar = "\n".join(textwrap.wrap(f"<p>{par}</p>", width=80))
|
||||
result += f"\n{jpar}\n"
|
||||
else:
|
||||
result += f"\n{par}\n"
|
||||
result = re.compile(r"\s*</(\w+)>\n<\1>").sub(r"\n</\1>\n\n<\1>", result)
|
||||
return f"{before}{result}{after}"
|
||||
pat = re.compile(r'(.*?<div id="middle">)(.*?)(</div>\s*<div id="bottom">.*)', re.S)
|
||||
match = pat.match(html_text)
|
||||
result = pat.sub(replace, html_text)
|
||||
result = re.compile(r"</td>\s+<td>", re.S).sub("</td>\n<td>", result)
|
||||
result = re.compile(r"\n *(\n *)+", re.S).sub("\n\n", result)
|
||||
return result
|
||||
|
||||
|
||||
def indent_html(filename, K):
|
||||
#print('indent_html')
|
||||
html_util.html_indent(filename)
|
||||
|
||||
def copy_html_resources(filename, K):
|
||||
output_dir = os.path.dirname(K._output_filename) or "."
|
||||
css_dir = f"{output_dir}/{html_util.css_reldir()}"
|
||||
kutil.make_dir_if_necessary(css_dir, delete_contents=True)
|
||||
for basename, source in kutil.sks_files_of_type("css"):
|
||||
command = f"cp {source} {css_dir}/{basename}.css"
|
||||
os.system(command)
|
||||
|
||||
|
||||
# def latex_escapes(latex_text):
|
||||
# return latex_text
|
||||
# result = latex_text
|
||||
# result = re.sub('#', '\\#', result)
|
||||
# result = re.sub('\^', '\\^', result)
|
||||
# result += "LATEX"
|
||||
# return result
|
||||
|
||||
# def restore_backslash(latex_text, K): # ?
|
||||
# return latex_text
|
||||
# print(latex_text)
|
||||
# print("restore_backslash")
|
||||
# result = latex_text
|
||||
# result = re.sub('\b', r'\\b', result)
|
||||
# result = re.sub(r'\\t', r'\\b', result)
|
||||
# return result
|
||||
|
||||
|
||||
def make_pdf_from_tex(filename, K, twice=True):
|
||||
#debug_mode = int(K.K_verbose_level) > 1
|
||||
verbose_level = int(K.K_verbose_level)
|
||||
pathname = os.path.abspath(filename)
|
||||
dirname, filename = os.path.split(pathname)
|
||||
basename, ext = os.path.splitext(filename)
|
||||
command = f"mv {pathname} {dirname}/{basename}.tex"
|
||||
log_file = f"{dirname}/{basename}.log"
|
||||
pdf_file = f"{dirname}/{basename}.pdf"
|
||||
result = os.system(command)
|
||||
#debug_mode = True
|
||||
if verbose_level == 3:
|
||||
remove_log = ''
|
||||
else:
|
||||
dbg_log = "/dev/null"
|
||||
remove_log = ' >{} 2>&1'.format(dbg_log)
|
||||
|
||||
env_var = "KLAMMERTEXT_TEXLIVE_BIN"
|
||||
texbin = os.environ.get(env_var)
|
||||
if texbin is None:
|
||||
msg = f"The directory of TeX Live commands must be defined by ${env_var}"
|
||||
raise Exception(msg)
|
||||
|
||||
latex_command = f"{texbin}/xelatex"
|
||||
#latex_command = f"{texbin}/pdflatex"
|
||||
|
||||
if not os.path.exists(latex_command):
|
||||
raise Exception(f"LaTeX command not found: {latex_command}")
|
||||
|
||||
flags = "--halt-on-error"
|
||||
#flags = "-file-line-error --shell-escape -halt-on-error -interaction nonstopmode -output-directory"
|
||||
|
||||
env = "export max_print_line=1000 ; export TEXINPUTS=${KLAMMERTEXT_HOME}/sks//: ;"
|
||||
command = f"{env} cd {dirname} ; {latex_command} {flags} {basename}.tex {remove_log}"
|
||||
#print(command)
|
||||
result = os.system(command)
|
||||
|
||||
if result != 0:
|
||||
os.system(f"tail -n 20 {log_file}")
|
||||
msg = f"Error in LaTeX processing. See file {os.path.relpath(log_file)}"
|
||||
#print(f"Error in LaTeX processing. See file {os.path.relpath(log_file)}")
|
||||
raise Exception(msg)
|
||||
|
||||
# No error; is repetition necessary for links, etc? Check log for this.
|
||||
# result = os.system(command)
|
||||
# print(f"Wrote {os.path.relpath(pdf_file)}")
|
||||
if twice:
|
||||
result = os.system(command)
|
||||
|
||||
aux_files = 'aux out toc'.split()
|
||||
if verbose_level < 2:
|
||||
aux_files += 'tex log'.split()
|
||||
for unused_ext in aux_files:
|
||||
os.system('rm -rf {}/{}.{}'.format(dirname, basename, unused_ext))
|
||||
|
||||
import re
|
||||
import textwrap
|
||||
|
||||
# Emitted by @vspace.txt (sks/block/block.k), one marker per line of space.
|
||||
# An @eval result consisting only of whitespace is trimmed away by the
|
||||
# Klammermachine, so vertical space must travel as markers and become
|
||||
# newlines here, after blank-line runs have been normalized.
|
||||
vspace_marker = "__VSPACE__"
|
||||
|
||||
def justify_blocks(text, K=None):
|
||||
def txt_nl_marker():
|
||||
return "__KTEXTNEWLINE__"
|
||||
|
||||
def txt_vspace_marker():
|
||||
return "__KTEXTVSPACE__"
|
||||
|
||||
def txt_justify_blocks(text, K):
|
||||
delim = '__DIVIDE__'
|
||||
text = re.sub("\n\n+", delim, text)
|
||||
result = ""
|
||||
for par in text.split(delim):
|
||||
stripped = par.strip()
|
||||
n = stripped.count(vspace_marker)
|
||||
if n and stripped == vspace_marker * n:
|
||||
n = stripped.count(txt_vspace_marker())
|
||||
if n and stripped == txt_vspace_marker() * n:
|
||||
# A paragraph of only @vspace markers: n blank lines in addition
|
||||
# to the normal paragraph separation.
|
||||
result += "\n" * n
|
||||
continue
|
||||
if par and par[0] not in {' ', '['}:
|
||||
par = "\n".join(textwrap.wrap(par, width=80))
|
||||
result += par + "\n\n"
|
||||
return result.replace(vspace_marker, "\n")
|
||||
|
||||
|
||||
|
||||
if par and par[0] != ' ':
|
||||
label = re.match(r"~*\[\d+\] ", par)
|
||||
par = "\n".join(textwrap.wrap(
|
||||
par, width=int(K.Target_txt_width), break_long_words=False,
|
||||
subsequent_indent=" " * len(label.group(0)) if label else ""))
|
||||
# if par and par[0] != ' ':
|
||||
# par = "\n".join(textwrap.wrap(
|
||||
# par, width=int(K.Target_txt_width), break_long_words=False))
|
||||
result += par + "\n\n"
|
||||
result = re.sub(" *" + txt_vspace_marker() + " *", "\n", result)
|
||||
result = re.sub(r"\n? *" + txt_nl_marker() + r" *\n?", "\n", result)
|
||||
result = re.sub("~", " ", result)
|
||||
return result
|
||||
|
||||
def txt_footnote_marker():
|
||||
return "KTEXTFOOTNOTE"
|
||||
|
||||
def txt_format_footnotes(text, K):
|
||||
footnote_number = 0
|
||||
footnotes = []
|
||||
def replace(m):
|
||||
nonlocal footnote_number, footnotes
|
||||
footnote_number += 1
|
||||
footnotes.append(m.group(1))
|
||||
return f"[{footnote_number}]"
|
||||
start_mark = "__" + txt_footnote_marker()
|
||||
end_mark = txt_footnote_marker() + "__"
|
||||
footnote_rgx = re.compile(rf"\s*{start_mark}\s+(.*?)\s+{end_mark}", re.S)
|
||||
result = footnote_rgx.sub(replace, text).rstrip() + "\n\n"
|
||||
if footnotes:
|
||||
rule_count = int(round(float(K.Target_txt_width) * 0.4))
|
||||
result += "-" * rule_count + "\n\n"
|
||||
# Calculate maximum width for right-justified numbers:
|
||||
width = len(f"[{len(footnotes)}]")
|
||||
for n, footnote in enumerate(footnotes, 1):
|
||||
label = f"[{n}]"
|
||||
result += "~" * (width - len(label)) + label + " " + footnote + "\n\n"
|
||||
return result
|
||||
|
||||
def html_footnote_marker():
|
||||
return "kt-footnote"
|
||||
|
||||
def html_format_footnotes(text, K=None):
|
||||
fnum = 0
|
||||
footnotes = []
|
||||
def replace(m):
|
||||
nonlocal fnum, footnotes
|
||||
fnum += 1
|
||||
footnotes.append(m.group(1))
|
||||
href = f'href ="#_footnote_{fnum}"'
|
||||
id = f'id="_footnote_src_{fnum}"'
|
||||
return f'<a {id} {href}><sup class="footnote_in_text">{fnum}</sup></a>'
|
||||
#return f'<a href="#_footnote_{fnum}" id="_footnote_src_{fnum}"><sup>{fnum}</sup></a>'
|
||||
start_mark = f"<{html_footnote_marker()}>"
|
||||
end_mark = f"</{html_footnote_marker()}>"
|
||||
footnote_rgx = re.compile(rf"\s*{start_mark}\s*(.*?)\s*{end_mark}", re.S)
|
||||
result = footnote_rgx.sub(replace, text).rstrip() + "\n\n"
|
||||
if footnotes:
|
||||
result += '<hr class="footnote_rule"/>\n'
|
||||
for n, footnote in enumerate(footnotes, 1):
|
||||
result += f'<a href="#_footnote_src_{n}" id="_footnote_{n}"><sup>{n} </sup></a>{footnote}<br/>\n'
|
||||
# Temporarily add room for footnotes to move to the top of the window for the test:
|
||||
result += '<div style="height: 100lh"></div>\n'
|
||||
return result
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user