A snapshot of the development tree. The substantial changes since the last one:
COMMAND OUTPUT POLICY. The three commands display text in exactly three cases,
and each owns a stream: LOGGING under "-v" greater than 0 and an ERROR before
termination go to STDERR; OUTPUT THE USER ASKED FOR goes to STDOUT. For ktext
that output is a document, so "ktext doc.kt -d | ..." is now safe -- logging
used to share the stream and land inside the document. A bare command prints
its usage and succeeds rather than failing. Colour is emitted only to a
terminal, per stream, and NO_COLOR is honoured.
"-v 1" reports every decision whose outcome you could not have read off your own
input: the klammerset that was loaded and from which file, a font's directory, a
":files" name's file, how "-o" was expanded. Higher levels are the trace.
The commands no longer warn and continue: an anomaly is an error, described with
its location. Two exceptions remain, each for a stated reason -- a condition
that is expected and temporary by design, and a judgment that is a heuristic
rather than exact.
@cond IS NOW A TRUE SPECIAL FORM, resolved at APPLICATION time rather than when
the file is read. Two consequences for a writer:
* a state variable reaches the predicate. "@@@state Flag :value true @@@
@cond *Flag* | T | F @" renders "T"; it used to see the literal "*Flag*" and
silently take the false branch. The document now behaves like a klammer
body, whose arguments are bound before its conditionals are decided.
* nothing in a discarded branch happens -- it is not read, not evaluated, not
expanded. An @eval in the branch not taken used to run anyway.
Its predicate relation is total and strict: true, True, 1; false, False, 0, and
empty; anything else is an error at the @cond rather than silently false.
@eval REACHING OUTSIDE. ":shell" and ":haskell" now keep the command's standard
error out of the document (it appears under "-v 1") and treat a nonzero exit as
an error naming what the command reported. A command that exits nonzero on
purpose -- "grep" finding no match -- says so with "|| true".
KLAMMER SETS. Several combine: "--klammersets a b c" loads all three in the
order given, sharing one namespace, with the definition modes deciding
collisions. "none" means none and may not be combined with other symbols. A
klammerset with symbol X is declared in a file X/X.k, which is what lets two
sets require the same third set without loading it twice.
TESTS. Four new suites: the kdiag command's interface, the @eval primitive's
contract with the outside world, and verbosity at both tiers. Three suites
that could not run on macOS at all now do.
Assembled from dev commit 6c8ee6c22fca.
359 lines
15 KiB
Python
359 lines
15 KiB
Python
if __name__ == "__main__":
|
|
import sys
|
|
sys.path.append("../kutil")
|
|
sys.path.append("../target")
|
|
|
|
import sys
|
|
import re
|
|
import klammer_base
|
|
import kutil
|
|
import html_util
|
|
from html_util import E
|
|
import latex_util as L
|
|
import pprint
|
|
import phases
|
|
|
|
def escape_newlines(s):
|
|
# A newline becomes the marker the html paragraph pass turns back into a
|
|
# line break, so a verbatim source file keeps its lines. Used by Source.
|
|
return re.sub("\n", " ___NL___ ", s)
|
|
|
|
def undash(s):
|
|
# Verbatim text must show the hyphens the writer typed: the target's
|
|
# "--"/"---" transforms have already run, so put them back. Used by
|
|
# Code_fragment.
|
|
result = re.sub("__MDASH__", "---", s)
|
|
return re.sub("__NDASH__", "--", result)
|
|
|
|
def is_comment(s):
|
|
return s.strip().startswith("//")
|
|
|
|
def split_blocks(s):
|
|
blocks = []
|
|
in_code = True
|
|
block = ""
|
|
for line in s.rstrip().split("\n"):
|
|
if is_comment(line):
|
|
if in_code:
|
|
blocks.append(block)
|
|
block = line + "\n"
|
|
in_code = False
|
|
else:
|
|
block += line + "\n"
|
|
else:
|
|
if not in_code:
|
|
blocks.append(block)
|
|
block = line + "\n"
|
|
in_code = True
|
|
else:
|
|
block += line + "\n"
|
|
if block:
|
|
blocks.append(block)
|
|
return blocks
|
|
|
|
def get_lines(comment, code, n):
|
|
lines = code.split("\n")
|
|
if len(lines) < n:
|
|
raise Exception(
|
|
f"Error in @code block comment:\n{comment}\n{code}\nNeed {n} lines, but there are only {len(lines)}")
|
|
if len(lines) == n:
|
|
return (code, "")
|
|
else:
|
|
return ("\n".join(lines[:n]), "\n".join(lines[n:]))
|
|
|
|
def parse_blocks(blocks):
|
|
box_comment_rgx = re.compile("\\s*//(\\d+)\\s+.*", re.S)
|
|
i = 0
|
|
block_pairs = []
|
|
while i < len(blocks):
|
|
match = box_comment_rgx.match(blocks[i])
|
|
if match:
|
|
boxed_code, rest_code = get_lines(blocks[i], blocks[i+1], int(match.group(1)))
|
|
block_pairs.append([boxed_code, blocks[i]])
|
|
block_pairs.append([rest_code, ""])
|
|
i += 2
|
|
else:
|
|
block_pairs.append([blocks[i], ""])
|
|
i += 1
|
|
return block_pairs
|
|
|
|
|
|
# Characters that are special to LATEX but not to Klammertext. # and ^ are
|
|
# deliberately ABSENT: they are Klammertext specials and are handled by
|
|
# quote_specials() below, not here -- see tex_line().
|
|
latex_escapes = {
|
|
"\\": "\\textbackslash{}", "{": "\\{", "}": "\\}", "%": "\\%",
|
|
"$": "\\$", "&": "\\&", "_": "\\_", "~": "\\textasciitilde{}",
|
|
"<": "\\textless{}", ">": "\\textgreater{}",
|
|
}
|
|
latex_escape_rgx = re.compile("[" + re.escape("".join(latex_escapes)) + "]")
|
|
|
|
def escape_latex(s):
|
|
r"""Escape LaTeX special characters in code text.
|
|
|
|
ONE pass, not a sequence of str.replace() calls: a sequence corrupts
|
|
the replacements it has already made. Replacing \ first yields
|
|
\textbackslash{}, whose braces the later { and } replacements then
|
|
escape in turn -- so a C++ "\n" typesets as \{}n. A single regex pass
|
|
over the original string cannot revisit its own output.
|
|
"""
|
|
return latex_escape_rgx.sub(lambda m: latex_escapes[m.group()], s)
|
|
|
|
def quote_specials(s):
|
|
r"""Quote the Klammertext special characters in code text.
|
|
|
|
An @eval result is RE-READ as Klammertext, so a special character that
|
|
reaches the result raw is interpreted again: a # in "#include" starts a
|
|
text removal and silently eats the rest of the line. Quoting is the
|
|
fix, and it must be quoting rather than LaTeX escaping -- \# would put
|
|
the # back and be eaten in turn.
|
|
|
|
The quoted forms resolve at final processing: the tex target declares
|
|
^# -> \# and ^^ -> \textasciicircum{} in its :escape list, and any
|
|
other quoted special decodes back to its own character. So the target,
|
|
not this code, decides what a # becomes in LaTeX.
|
|
|
|
^ is quoted FIRST, since the other quotings introduce ^ characters.
|
|
"""
|
|
for ch in "^@#|":
|
|
s = s.replace(ch, "^" + ch)
|
|
return s
|
|
|
|
def tex_line(line):
|
|
r"""One code line, ready to be a \klline argument.
|
|
|
|
Three steps in this order: LaTeX-escape the characters only LaTeX cares
|
|
about; quote the ones Klammertext would re-interpret (the escapes above
|
|
introduce none of them, so the two passes cannot interfere); then make
|
|
every space a ~ -- a non-breaking space, exactly one character wide in
|
|
a monospace font, which LaTeX will not collapse -- so the code's
|
|
indentation and internal alignment survive.
|
|
"""
|
|
return quote_specials(escape_latex(line)).replace(" ", "~")
|
|
|
|
def widest_line(lines):
|
|
r"""The line to measure the block's width with.
|
|
|
|
Measured on the RAW lines: in a monospace font character count is
|
|
exact, whereas the escaped text is longer than it typesets
|
|
(\textbackslash{} is eleven characters and one glyph). The chosen
|
|
line is escaped afterwards, for \settowidth.
|
|
"""
|
|
return max(lines, key=len) if lines else ""
|
|
|
|
def tex_block(code, comment):
|
|
r"""One block: its lines, shaded if the block is annotated, beside its
|
|
comment. See the "Annotated code listings" section of sty/code.sty,
|
|
which owns the layout; this only supplies the three arguments."""
|
|
lines = code.strip("\n").split("\n")
|
|
if not any(line.strip() for line in lines):
|
|
return ""
|
|
# Every line is a plain \klline; whether the BLOCK is shaded is decided
|
|
# by \klcodebox from the comment, because the shading is one box around
|
|
# the block (that is what gives it vertical padding).
|
|
body = "\\\\\n".join(f"\\klline{{{tex_line(line)}}}" for line in lines)
|
|
return ("\\klblock{" + tex_line(widest_line(lines)) + "}{%\n"
|
|
+ body + "}{" + comment + "}\n")
|
|
|
|
def comment_text(comment):
|
|
"""The prose of a block comment: the // and the line count removed."""
|
|
return re.sub(r"^\s*//\d*\s*", "", comment.strip(), flags=re.S).strip()
|
|
|
|
html_escapes = {"&": "&", "<": "<", ">": ">"}
|
|
html_escape_rgx = re.compile("[" + re.escape("".join(html_escapes)) + "]")
|
|
|
|
def html_line(text):
|
|
"""Code text as html: the markup characters escaped, then the
|
|
Klammertext specials quoted for the @eval read-back (same reason as
|
|
tex_line -- a raw # in "#include" would start a text removal). The
|
|
html entities introduce no Klammertext special, so the two passes
|
|
cannot interfere. No target :escape entries apply here, so each
|
|
quoted special decodes back to its own character."""
|
|
return quote_specials(html_escape_rgx.sub(
|
|
lambda m: html_escapes[m.group()], text))
|
|
|
|
def html_block(code, comment):
|
|
r"""One block: its lines beside its comment.
|
|
|
|
The layout is CSS (sks/code/css/code.css): .code_block is a flex row
|
|
with align-items center -- the same model as the LaTeX \parbox[c]
|
|
pair -- and .code_text is an inline-block with white-space: pre, so
|
|
its shrink-to-fit width IS the block's longest line and the shading
|
|
is one solid rectangle with no per-line work. Nothing is measured
|
|
here: unlike LaTeX, the browser does the layout.
|
|
"""
|
|
lines = code.strip("\n").split("\n")
|
|
if not any(line.strip() for line in lines):
|
|
return ""
|
|
shade = "code_border" if comment else "code_no_border"
|
|
result = f'<div class="code_text {shade}">{html_line(code.strip(chr(10)))}</div>'
|
|
if comment:
|
|
result += f'<div class="code_comment">{comment}</div>'
|
|
return f'<div class="code_block">{result}</div>\n'
|
|
|
|
|
|
class Code(klammer_base.Klammer_base):
|
|
id = 0
|
|
def __init__(self, K):
|
|
super().__init__(K)
|
|
self.text = phases.expand_whitespace_markers(self.text)
|
|
|
|
def annotated(self, pairs):
|
|
"""Does any block of this listing carry a comment?"""
|
|
return any(comment.strip() for _, comment in pairs)
|
|
|
|
def tex(self):
|
|
# No table: comments sit a fixed distance from their own block and
|
|
# deliberately do not align with each other, and each block's box is
|
|
# as wide as that block's longest line — so there is no column to
|
|
# align and nothing for a table to do. The layout is in
|
|
# sty/code.sty; see its "Annotated code listings" section.
|
|
#
|
|
# This is a RENDERER (final LaTeX, no klammers in the result), so
|
|
# the Klammermachine leaves it alone. It must never emit @code:
|
|
# a klammer that generates itself re-enters its own body with no
|
|
# base case, which the depth guard catches at 200 levels.
|
|
pairs = parse_blocks(split_blocks(self.text))
|
|
blocks = "".join(tex_block(code, comment_text(comment))
|
|
for code, comment in pairs)
|
|
listing = "\\begin{klcode}\n" + blocks + "\\end{klcode}\n"
|
|
captioned = bool(self.number or self.caption)
|
|
annotated = self.annotated(pairs)
|
|
inset = L.offset_length(self.offset)
|
|
|
|
# A caption does NOT by itself require a box, and boxing a listing
|
|
# costs page breaking -- a minipage cannot break, and listings are
|
|
# often long. Only two things genuinely need the shared box that
|
|
# add_caption builds:
|
|
#
|
|
# * a caption BESIDE the listing (:caption_side left or right),
|
|
# which has to know the listing's width;
|
|
# * an unannotated listing that must be centered or right-aligned,
|
|
# which has to be measured before it can be moved.
|
|
#
|
|
# Everything else is placed unboxed, and keeps breaking: the offset
|
|
# becomes \klshift/\klindent inside the environment, and the caption
|
|
# becomes a paragraph above or below, indented and width-matched to
|
|
# the listing. An annotated listing never needs measuring -- it
|
|
# spans the text column by construction.
|
|
beside = captioned and self.caption_side in ("left", "right")
|
|
must_measure = not annotated and (self.hpos != "left" or inset != "0pt")
|
|
if beside or must_measure:
|
|
return L.block(self.tex_boxed(listing, pairs, annotated,
|
|
captioned, inset))
|
|
return L.block(self.tex_unboxed(listing, annotated, captioned, inset))
|
|
|
|
def tex_boxed(self, listing, pairs, annotated, captioned, inset):
|
|
r"""The listing as a box: placed and captioned like a table or an
|
|
image, at the cost of not breaking across pages."""
|
|
measure = ""
|
|
if annotated:
|
|
# It spans the text column, so an offset narrows it rather than
|
|
# moving it; a centered element has no margin to be inset from.
|
|
width = ("\\linewidth" if self.hpos == "center"
|
|
else f"\\dimexpr\\linewidth-{inset}\\relax")
|
|
else:
|
|
# The \settowidth must stay OUTSIDE the minipage it sizes.
|
|
lines = [line for code, _ in pairs
|
|
for line in code.strip("\n").split("\n")]
|
|
measure = ("\\settowidth{\\kllistingwidth}{\\ttfamily "
|
|
+ tex_line(widest_line(lines)) + "}\n")
|
|
width = "\\kllistingwidth"
|
|
boxed = L.minipage(listing, width, vertical="t", center=False)
|
|
if captioned:
|
|
# add_caption attaches the caption to the box and hands the pair
|
|
# to caption_wrapper, so :pos and :offset move both together.
|
|
return measure + L.add_caption(
|
|
boxed, "Listing", self.number, self.caption, width,
|
|
hpos=self.hpos, side=self.caption_side,
|
|
font_symbol=self.caption_font,
|
|
font_size=self.caption_font_size, offset=self.offset)
|
|
return measure + L.caption_wrapper(boxed, self.hpos,
|
|
offset=self.offset)
|
|
|
|
def tex_unboxed(self, listing, annotated, captioned, inset):
|
|
r"""The listing in the running vertical list, so it can break across
|
|
pages. The offset is \leftskip plus a matching reduction of the
|
|
width \klblock computes its comment column from; the caption is a
|
|
paragraph on the same indent and width."""
|
|
shift = inset if self.hpos == "left" else "0pt"
|
|
# \hpos center cannot move a full-width listing, so it takes no
|
|
# offset -- the same rule the boxed path and caption_wrapper use.
|
|
indent = "0pt" if (annotated and self.hpos == "center") else inset
|
|
setup = (f"\\setlength{{\\klshift}}{{{shift}}}"
|
|
f"\\setlength{{\\klindent}}{{{indent}}}\n")
|
|
result = setup + listing
|
|
if captioned:
|
|
caption = L.make_caption_text(
|
|
self.number, "Listing", self.caption,
|
|
self.caption_font, self.caption_font_size)
|
|
# \nobreak: a caption must not be separated from its listing by
|
|
# a page break, even though the listing itself may break.
|
|
block = (f"\\noindent\\hspace*{{{shift}}}"
|
|
f"\\parbox[t]{{\\dimexpr\\linewidth-{indent}\\relax}}"
|
|
f"{{{caption}}}\\par")
|
|
if self.caption_side == "top":
|
|
result = block + "\\nobreak\n" + result
|
|
else:
|
|
result = result + "\\nobreak\n" + block + "\n"
|
|
return result
|
|
|
|
def html(self):
|
|
pairs = parse_blocks(split_blocks(self.text))
|
|
result = "".join(html_block(code, comment_text(comment))
|
|
for code, comment in pairs)
|
|
# The same two cases as tex_place, with the browser doing the work.
|
|
# An annotated listing's rows must stay full width so the comment
|
|
# column's flex: 1 has a remainder to take; an unannotated one gets
|
|
# width: fit-content, which is what lets the position container
|
|
# center or right-align it.
|
|
listing = E("div").cls("code_listing").body(result)
|
|
if not self.annotated(pairs):
|
|
listing.cls("code_listing_box")
|
|
# As in tex: the caption goes through the shared helper, so it sits
|
|
# in the same position container as the listing and moves with it.
|
|
if self.number or self.caption:
|
|
return html_util.add_caption(
|
|
listing, "Listing", self.number, self.caption,
|
|
self.caption_font, self.hpos, self.caption_side, True,
|
|
self.caption_font_size, offset=self.offset)
|
|
return html_util.hpos_container(listing, self.hpos, self.offset).str()
|
|
|
|
|
|
class Code_fragment(klammer_base.Klammer_base):
|
|
def __init__(self, K):
|
|
super().__init__(K)
|
|
|
|
def html(self):
|
|
#print(f"code: |{self.code_text}|")
|
|
result = self.code_text.strip()
|
|
result = undash(result)
|
|
#result = re.escape(result)
|
|
result = re.sub("<", "<", result)
|
|
result = re.sub(" ", " ", result)
|
|
#print(f"code: |{self.code_text}| -> |{result}|")
|
|
return f'<span class="code">{result}</span>'
|
|
|
|
def tex(self):
|
|
return f"{{\\tt {self.code_text.strip()}}}"
|
|
|
|
|
|
class Source(klammer_base.Klammer_base):
|
|
def __init__(self, K):
|
|
super().__init__(K)
|
|
with open(self.filename) as fp:
|
|
self.src = fp.read()
|
|
|
|
def tex(self):
|
|
result = self.src
|
|
# result = re.sub("#", "^#", result)
|
|
# result = re.sub("\\^", "\\^", result)
|
|
result = f"\\begin{{verbatim}}\n{result}\n\\end{{verbatim}}\n"
|
|
return result
|
|
|
|
def html(self):
|
|
result = escape_newlines(self.src.strip()) + "\n"
|
|
result = re.sub("@", "^@", result)
|
|
result = E("div").body(result).cls("code_text").str()
|
|
return result
|