Initial commit: Klammertext source distribution
Curated source subset assembled by klammertext-dev's doc/make_dist.sh: the Klammermachine (mac), the Standard Klammer Set (sks), the commands (com), editor plugins and install guides (doc), a test subset (tst), and lib/bin placeholders. Builds with 'make -C com'. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
320
sks/target/phases.py
Normal file
320
sks/target/phases.py
Normal file
@@ -0,0 +1,320 @@
|
||||
import sys, os, re, textwrap, pprint
|
||||
|
||||
basedir = f'{os.environ["KLAMMERTEXT_HOME"]}/sks'
|
||||
sys.path = [f"{basedir}/kutil"] + sys.path
|
||||
sys.path = [f"{basedir}/target"] + sys.path
|
||||
|
||||
import kutil
|
||||
import html_util
|
||||
|
||||
def tex_to_pdf(text):
|
||||
print("in tex_to_pdf")
|
||||
|
||||
|
||||
|
||||
def expand_whitespace_markers(text, K=None):
|
||||
def count(count_match):
|
||||
count = int(count_match) if count_match else 1
|
||||
return count
|
||||
def replace_spaces(match):
|
||||
return " " * count(match.group(1))
|
||||
def replace_newlines(match):
|
||||
return "\n" * count(match.group(1))
|
||||
|
||||
result = text
|
||||
|
||||
# result = re.sub(r"\s*#-\s*", "", result)
|
||||
|
||||
# space_pat = re.compile(r" *#\+(\d*) *", re.S)
|
||||
# print("hits:", space_pat.findall(text))
|
||||
|
||||
# for p in space_pat.findall(text):
|
||||
# print("FOUND:", p)
|
||||
|
||||
|
||||
#
|
||||
#result = re.sub(" ", "SPACE", result)
|
||||
#result = re.sub("#/", "\n", result)
|
||||
#result = re.sub("X", " ", result)
|
||||
|
||||
result = re.compile(r" *#- *", re.S).sub("", result)
|
||||
result = re.compile(r" *#\+(\d*) *", re.S).sub(replace_spaces, result)
|
||||
result = re.compile(r"\s*\#\/(\d*)\s*", re.S).sub(replace_newlines, result)
|
||||
|
||||
#print(result)
|
||||
#sys.exit(0)
|
||||
#print("expand_whitespace_markers", text, result)
|
||||
return result
|
||||
|
||||
|
||||
def handle_dashes(html_text, K=None):
|
||||
def replace(match):
|
||||
tag_start, text, tag_end = match.groups()
|
||||
text = re.sub("__MDASH__", "---", text)
|
||||
text = re.sub("__NDASH__", "--", text)
|
||||
result = f"{tag_start}{text}{tag_end}"
|
||||
return result
|
||||
result = html_text
|
||||
result = re.compile('(<span class="monospace">)(.*?)(</span>)', re.S).sub(replace, result)
|
||||
result = re.sub("__MDASH__", "—", result);
|
||||
result = re.sub("__NDASH__", "–", result);
|
||||
return result
|
||||
|
||||
def add_tex_caption_numbers(tex_text, K=None):
|
||||
chapter_pat = re.compile(r"\\section\\{")
|
||||
index = {}
|
||||
result = ""
|
||||
chapter_number = 1
|
||||
for line in tex_text.split("\n"):
|
||||
if chapter_pat.search(line):
|
||||
chapter_number += 1
|
||||
reset_indices(index)
|
||||
if kutil.caption_delimiter() in line:
|
||||
before, caption_type, caption, after = line.split(kutil.caption_delimiter())
|
||||
if index.get(caption_type) is None:
|
||||
index[caption_type] = 1
|
||||
n = index[caption_type]
|
||||
number = f"{chapter_number}.{n}" if chapter_number != 0 else n
|
||||
line = f"{before}{caption_type} {number} {caption}{after}"
|
||||
index[caption_type] += 1
|
||||
result += line + "\n"
|
||||
return result
|
||||
|
||||
|
||||
def remove_redundant_vspace(tex_text, K=None):
|
||||
#print("remove_redundant_vspace")
|
||||
rgx = re.compile(r"(\\vspace\*\{-[^}]+\})\s*\\vspace\*\{-[^}]+\}", re.S)
|
||||
return rgx.sub(r"\1", tex_text)
|
||||
|
||||
def restore_backslash(tex_text, K=None):
|
||||
#print("restore backslash")
|
||||
#return re.compile(r"\^/").sub(r"\\", tex_text)
|
||||
return re.sub("\6", "", tex_text)
|
||||
|
||||
#--------------------------------------------------------------------------------
|
||||
|
||||
def levels_to_section(levels):
|
||||
result = ".".join([str(e) for e in levels])
|
||||
result = re.sub(r"\.0", "", result)
|
||||
return result
|
||||
|
||||
def chapter_title_span():
|
||||
return '<span class="chapter_title">'
|
||||
|
||||
def add_html_section_numbers(html_text, K=None, start=0):
|
||||
depth = 6
|
||||
levels = [start-1] + ([0] * (depth-1))
|
||||
section_pat = re.compile(r"(.*?)<h(\d)(.*?)>(.*?)</h\2>")
|
||||
id_pat = re.compile(r'.*?id="([-\w]+)".*', re.S)
|
||||
result = ""
|
||||
id_number = 1
|
||||
id_map = []
|
||||
for line in html_text.split('\n'):
|
||||
match = section_pat.match(line)
|
||||
if match:
|
||||
pre, level, attr, text = match.groups()
|
||||
id_match = id_pat.fullmatch(attr)
|
||||
if id_match:
|
||||
id = id_match.group(1)
|
||||
else:
|
||||
id = f"id{id_number}"
|
||||
attr = " " + f'id="{id}" {attr}'.strip()
|
||||
id_number += 1
|
||||
level = int(level)
|
||||
levels[level-1] = levels[level-1] + 1
|
||||
for i in range(level, depth):
|
||||
levels[i] = 0
|
||||
section = levels_to_section(levels)
|
||||
line = f'{pre}<h{level}{attr}>{chapter_title_span()}{section}  </span>{text}</h{level}>'
|
||||
id_map.append([level, section, id, text])
|
||||
result += line + "\n"
|
||||
[print(e) for e in id_map]
|
||||
return result, id_map
|
||||
|
||||
def make_html_table_of_contents(titles):
|
||||
result = '<div id="_toc" class="toc-title">Table of contents</div>\n'
|
||||
for level, section, id, text in titles:
|
||||
result += f'<div class="level{level}"><a href="#{id}" class="level">{section} {text}</a></div>\n'
|
||||
return result
|
||||
|
||||
def process_html_sections(html_text, K=None, start=1):
|
||||
html, id_map = add_html_section_numbers(html_text)
|
||||
toc = make_html_table_of_contents(id_map)
|
||||
result = re.sub(r'(<div id="middle">)', rf"\1\n{toc}", html)
|
||||
return result
|
||||
|
||||
|
||||
def reset_indices(indices):
|
||||
for key in indices.keys():
|
||||
indices[key] = 1
|
||||
|
||||
def add_html_caption_numbers(html_text, K=None):
|
||||
chapter_pat = re.compile(rf"{chapter_title_span()}(\d+)</span>")
|
||||
index = {}
|
||||
result = ""
|
||||
chapter_number = ""
|
||||
for line in html_text.split("\n"):
|
||||
if '<span' in line:
|
||||
match = chapter_pat.search(line)
|
||||
print(match.groups())
|
||||
if match and match.group(1) != "0":
|
||||
chapter_number = f"{match.group(1)}."
|
||||
reset_indices(index)
|
||||
if kutil.caption_delimiter() in line:
|
||||
before, caption_type, caption, after = line.split(kutil.caption_delimiter())
|
||||
if index.get(caption_type) is None:
|
||||
index[caption_type] = 1
|
||||
n = index[caption_type]
|
||||
line = f"{before}{caption_type} {chapter_number}{n} {caption}{after}"
|
||||
index[caption_type] += 1
|
||||
result += line + "\n"
|
||||
return result
|
||||
|
||||
|
||||
|
||||
def escape_pre_angle_brackets(code, K=None):
|
||||
def replace(match):
|
||||
code = re.sub(r"<", "<", match.group(1))
|
||||
return f"<pre>{code}</pre>"
|
||||
pat = re.compile("<pre>(.*?)</pre>", re.S)
|
||||
return pat.sub(replace, code)
|
||||
|
||||
def block_elements():
|
||||
return """
|
||||
address article aside blockquote details dialog dd div dl dt fieldset
|
||||
figcaption figure footer form h0 h1 h2 h3 h4 h5 h6 header hgroup hr li
|
||||
main nav ol p section table ul td tr pre
|
||||
""".strip().split()
|
||||
|
||||
|
||||
def not_a_block(par):
|
||||
if par.strip()[0] != "<":
|
||||
return True
|
||||
pat = re.compile(r'^</?([^\s>]+).*?>', re.S|re.M)
|
||||
match = pat.match(par.strip())
|
||||
return match.group(1) not in block_elements()
|
||||
|
||||
def make_paragraphs(html_text, K):
|
||||
def replace(match):
|
||||
before, content, after = match.groups()
|
||||
result = ""
|
||||
for par in [e.strip() for e in
|
||||
re.compile(r"\n *(\n *)+", re.S).split(content)]:
|
||||
if par.strip() and not_a_block(par):
|
||||
#jpar = "\n".join(textwrap.wrap(par, width=80))
|
||||
#result += f"\n<p>\n{jpar}\n</p>\n"
|
||||
jpar = "\n".join(textwrap.wrap(f"<p>{par}</p>", width=80))
|
||||
result += f"\n{jpar}\n"
|
||||
else:
|
||||
result += f"\n{par}\n"
|
||||
result = re.compile(r"\s*</(\w+)>\n<\1>").sub(r"\n</\1>\n\n<\1>", result)
|
||||
return f"{before}{result}{after}"
|
||||
pat = re.compile(r'(.*?<div id="middle">)(.*?)(</div>\s*<div id="bottom">.*)', re.S)
|
||||
match = pat.match(html_text)
|
||||
result = pat.sub(replace, html_text)
|
||||
result = re.compile(r"</td>\s+<td>", re.S).sub("</td>\n<td>", result)
|
||||
result = re.compile(r"\n *(\n *)+", re.S).sub("\n\n", result)
|
||||
return result
|
||||
|
||||
|
||||
def indent_html(filename, K):
|
||||
#print('indent_html')
|
||||
html_util.html_indent(filename)
|
||||
|
||||
def copy_html_resources(filename, K):
|
||||
output_dir = os.path.dirname(K._output_filename) or "."
|
||||
css_dir = f"{output_dir}/{html_util.css_reldir()}"
|
||||
kutil.make_dir_if_necessary(css_dir, delete_contents=True)
|
||||
for basename, source in kutil.sks_files_of_type("css"):
|
||||
command = f"cp {source} {css_dir}/{basename}.css"
|
||||
os.system(command)
|
||||
|
||||
|
||||
# def latex_escapes(latex_text):
|
||||
# return latex_text
|
||||
# result = latex_text
|
||||
# result = re.sub('#', '\\#', result)
|
||||
# result = re.sub('\^', '\\^', result)
|
||||
# result += "LATEX"
|
||||
# return result
|
||||
|
||||
# def restore_backslash(latex_text, K): # ?
|
||||
# return latex_text
|
||||
# print(latex_text)
|
||||
# print("restore_backslash")
|
||||
# result = latex_text
|
||||
# result = re.sub('\b', r'\\b', result)
|
||||
# result = re.sub(r'\\t', r'\\b', result)
|
||||
# return result
|
||||
|
||||
|
||||
def make_pdf_from_tex(filename, K, twice=True):
|
||||
#debug_mode = int(K.K_verbose_level) > 1
|
||||
verbose_level = int(K.K_verbose_level)
|
||||
pathname = os.path.abspath(filename)
|
||||
dirname, filename = os.path.split(pathname)
|
||||
basename, ext = os.path.splitext(filename)
|
||||
command = f"mv {pathname} {dirname}/{basename}.tex"
|
||||
log_file = f"{dirname}/{basename}.log"
|
||||
pdf_file = f"{dirname}/{basename}.pdf"
|
||||
result = os.system(command)
|
||||
#debug_mode = True
|
||||
if verbose_level == 3:
|
||||
remove_log = ''
|
||||
else:
|
||||
dbg_log = "/dev/null"
|
||||
remove_log = ' >{} 2>&1'.format(dbg_log)
|
||||
|
||||
env_var = "KLAMMERTEXT_TEXLIVE_BIN"
|
||||
texbin = os.environ.get(env_var)
|
||||
if texbin is None:
|
||||
msg = f"The directory of TeX Live commands must be defined by ${env_var}"
|
||||
raise Exception(msg)
|
||||
|
||||
latex_command = f"{texbin}/xelatex"
|
||||
#latex_command = f"{texbin}/pdflatex"
|
||||
|
||||
if not os.path.exists(latex_command):
|
||||
raise Exception(f"LaTeX command not found: {latex_command}")
|
||||
|
||||
flags = "--halt-on-error"
|
||||
#flags = "-file-line-error --shell-escape -halt-on-error -interaction nonstopmode -output-directory"
|
||||
|
||||
env = "export max_print_line=1000 ; export TEXINPUTS=${KLAMMERTEXT_HOME}/sks//: ;"
|
||||
command = f"{env} cd {dirname} ; {latex_command} {flags} {basename}.tex {remove_log}"
|
||||
#print(command)
|
||||
result = os.system(command)
|
||||
|
||||
if result != 0:
|
||||
os.system(f"tail -n 20 {log_file}")
|
||||
msg = f"Error in LaTeX processing. See file {os.path.relpath(log_file)}"
|
||||
#print(f"Error in LaTeX processing. See file {os.path.relpath(log_file)}")
|
||||
raise Exception(msg)
|
||||
|
||||
# No error; is repetition necessary for links, etc? Check log for this.
|
||||
# result = os.system(command)
|
||||
# print(f"Wrote {os.path.relpath(pdf_file)}")
|
||||
if twice:
|
||||
result = os.system(command)
|
||||
|
||||
aux_files = 'aux out toc'.split()
|
||||
if verbose_level < 2:
|
||||
aux_files += 'tex log'.split()
|
||||
for unused_ext in aux_files:
|
||||
os.system('rm -rf {}/{}.{}'.format(dirname, basename, unused_ext))
|
||||
|
||||
|
||||
def justify_blocks(text, K=None):
|
||||
rgx = re.compile("\n\n+", re.S)
|
||||
delim = '__DIVIDE__'
|
||||
text = rgx.sub(delim, text)
|
||||
result = ""
|
||||
for par in text.split(delim):
|
||||
#print(par)
|
||||
if par[0] not in {' ', '['}:
|
||||
par = "\n".join(textwrap.wrap(par, width=80))
|
||||
result += par + "\n\n"
|
||||
return result
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user