diff options
Diffstat (limited to 'src/sisudoc')
| -rw-r--r-- | src/sisudoc/ocda/meta/rgx.d | 9 | ||||
| -rw-r--r-- | src/sisudoc/outputs/io_out/hub.d | 12 | ||||
| -rw-r--r-- | src/sisudoc/outputs/io_out/paths_output.d | 48 | ||||
| -rw-r--r-- | src/sisudoc/outputs/io_out/rgx.d | 13 | ||||
| -rw-r--r-- | src/sisudoc/outputs/io_out/typst.d | 846 | ||||
| -rw-r--r-- | src/sisudoc/spine.d | 24 |
6 files changed, 947 insertions, 5 deletions
diff --git a/src/sisudoc/ocda/meta/rgx.d b/src/sisudoc/ocda/meta/rgx.d index 125fb52..f418582 100644 --- a/src/sisudoc/ocda/meta/rgx.d +++ b/src/sisudoc/ocda/meta/rgx.d @@ -226,6 +226,10 @@ static template spineRgxIn() { static br_line = ctRegex!(`\s*┘\s*`, "mg"); static br_line_inline = ctRegex!(`\s*┙\s*`, "mg"); static br_line_spaced = ctRegex!(`\s*┚\s*`, "mg"); + /+ ↓ the line break the markup asked for with a trailing " \\\\". xhtml writers + carry their own copy of this in rgx_xhtml may be needed by other writers + +/ + static line_break = ctRegex!(` [\\]{2}`, "m"); /+ inline markup footnotes endnotes +/ static inline_notes_al = ctRegex!(`【(?:[*+]\s+|\s*)(.+?)】`, "mg"); static inline_notes_al_special = ctRegex!(`【(?:[*+]\s+)(.+?)】`, "mg"); // TODO remove match when special footnotes are implemented @@ -258,6 +262,11 @@ static template spineRgxIn() { static inline_link_clean = ctRegex!(`┤(?:.+?)├|[┥┝]`, "mg"); static inline_link_toc_to_backmatter = ctRegex!(`┤#(?P<link>endnotes|bibliography|bookindex|glossary|blurb)├`, "mg"); static find_bookindex_ocn_link_and_comma = ctRegex!(`[, ]*┥.+?┝┤#?\S+?├`, "mg"); + /+ ↓ an internal link target on its own, with or without the segment prefix a segmented output uses. Read where the + *target* is the question and the surrounding link may be malformed: gpl3`s copyright line nests one link inside + another, which the flat markers cannot express, so the target is there and the pair around it is not + +/ + static any_internal_target = ctRegex!(`┤(?:¤.+?\.fnSuffix)?#(?P<frag>[^├]+)├`, "mg"); static url = ctRegex!(`https?://`, "mg"); static uri = ctRegex!(`(?:https?|git)://`, "mg"); static uri_identify_components = ctRegex!(`(?P<type>(?:https?|git)://)(?P<path>\S+?/)(?P<file>[^/]+)$`, "mg"); diff --git a/src/sisudoc/outputs/io_out/hub.d b/src/sisudoc/outputs/io_out/hub.d index 254eaa5..1a0ed70 100644 --- a/src/sisudoc/outputs/io_out/hub.d +++ b/src/sisudoc/outputs/io_out/hub.d @@ -77,7 +77,7 @@ template outputHub() { mixin spinePo4aConfig; write(po4aConfig(doc.matters)); } - enum outTask { source_or_pod, sqlite, sqlite_multi, latex, odt, epub, html_scroll, html_seg, html_stuff, text, skel } + enum outTask { source_or_pod, sqlite, sqlite_multi, latex, typst, odt, epub, html_scroll, html_seg, html_stuff, text, skel } void Scheduled(D)(int sched, D doc) { auto msg = Msg!()(doc.matters); if (sched == outTask.source_or_pod) { @@ -133,6 +133,16 @@ template outputHub() { outputLaTeX!()(doc.abstraction, doc.matters); msg.vv("latex done"); } + if (sched == outTask.typst) { + msg.v("typst processing... (the .typ, and the pdf when --pdf)"); + import sisudoc.outputs.io_out.typst; + outputTypst!()(doc.abstraction, doc.matters); + if (doc.matters.opt.action.pdf) { + msg.v("typst compile to pdf... "); + outputTypstPdf!()(doc.matters); + } + msg.vv("typst done"); + } if (sched == outTask.text) { msg.v("text processing... "); import sisudoc.outputs.io_out.text; diff --git a/src/sisudoc/outputs/io_out/paths_output.d b/src/sisudoc/outputs/io_out/paths_output.d index 3d8323d..df0583c 100644 --- a/src/sisudoc/outputs/io_out/paths_output.d +++ b/src/sisudoc/outputs/io_out/paths_output.d @@ -669,6 +669,54 @@ template spinePathsSQLite() { } } +/+ ↓ where the typst output goes. + One .typ per document-language, because the paper is an input to the + compile rather than a thing compiled in, and the images beside it because a + .typ names them by path. The pdfs go where the latex path's pdfs go, under + the same names, so that a site built either way looks the same. ++/ +template spinePathsTypst() { + import std.conv; + auto spinePathsTypst(M)( + M doc_matters, + ) { + auto out_pth = spineOutPaths!()(doc_matters.output_path, doc_matters.src.language); + struct _PathsStruct { + /+ ↓ as latex: the .typ and the pdfs made from it carry the names the + .tex and its pdfs carried, so that a site built either way is the + same site and the html's pdf links resolve + +/ + string base_filename(string fn_src) { + return fn_src.baseName.stripExtension; + } + string base_pth() { + return (((out_pth.output_root).chainPath("typst")).asNormalizedPath).array; + } + string typst_file() { + // return ((base_pth.chainPath(doc_matters.src.doc_uid_out ~ ".typ")).asNormalizedPath).array; + return ((base_pth.chainPath(base_filename(doc_matters.src.filename) + ~ "." ~ doc_matters.src.language + ~ ".typ") + ).asNormalizedPath).array; + } + string images() { + return (((base_pth).chainPath("image")).asNormalizedPath).array; + } + string pdf_pth() { + return (((out_pth.output_root).chainPath("pdf")).asNormalizedPath).array; + } + string pdf_file_with_path(string paper_size_orientation) { + // return ((pdf_pth.chainPath(doc_matters.src.doc_uid_out + return ((pdf_pth.chainPath(base_filename(doc_matters.src.filename) + ~ "." ~ doc_matters.src.language + ~ "." ~ paper_size_orientation + ~ ".pdf") + ).asNormalizedPath).array; + } + } + return _PathsStruct(); + } +} template spinePathsText() { import std.conv; auto spinePathsText(M)( diff --git a/src/sisudoc/outputs/io_out/rgx.d b/src/sisudoc/outputs/io_out/rgx.d index 145b19b..bafd8b6 100644 --- a/src/sisudoc/outputs/io_out/rgx.d +++ b/src/sisudoc/outputs/io_out/rgx.d @@ -81,6 +81,10 @@ static template spineRgxOut() { static br_line = ctRegex!(`\s*┘\s*`, "mg"); static br_line_inline = ctRegex!(`\s*┙\s*`, "mg"); static br_line_spaced = ctRegex!(`\s*┚\s*`, "mg"); + /+ ↓ the line break the markup asked for with a trailing " \\\\". xhtml writers + carry their own copy of this in rgx_xhtml may be needed by other writers + +/ + static line_break = ctRegex!(` [\\]{2}`, "m"); /+ quotation marks +/ static quotes_open_and_close = ctRegex!(`[“”]`, "mg"); /+ inline markup footnotes endnotes +/ @@ -115,6 +119,11 @@ static template spineRgxOut() { static inline_link_clean = ctRegex!(`┤(?:.+?)├|[┥┝]`, "mg"); static inline_link_toc_to_backmatter = ctRegex!(`┤#(?P<link>endnotes|bibliography|bookindex|glossary|blurb)├`, "mg"); static find_bookindex_ocn_link_and_comma = ctRegex!(`[, ]*┥.+?┝┤#?\S+?├`, "mg"); + /+ ↓ an internal link target on its own, with or without the segment prefix a segmented output uses. Read where the + *target* is the question and the surrounding link may be malformed: gpl3`s copyright line nests one link inside + another, which the flat markers cannot express, so the target is there and the pair around it is not + +/ + static any_internal_target = ctRegex!(`┤(?:¤.+?\.fnSuffix)?#(?P<frag>[^├]+)├`, "mg"); static url = ctRegex!(`https?://`, "mg"); static uri = ctRegex!(`(?:https?|git)://`, "mg"); static uri_identify_components = ctRegex!(`(?P<type>(?:https?|git)://)(?P<path>\S+?/)(?P<file>[^/]+)$`, "mg"); @@ -159,5 +168,9 @@ static template spineRgxOut() { static grouped_para_bullet_indent_9 = ctRegex!(`^_9[*] `, "m"); static grouped_para_bullet_indent = ctRegex!(`^_(?P<indent>[1-9])[*] `, "m"); static grouped_para_indent_hang = ctRegex!(`^_(?P<hang>[0-9])_(?P<indent>[0-9])[ ]`, "m"); + /+ ↓ the indent as a number rather than one pattern per level: a writer that renders the indent as a measurement wants + the number, where one that writes a fixed run of spaces per level wants the nine above + +/ + static grouped_para_indent = ctRegex!(`^_(?P<indent>[1-9])[ ]`, "m"); } } diff --git a/src/sisudoc/outputs/io_out/typst.d b/src/sisudoc/outputs/io_out/typst.d new file mode 100644 index 0000000..e9f5f32 --- /dev/null +++ b/src/sisudoc/outputs/io_out/typst.d @@ -0,0 +1,846 @@ +/+ +- Name: SisuDoc Spine, Doc Reform [a part of] + - Description: documents, structuring, processing, publishing, search + - static content generator + + - Author: Ralph Amissah + [ralph.amissah@gmail.com] + + - Copyright: (C) 2015 (continuously updated, current 2026) Ralph Amissah, All Rights Reserved. + + - License: AGPL 3 or later: + + Spine (SiSU), a framework for document structuring, publishing and + search + + Copyright (C) Ralph Amissah + + This program is free software: you can redistribute it and/or modify it + under the terms of the GNU AFERO General Public License as published by the + Free Software Foundation, either version 3 of the License, or (at your + option) any later version. + + This program is distributed in the hope that it will be useful, but WITHOUT + ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or + FITNESS FOR A PARTICULAR PURPOSE. See the GNU General Public License for + more details. + + You should have received a copy of the GNU General Public License along with + this program. If not, see [https://www.gnu.org/licenses/]. + + If you have Internet connection, the latest version of the AGPL should be + available at these locations: + [https://www.fsf.org/licensing/licenses/agpl.html] + [https://www.gnu.org/licenses/agpl.html] + + - Spine (by Doc Reform, related to SiSU) uses standard: + - docReform markup syntax + - standard SiSU markup syntax with modified headers and minor modifications + - docReform object numbering + - standard SiSU object citation numbering & system + + - Homepages: + [https://www.sisudoc.org] + [https://www.doc-reform.org] + + - Git + [https://git.sisudoc.org/] + ++/ +module sisudoc.outputs.io_out.typst; +@safe: +/+ ↓ typst output: the document as typst markup, for a pdf. + . + The new default pdf path (beside the latex one). + --typst (--typ) writes .typ; + --pdf writes the .typ and compiles it. + --latex writes .tex (as before); + . + justification for typst as default: pdf is the format on which spine + depends most on an external toolchain, and various useful parts of an + object-centric pdf such as: object numbers in the margin; a link from + an object to its note and back; table of contents using object numbers + instead of page numbers - are a few lines each here. (Additionally, + measured against the latex build on the document sample set documents: + typst has a toolchain of 108 MB where texlive-combined-full is 6.6 GB, + and requires one compile per pdf rather than two, resulting in a tagged + pdf, and currently better CJK for Japanese than the latex). + . + One .typ per document-language, not one per paper size. The paper and the + orientation are read from `sys.inputs` with a default, so the same file + compiles to every paper spine is configured for. The objectcentric table + of contents means the document does not depend on pages. + . + it compiles silently, which means every internal link in it resolves, + because a link to a label that does not exist is a typst compile error. ++/ +template outputTypst() { + import sisudoc.outputs.io_out; + import sisudoc.outputs.io_out.rgx; + import sisudoc.outputs.io_out.paths_output; + import std.algorithm : max, min; + import std.array : appender, array, join, replace; + import std.array : asplit = split; + import std.conv : to; + import std.exception : ErrnoException; + import std.file; + import std.regex : matchAll, matchFirst, replaceAll, split; + import std.stdio; + import std.string : strip; + mixin spineRgxOut; + static auto rgx = RgxO(); + enum newline = "\n"; + enum newlines = "\n\n"; + /+ ↓ the markers that protect emitted typst from the escaping that follows. + . + escaping must come last: the abstraction's own markers are built from + characters typst reads as syntax, so escaping first destroys them. A + private-use pair, because they cannot occur in a document, and the + protection nests, so the escaper counts it rather than flagging it. + +/ + enum keep_open = ""; + enum keep_close = ""; + string keep(string typst_source) { + return keep_open ~ typst_source ~ keep_close; + } + string unprotect(string txt) { + return txt.replace(keep_open, "").replace(keep_close, ""); + } + /+ ↓ a label name typst will accept, from a name the markup chose. + Prefixed, because a label may not begin with a digit and every object + number does. Codepoints, not bytes, so that one character gives one + underscore. + +/ + string labelOf(string name) { + string _out = "t"; + foreach (dchar c; name) { + bool ok = (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z') + || (c >= '0' && c <= '9') || c == '_' || c == '-' || c == '.'; + if (ok) { _out ~= c; } else { _out ~= '_'; } + } + return _out; + } + /+ ↓ a typst string literal +/ + string str(string s) { + auto _out = appender!string; + _out ~= '"'; + foreach (dchar c; s) { + if (c == '"' || c == '\\') { _out ~= '\\'; _out ~= c; } + else if (c == '\n') { _out ~= "\\n"; } + else { _out ~= c; } + } + _out ~= '"'; + return _out.data; + } + /+ ↓ everything outside a protected run is escaped. + . + The rule is deliberately blunt: every character typst can read as syntax + is escaped wherever it occurs, not only where it would be read that way. + typst continues a function call across whatever follows it - + `#emph[a](b)` reads "(b)" as a second argument list and `#emph[a].b` as a + field access - so a "(" or a "." of the author's own, landing after any + call this writer emits, is a compile error. Escaping them everywhere + makes the set closed, and a closed set is checkable. + . + The quotation mark is not typst syntax in markup and is escaped anyway, + so that a bare one in the written source always delimits a string + literal; `guardBlockStarts` relies on that to leave a code object alone. + +/ + string escape(string txt) { + auto _out = appender!string; + int depth = 0; + dchar previous = '\0'; + foreach (dchar c; txt) { + if (c == keep_open.to!dchar) { depth++; continue; } + if (c == keep_close.to!dchar) { if (depth > 0) { depth--; } previous = '\0'; continue; } + if (depth > 0) { _out ~= c; continue; } + switch (c) { + case '\\': case '#': case '[': case ']': case '*': case '_': + case '$': case '@': case '<': case '>': case '`': case '~': + case '(': case ')': case '.': case '/': case '"': + _out ~= '\\'; _out ~= c; break; + case '-': + /+ ↓ a second dash would be an en dash, so it is escaped +/ + if (previous == '-') { _out ~= '\\'; } + _out ~= c; + break; + default: + _out ~= c; + break; + } + previous = c; + } + return _out.data; + } + /+ ↓ a "- ", "+ " or "= " that begins a line or a content block, which typst + reads as a list item, an enumeration item and a heading. Special only in + that position, so guarded there rather than escaped in every hyphen; and + not inside a string literal, where a guard would be a backslash in the + middle of somebody's program. + +/ + string guardBlockStarts(string s) { + auto _out = appender!string; + bool in_string = false; + for (size_t i = 0; i < s.length; i++) { + char c = s[i]; + _out ~= c; + if (in_string) { + if (c == '\\' && i + 1 < s.length) { _out ~= s[++i]; } + else if (c == '"') { in_string = false; } + continue; + } + if (c == '"') { in_string = true; continue; } + if ((c == '[' || c == '\n') && i + 2 < s.length + && (s[i + 1] == '-' || s[i + 1] == '+' || s[i + 1] == '=') + && s[i + 2] == ' ') { + _out ~= '\\'; + } + } + return _out.data; + } + /+ ↓ which sections, in which order. + The latex sequence, less the endnotes: a note is a footnote here, at the + foot of the page that refers to it, so a gathered endnotes section would + put every note in the document twice. + +/ + string[] typstSections(M)(M doc_matters) { + string[] _seq; + foreach (part; doc_matters.has.keys_seq.latex) { + if (part == "endnotes") { continue; } + _seq ~= part; + } + return _seq; + } + /+ ↓ every label this document will define. + A link to an undefined label does not compile, so the writer has to know + what it will define before it writes the first link. + +/ + bool[string] definedTargets(D,M)(const D doc_abstraction, M doc_matters) { + bool[string] _targets; + foreach (part; typstSections!()(doc_matters)) { + foreach (obj; doc_abstraction[part]) { + if (obj.metainfo.is_of_type == "comment") { continue; } + if (obj.metainfo.ocn != 0) { _targets[labelOf(obj.metainfo.ocn.to!string)] = true; } + if (obj.metainfo.identifier.length > 0) { + _targets[labelOf(obj.metainfo.identifier.to!string)] = true; + } + if (obj.tags.heading_lev_anchor_tag.length > 0) { + _targets[labelOf(obj.tags.heading_lev_anchor_tag.to!string)] = true; + } + foreach (at; obj.tags.anchor_tags) { + if (at.length > 0) { _targets[labelOf(at.to!string)] = true; } + } + foreach (m; obj.text.matchAll(rgx.inline_link_anchor)) { + _targets[labelOf(m["anchor"].to!string)] = true; + } + } + } + return _targets; + } + /+ ↓ the labels placed by `ocn`, on the number in the margin: always, and for + every numbered object. Claimed first, so that an anchor or an identifier + carrying the same name is dropped rather than defining a label twice, + which typst rejects. + +/ + bool[string] ocnLabels(D,M)(const D doc_abstraction, M doc_matters) { + bool[string] _placed; + foreach (part; typstSections!()(doc_matters)) { + foreach (obj; doc_abstraction[part]) { + if (obj.metainfo.is_of_type == "comment") { continue; } + if (obj.metainfo.ocn != 0) { _placed[labelOf(obj.metainfo.ocn.to!string)] = true; } + } + } + return _placed; + } + /+ ↓ the preamble: the page, the text, and the three functions the body leans + on. + . + The paper and the orientation come from `sys.inputs`, so one .typ is + every paper size spine is configured for: + typst compile --input paper=letter --input orientation=landscape + +/ + string typstHead(M)(M doc_matters) { + string _lang = doc_matters.src.language; + string _region; + auto _parts = _lang.replace("_", "-").asplit("-"); + _lang = _parts[0]; + if (_parts.length > 1) { _region = _parts[1]; } + string _title = doc_matters.conf_make_meta.meta.title_full; + auto _out = appender!string; + _out ~= "// " ~ doc_matters.generator_program.project_name.strip + ~ ": the object-centric document abstraction, written as typst\n"; + _out ~= "// Compile with: typst compile <this file>\n"; + _out ~= "// The paper is an input, so this one file is every paper size:\n"; + _out ~= "// typst compile --input paper=letter --input orientation=landscape ...\n\n"; + _out ~= "#let paper = sys.inputs.at(\"paper\", default: \"a4\")\n"; + _out ~= "#let landscape = sys.inputs.at(\"orientation\", default: \"portrait\")" + ~ " == \"landscape\"\n\n"; + /+ ↓ the pdf's own metadata, from the document's own. The date is the + document's and never the clock's: a document typeset twice is one + document. + +/ + _out ~= "#set document(\n title: " ~ str(_title) ~ ",\n"; + if (doc_matters.conf_make_meta.meta.creator_author.length > 0) { + _out ~= " author: " ~ str(doc_matters.conf_make_meta.meta.creator_author) ~ ",\n"; + } + _out ~= " date: none,\n)\n"; + /+ ↓ the left margin is wide because the object numbers live in it +/ + _out ~= "#set page(\n paper: paper,\n flipped: landscape,\n" + ~ " margin: (left: 3.4cm, right: 2cm, top: 2.2cm, bottom: 2.2cm),\n" + ~ " numbering: \"1\",\n)\n"; + _out ~= "#set text(\n font: (\"DejaVu Serif\", \"Noto Sans CJK JP\"),\n" + ~ " size: 10pt,\n lang: " ~ str(_lang); + if (_region.length > 0) { _out ~= ",\n region: " ~ str(_region); } + _out ~= ",\n)\n"; + _out ~= "#set par(justify: true, leading: 0.65em)\n"; + /+ ↓ the author's quotation marks and dashes, not typst's. A document is + reproduced, not restyled + +/ + _out ~= "#set smartquote(enabled: false)\n"; + _out ~= "#set heading(numbering: none)\n"; + _out ~= "#show heading: it => block(above: 1.4em, below: 0.7em, it)\n"; + _out ~= "#show raw.where(block: true): it => block(\n" + ~ " fill: luma(245), inset: 6pt, radius: 2pt, width: 100%, it)\n\n"; + /+ ↓ the object number, in the margin, beside the object's first line, and + carrying the label every citation of the object points at. + . + Typst gets this right being one `place` at a coordinate inside the + object's own block. (this can be a problem for the latex \ocn + which is a \marginpar with a \hypertarget, and can fail inside a + float or a footnote and migrate to the next page when it will not + fit.) + +/ + _out ~= "#let ocn(n, body) = block(breakable: true, width: 100%, {\n" + ~ " place(left, dx: -1.8cm, dy: 0.2em,\n" + ~ " text(size: 7pt, fill: luma(130))[#n#label(\"t\" + str(n))])\n" + ~ " body\n})\n"; + /+ ↓ an object with no number of its own still needs its block +/ + _out ~= "#let obj(body) = block(breakable: true, width: 100%, body)\n"; + /+ ↓ monospace as a font rather than as `raw`: typst's raw() takes a + ,*string*, so it cannot carry the nested markup an object may have + inside a monospace run. A code block is a different thing and uses + raw. + +/ + _out ~= "#let mono(body) = text(font: (\"DejaVu Sans Mono\",), body)\n"; + /+ ↓ a named target with no output of its own +/ + _out ~= "#let anchor(name) = [#box(width: 0pt)#label(name)]\n\n"; + return _out.data; + } + /+ ↓ an image, with its own width where the document declared one. Capped at + the text width: the declared width is in pixels and a 600 pixel figure + read as 600 points is wider than the page. + . + The opening link marker is kept, as the html writer keeps it: an image in + the markup sits inside a link whose target may be empty or may be a url, + and the link handling that follows needs the pair. + +/ + string images(string txt) { + return replaceAll!((m) { + string _file = m["img"].to!string; + int _w = 0; + try { _w = m["width"].to!int; } catch (Exception) {} + string _width = (_w > 0) + ? ("width: " ~ min(_w, 440).to!string ~ "pt") + : "width: 100%"; + string _rest = m["post"].to!string; + string _alt; + if (auto a = _rest.matchFirst(rgx.inline_image_alt)) { + _alt = a["alt"].to!string; + _rest = a.post.to!string; + } + string _img = "#image(" ~ str("image/" ~ _file) ~ ", " ~ _width; + if (_alt.length > 0) { _img ~= ", alt: " ~ str(_alt); } + _img ~= ")"; + return m["pre"].to!string ~ keep(_img) ~ _rest; + })(txt, rgx.inline_image); + } + /+ ↓ the font faces, as typst functions rather than typst markup: a function + call cannot be broken by the character beside it, and the markup form can + +/ + string fontFace(string txt) { + string wrapWith(string fn) { + return keep("#" ~ fn ~ "[") ~ "$1" ~ keep("]"); + } + return txt + .replaceAll(rgx.inline_emphasis, wrapWith("emph")) + .replaceAll(rgx.inline_bold, wrapWith("strong")) + .replaceAll(rgx.inline_italics, wrapWith("emph")) + .replaceAll(rgx.inline_underscore, wrapWith("underline")) + .replaceAll(rgx.inline_superscript, wrapWith("super")) + .replaceAll(rgx.inline_subscript, wrapWith("sub")) + .replaceAll(rgx.inline_mono, wrapWith("mono")) + .replaceAll(rgx.inline_strike, wrapWith("strike")) + .replaceAll(rgx.inline_insert, wrapWith("underline")) + .replaceAll(rgx.inline_cite, wrapWith("emph")); + } + /+ ↓ the line breaks the markup asked for. Not inside code, whose text is + preserved as it stands. + +/ + string breaks(O)(string txt, const O obj) { + if (obj.metainfo.is_a == "code") { return txt; } + string _br = keep("#linebreak()"); + return txt + .replaceAll(rgx.br_line, _br) + .replaceAll(rgx.br_line_inline, _br) + .replaceAll(rgx.br_line_spaced, _br ~ _br) + .replaceAll(rgx.line_break, _br) + .replaceAll(rgx.nbsp_char, " "); + } + /+ ↓ the line-leading markers of a grouped object: an indent as a measurement, + a hanging indent's inner level, a bullet. + +/ + string groupIndents(O)(string txt, const O obj) { + if (obj.metainfo.is_a != "group" && obj.metainfo.is_a != "block") { return txt; } + string pad(string n) { return keep("#h(" ~ n ~ " * 12pt)"); } + return txt + .replaceAll(rgx.grouped_para_indent_hang, pad("$2")) + .replaceAll(rgx.grouped_para_bullet_indent, pad("$1") ~ keep("●") ~ " ") + .replaceAll(rgx.grouped_para_bullet, keep("●") ~ " ") + .replaceAll(rgx.grouped_para_indent, pad("$1")); + } + /+ ↓ a link. An internal target becomes a link to a label, and only when the + label will exist: typst refuses to compile a link to one that will not, + which is the check and so also the constraint. + +/ + string links(O)(string txt, const O obj, bool[string] targets) { + auto _stow = obj.stow.link; + txt = replaceAll!((m) { + size_t _num = m["num"].to!size_t; + string _url = (_num < _stow.length) ? _stow[_num].to!string : ""; + return m["linked_text"].to!string ~ "┤" ~ _url ~ "├"; + })(txt, rgx.inline_link_number_only); + txt = txt.replaceAll(rgx.inline_link_empty, "$1"); + txt = replaceAll!((m) { + string _text = escape(fontFace(m.captures[1].to!string)); + string _target = m.captures[2].to!string; + /+ ↓ an internal link to another segment: this is one document, so the + segment is this document and only the fragment matters + +/ + if (auto im = _target.matchFirst(rgx.inline_link_seg_and_hash)) { + _target = "#" ~ im["hash"].to!string; + } + if (_target.length > 1 && _target[0] == '#') { + string _l = labelOf(_target[1 .. $]); + return keep((_l in targets) + ? "#link(label(" ~ str(_l) ~ "))[" ~ unprotect(_text) ~ "]" + : unprotect(_text)); + } + if (_target.length == 0) { return keep(unprotect(_text)); } + return keep("#link(" ~ str(_target) ~ ")[" ~ unprotect(_text) ~ "]"); + })(txt, rgx.inline_link); + /+ ↓ whatever marker is left where a link was not one +/ + return txt.replaceAll(rgx.mark_internal_site_lnk, ""); + } + /+ ↓ a note becomes a footnote at the point that refers to it, carrying the + document's own number rather than typst's count, and a way back to the + object it came from. + . + Object to note is the footnote mark, note to object is one link, and + typst puts the note at the foot of the page the reference is on. + +/ + string notes(O)(string txt, const O obj, bool[string] targets) { + string _back; + if (obj.metainfo.ocn != 0) { + string _l = labelOf(obj.metainfo.ocn.to!string); + if (_l in targets) { + _back = " #link(label(" ~ str(_l) ~ "))[#sym.arrow.l.hook]"; + } + } + return replaceAll!((m) { + string _mark = m["num"].to!string; + string _note = m["note"].to!string; + /+ ↓ the note's own text is content in its own right, so it goes through + the same pipeline - links, faces, escaping - and nothing else + +/ + string _body = escape(breaks!()(fontFace( + links!()(images(_note), obj, targets)), obj)); + return keep("#footnote(numbering: _ => " ~ str(_mark) ~ ")[" + ~ unprotect(_body) ~ _back ~ "]"); + })(txt, rgx.inline_notes_al_all_note); + } + /+ ↓ an anchor the markup placed in the text +/ + string inlineAnchors(string txt, bool[string] targets, ref bool[string] placed) { + string[] _names; + foreach (m; txt.matchAll(rgx.inline_link_anchor)) { + _names ~= m["anchor"].to!string; + } + foreach (name; _names) { + string _l = labelOf(name); + bool _place = (_l in targets) && !(_l in placed); + if (_place) { placed[_l] = true; } + txt = txt.replaceFirst(rgx.inline_link_anchor, + _place ? keep("#anchor(" ~ str(_l) ~ ")") : ""); + } + return txt; + } + /+ ↓ the named targets an object carries, as labels with no output +/ + string anchors(O)(const O obj, bool[string] targets, ref bool[string] placed) { + auto _out = appender!string; + string _own = (obj.metainfo.ocn == 0) + ? "" : labelOf(obj.metainfo.ocn.to!string); + void put(string name) { + if (name.length == 0) { return; } + string _l = labelOf(name); + /+ ↓ the ocn's own label is placed on the number in the margin +/ + if (_l == _own || (_l in placed)) { return; } + placed[_l] = true; + _out ~= "#anchor(" ~ str(_l) ~ ")"; + } + put(obj.metainfo.identifier.to!string); + put(obj.tags.heading_lev_anchor_tag.to!string); + foreach (at; obj.tags.anchor_tags) { put(at.to!string); } + return _out.data; + } + /+ ↓ the object's text as typst content +/ + string inlineText(O)( + string txt, const O obj, bool[string] targets, ref bool[string] placed + ) { + txt = notes!()(txt, obj, targets); + txt = images(txt); + txt = links!()(txt, obj, targets); + txt = fontFace(txt); + txt = inlineAnchors(txt, targets, placed); + txt = groupIndents!()(txt, obj); + txt = breaks!()(txt, obj); + txt = escape(txt); + return unprotect(txt); + } + /+ ↓ eight markup levels, eight heading levels. typst has no six-level + ceiling, so the document's own depth goes through unchanged - which the + html output cannot do. + +/ + int headingLevel(string level) { + switch (level) { + case "A": return 1; + case "B": return 2; + case "C": return 3; + case "D": return 4; + case "1": return 5; + case "2": return 6; + case "3": return 7; + case "4": return 8; + default: return 1; + } + } + /+ ↓ an object in its block, with its number in the margin when it has one +/ + string wrap(O)(const O obj, string content) { + return (obj.metainfo.ocn == 0) + ? "#obj[" ~ content ~ "]" + : "#ocn(" ~ obj.metainfo.ocn.to!string ~ ")[" ~ content ~ "]"; + } + /+ ↓ the object number a table of contents entry points at, where it points at + one. An entry may name a section by its anchor instead, and then there is + no number to show. + +/ + string tocTarget(O)(const O obj, bool[string] targets) { + foreach (m; obj.text.matchAll(rgx.any_internal_target)) { + string _frag = m["frag"].to!string; + if (_frag.length == 0) { continue; } + bool _digits = true; + foreach (c; _frag) { if (c < '0' || c > '9') { _digits = false; break; } } + if (_digits && (labelOf(_frag) in targets)) { return _frag; } + } + return ""; + } + /+ ↓ whether a table of contents entry points at anything this writer renders. + . + A table of contents names what is in the document, so an entry whose only + target is gone is not an entry: the endnotes are footnotes here, so the + section they had is gone and so is the line that pointed at it. Decided + from the target and not from the rendering, because a nested link - a + link inside a link, which the flat markers cannot express - leaves an + entry with no link to show and it is still an entry. + +/ + bool pointsAtSomething(O)(const O obj, bool[string] targets) { + bool _saw = false; + foreach (m; obj.text.matchAll(rgx.any_internal_target)) { + _saw = true; + if (labelOf(m["frag"].to!string) in targets) { return true; } + } + return !_saw; + } + /+ ↓ one object +/ + string object(O,M)( + const O obj, M doc_matters, bool[string] targets, ref bool[string] placed + ) { + /+ ↓ an object with no text of its own writes nothing. A poem is such an + object: a container that closes a run of verses and carries the same + object number as the last of them. + +/ + if (obj.text.strip.length == 0) { return ""; } + string _body; + string _anchors; + switch (obj.metainfo.is_a) { + case "heading": + /+ ↓ a dummy heading is generated furniture whose text repeats the heading + above it; its anchors are what the table of contents wants + +/ + if (obj.metainfo.dummy_heading) { + _anchors = anchors!()(obj, targets, placed); + return (_anchors.length > 0) ? _anchors ~ newlines : ""; + } + _body = inlineText!()(obj.text, obj, targets, placed); + if (_body.strip.length == 0) { return ""; } + _anchors = anchors!()(obj, targets, placed); + return wrap!()(obj, "#heading(level: " + ~ headingLevel(obj.metainfo.marked_up_level.to!string).to!string + ~ ")[" ~ _anchors ~ _body ~ "]") ~ newlines; + case "toc": + if (obj.metainfo.dummy_heading) { return ""; } + _body = inlineText!()(obj.text, obj, targets, placed); + if (_body.strip.length == 0) { return ""; } + if (!pointsAtSomething!()(obj, targets)) { return ""; } + /+ ↓ the entry's indent, the heading as a link, a dotted leader, and the + object number as the locator. + . + Not a page number, and not typst's #outline(), which would make one. + A page number is a fact about this typesetting: change the paper, the + orientation, the type size, the leading or the margins and every + number in the table moves. The object number does not. So the reader + is given the reference that holds across every output of this + document and across every setting of this one, and it is the same + number the margin carries and the same number a citation uses. + +/ + int _indent = (obj.attrib.indent_hang > 1) ? (obj.attrib.indent_hang - 1) : 0; + string _locator; + string _n = tocTarget!()(obj, targets); + if (_n.length > 0) { + _locator = " #box(width: 1fr, inset: (x: 3pt), repeat[.]) " + ~ "#link(label(" ~ str(labelOf(_n)) ~ "))[" ~ _n ~ "]"; + } + return "#block(inset: (left: " ~ (_indent * 8).to!string ~ "pt))[" + ~ _body ~ _locator ~ "]" ~ newlines; + case "table": + return table!()(obj, doc_matters, targets, placed); + case "code": + /+ ↓ nothing inside a code block is interpreted, which is what a raw block + is for. The one substitution is the non-breaking space marker spine + uses inside code to hold a leading indent. + +/ + string _lang = (obj.metainfo.syntax.length > 0) + ? ("lang: " ~ str(obj.metainfo.syntax.to!string) ~ ", ") : ""; + string _code = obj.text.replaceAll(rgx.nbsp_char, " "); + return wrap!()(obj, "#raw(block: true, " ~ _lang ~ str(_code) ~ ")") ~ newlines; + case "quote": + _body = inlineText!()(obj.text, obj, targets, placed); + if (_body.strip.length == 0) { return ""; } + _anchors = anchors!()(obj, targets, placed); + return wrap!()(obj, "#block(inset: (left: 18pt, right: 18pt))[" + ~ _anchors ~ _body ~ "]") ~ newlines; + case "group": case "block": + _body = inlineText!()(obj.text, obj, targets, placed); + if (_body.strip.length == 0) { return ""; } + _anchors = anchors!()(obj, targets, placed); + return wrap!()(obj, "#block[" ~ _anchors ~ _body ~ "]") ~ newlines; + case "verse": case "poem": + /+ ↓ every line break kept, and no justification, because a line of verse + ends where the poet ended it + +/ + _body = inlineText!()(obj.text, obj, targets, placed); + if (_body.strip.length == 0) { return ""; } + _anchors = anchors!()(obj, targets, placed); + return wrap!()(obj, "#block(inset: (left: 12pt))[#set par(justify: false);" + ~ _anchors ~ _body ~ "]") ~ newlines; + case "para": case "blurb": case "glossary": case "bibliography": + case "bookindex": + if (obj.metainfo.dummy_heading) { return ""; } + _body = inlineText!()(obj.text, obj, targets, placed); + if (_body.strip.length == 0) { return ""; } + _anchors = anchors!()(obj, targets, placed); + string _bullet = obj.attrib.bullet ? "●~" : ""; + string _content = (obj.attrib.indent_base > 0) + ? "#block(inset: (left: " ~ (obj.attrib.indent_base * 12).to!string + ~ "pt))[" ~ _anchors ~ _bullet ~ _body ~ "]" + : _anchors ~ _bullet ~ _body; + return wrap!()(obj, _content) ~ newlines; + default: + return ""; + } + } + /+ ↓ a table. The declared column widths become fractions, which is what a + width relative to the others means, and a header row becomes one. + +/ + string table(O,M)( + const O obj, M doc_matters, bool[string] targets, ref bool[string] placed + ) { + auto _rows = obj.text.split(rgx.table_delimiter_row); + int _cols = max(obj.table.number_of_columns.to!int, 1); + auto _out = appender!string; + _out ~= "#table(\n columns: ("; + if (obj.table.column_widths.length > 0) { + foreach (i, w; obj.table.column_widths) { + _out ~= (i > 0 ? ", " : "") ~ w.to!string ~ "fr"; + } + } else { + foreach (i; 0 .. _cols) { _out ~= (i > 0 ? ", " : "") ~ "1fr"; } + } + _out ~= "),\n align: ("; + foreach (i; 0 .. _cols) { + string _a = (i < obj.table.column_aligns.length) + ? obj.table.column_aligns[i].to!string : "l"; + _out ~= (i > 0 ? ", " : "") + ~ ((_a == "c") ? "center" : (_a == "r") ? "right" : "left"); + } + _out ~= "),\n stroke: 0.4pt + luma(180),\n inset: 4pt,\n"; + bool _first = true; + foreach (row; _rows) { + if (row.strip.length == 0) { continue; } + auto _cells = row.split(rgx.table_delimiter_col); + foreach (cell; _cells) { + string _c = inlineText!()(cell, obj, targets, placed); + _out ~= (_first && obj.table.heading) + ? " table.cell(fill: luma(240))[#strong[" ~ _c ~ "]],\n" + : " [" ~ _c ~ "],\n"; + } + _first = false; + } + _out ~= ")"; + return wrap!()(obj, _out.data) ~ newlines; + } + /+ ↓ the document's objects +/ + string typstBody(D,M)(const D doc_abstraction, M doc_matters) { + auto _targets = definedTargets!()(doc_abstraction, doc_matters); + auto _placed = ocnLabels!()(doc_abstraction, doc_matters); + auto _out = appender!string; + foreach (part; typstSections!()(doc_matters)) { + foreach (obj; doc_abstraction[part]) { + if (obj.metainfo.is_of_type == "comment") { continue; } + _out ~= guardBlockStarts(object!()(obj, doc_matters, _targets, _placed)); + } + } + return _out.data; + } + void outputTypst(D,M)( + const D doc_abstraction, + M doc_matters, + ) { + auto pth_typst = spinePathsTypst(doc_matters); + try { + if (!exists(pth_typst.base_pth)) { (pth_typst.base_pth).mkdirRecurse; } + } catch (ErrnoException ex) { + } + if (doc_matters.opt.action.vox_gt_1) { + writeln(" ", pth_typst.typst_file); + } + { + auto f = File(pth_typst.typst_file, "w"); + f.write(typstHead!()(doc_matters)); + f.write(typstBody!()(doc_abstraction, doc_matters)); + } + /+ ↓ the images the document names, beside the .typ, because a .typ names + its images by path and does not carry them + +/ + if (doc_matters.srcs.image_list.length > 0) { + try { + if (!exists(pth_typst.images)) { (pth_typst.images).mkdirRecurse; } + foreach (image; doc_matters.srcs.image_list) { + string _in = doc_matters.src.image_dir_path ~ "/" ~ image; + string _out = pth_typst.images ~ "/" ~ image; + if (exists(_in) && !exists(_out)) { _in.copy(_out); } + } + } catch (Exception ex) { + } + } + } +} +/+ ↓ the pdf: typst compiles the .typ (just) written by the writer. + . + This is the first time spine invokes a typesetter itself. By contrast sisu + (ruby) invoked xelatex after producing latex, but for spine this was + considered to be overly complicated and too slow. Spine's latex build + writes a .tex and leaves xelatex to the caller; --pdf does not, because one + static binary is a different proposition from a 6.6 GB texlive and because + a pdf is what was asked for. + . + One .typ, one pdf per paper size spine is configured for, the paper passed + in rather than compiled in: + typst compile --input paper=a4 --input orientation=portrait + . + If typst is not available on the machine this says so, naming the binary, + and the run continues with the .typ being written either way (as with latex + output it can be compiled later). ++/ +template outputTypstPdf() { + import sisudoc.outputs.io_out; + import sisudoc.outputs.io_out.paths_output; + import std.conv : to; + import std.file; + import std.process; + import std.stdio; + enum typst_binary = "typst"; + @trusted bool typstOnPath() { + try { + auto probe = execute([typst_binary, "--version"]); + return probe.status == 0; + } catch (Exception ex) { + return false; + } + } + /+ ↓ spine`s paper names and typst`s are not the same vocabulary. Mapped + rather than passed through, and a paper typst does not have is named and + skipped. + +/ + string typstPaper(string spine_paper) { + switch (spine_paper) { + case "a3": return "a3"; + case "a4": return "a4"; + case "a5": return "a5"; + case "b4": return "iso-b4"; + case "b5": return "iso-b5"; + case "letter": return "us-letter"; + case "legal": return "us-legal"; + default: return ""; + } + } + @trusted void outputTypstPdf(M)(M doc_matters) { + auto pth_typst = spinePathsTypst(doc_matters); + if (!exists(pth_typst.typst_file)) { + writeln("spine: --pdf found no ", pth_typst.typst_file, " to compile"); + return; + } + if (!typstOnPath) { + writeln("spine: --pdf needs the \"", typst_binary, + "\" binary on the path and did not find it; the .typ is written:"); + writeln(" ", pth_typst.typst_file); + writeln(" typst compile --root ", pth_typst.base_pth, " <that file> <a pdf>"); + return; + } + try { + if (!exists(pth_typst.pdf_pth)) { (pth_typst.pdf_pth).mkdirRecurse; } + } catch (Exception ex) { + } + foreach (paper_size_orientation; doc_matters.conf_make_meta.conf.set_papersize) { + string _paper = paper_size_orientation; + string _orientation = "portrait"; + foreach (i, c; paper_size_orientation) { + if (c == '.') { + _paper = paper_size_orientation[0 .. i]; + _orientation = paper_size_orientation[(i + 1) .. $]; + break; + } + } + string _typst_paper = typstPaper(_paper); + if (_typst_paper.length == 0) { + writeln("spine: typst has no paper named \"", _paper, + "\"; skipping ", paper_size_orientation); + continue; + } + string _pdf = pth_typst.pdf_file_with_path(paper_size_orientation); + auto _cmd = [ + typst_binary, "compile", + "--root", pth_typst.base_pth, + "--input", "paper=" ~ _typst_paper, + "--input", "orientation=" ~ _orientation, + pth_typst.typst_file, + _pdf, + ]; + if (doc_matters.opt.action.vox_gt_1) { writeln(" ", _pdf); } + auto _run = execute(_cmd); + if (_run.status != 0) { + writeln("spine: typst could not compile ", pth_typst.typst_file, + " for ", paper_size_orientation); + writeln(_run.output); + } + } + } +} diff --git a/src/sisudoc/spine.d b/src/sisudoc/spine.d index 0494a34..e42a5f1 100644 --- a/src/sisudoc/spine.d +++ b/src/sisudoc/spine.d @@ -177,6 +177,8 @@ string program_name = "spine"; "parallel" : false, "parallel-subprocesses" : false, "pdf" : false, + "typst" : false, + "typ" : false, "pdf-color-links" : false, "pdf-init" : false, "po4a-cfg" : false, @@ -288,7 +290,7 @@ string program_name = "spine"; "html-seg", "process html output", &opts["html-seg"], "html-scroll", "process html output", &opts["html-scroll"], "lang", "=[lang code e.g. =en or =en,es]", &settings["lang"], - "latex", "latex output (for pdfs)", &opts["latex"], + "latex", "latex output (.tex, for further processing to a pdf)", &opts["latex"], "latex-color-links", "mono or color links for pdfs", &opts["latex-color-links"], "latex-init", "initialise latex shared files (see latex-header-sty)", &opts["latex-init"], "latex-header-sty", "latex document header sty files", &opts["latex-header-sty"], @@ -307,7 +309,9 @@ string program_name = "spine"; "abstraction-source", "=/path/to/(.sst|pod|.ssp|.ocda.db) identify it, and load it if it is an abstraction", &settings["abstraction-source"], "parallel", "parallelise document processing (opt-in; slower than serial)", &opts["parallel"], "parallel-subprocesses", "nested parallelisation", &opts["parallel-subprocesses"], - "pdf", "latex output for pdfs", &opts["pdf"], + "pdf", "pdf output (writes the .typ and compiles it with typst)", &opts["pdf"], + "typst", "typst output (.typ, for further processing to a pdf)", &opts["typst"], + "typ", "typst output (--typst)", &opts["typ"], "pdf-color-links", "mono or color links for pdfs", &opts["pdf-color-links"], "pdf-init", "initialise latex shared files (see latex-header-sty)", &opts["pdf-init"], "po4a-cfg", "print the po4a configuration for this document", &opts["po4a-cfg"], @@ -578,7 +582,7 @@ Verifying: stdout.flush; exit(_loaded.loaded ? 0 : 1); } - enum outTask { source_or_pod, sqlite, sqlite_multi, latex, odt, epub, html_scroll, html_seg, html_stuff, text, skel } + enum outTask { source_or_pod, sqlite, sqlite_multi, latex, typst, odt, epub, html_scroll, html_seg, html_stuff, text, skel } struct OptActions { @trusted bool allow_downloads() { return opts["allow-downloads"]; @@ -703,8 +707,17 @@ Verifying: @trusted bool html_stuff() { return (opts["html"] || opts["html-scroll"] || opts["html-seg"]) ? true : false; } + /+ ↓ --pdf is produced via typst (it no longer writes latex: --latex writes .tex, --typst (--typ) writes .typ, and + --pdf writes .typ and compiles it. + +/ @trusted bool latex() { - return (opts["latex"] || opts["pdf"]) ? true : false; + return opts["latex"]; + } + @trusted bool typst() { + return (opts["typst"] || opts["typ"] || opts["pdf"]) ? true : false; + } + @trusted bool pdf() { + return opts["pdf"]; } @trusted bool latex_color_links() { return (opts["latex-color-links"] || opts["pdf-color-links"]) ? true : false; @@ -1060,6 +1073,7 @@ Verifying: if (html_stuff) schedule ~= outTask.html_stuff; if (odt) schedule ~= outTask.odt; if (latex) schedule ~= outTask.latex; + if (typst) schedule ~= outTask.typst; if (text) schedule ~= outTask.text; if (skel) schedule ~= outTask.skel; return schedule.sort().uniq; @@ -1077,6 +1091,7 @@ Verifying: || epub || odt || latex + || typst || manifest || sqlite_discrete || sqlite_delete @@ -1094,6 +1109,7 @@ Verifying: || html_seg || html_scroll || latex + || typst || odt || manifest || show_abstraction |
