diff options
| -rw-r--r-- | org/default_paths.org | 61 | ||||
| -rw-r--r-- | org/default_regex.org | 5 | ||||
| -rw-r--r-- | org/out_markdown.org | 608 | ||||
| -rw-r--r-- | org/out_metadata.org | 5 | ||||
| -rw-r--r-- | org/output_hub.org | 15 | ||||
| -rw-r--r-- | org/spine.org | 20 | ||||
| -rw-r--r-- | src/sisudoc/ocda/meta/rgx.d | 5 | ||||
| -rw-r--r-- | src/sisudoc/outputs/io_out/hub.d | 8 | ||||
| -rw-r--r-- | src/sisudoc/outputs/io_out/markdown.d | 602 | ||||
| -rw-r--r-- | src/sisudoc/outputs/io_out/metadata.d | 5 | ||||
| -rw-r--r-- | src/sisudoc/outputs/io_out/paths_output.d | 54 | ||||
| -rw-r--r-- | src/sisudoc/outputs/io_out/rgx.d | 5 | ||||
| -rw-r--r-- | src/sisudoc/spine.d | 20 |
13 files changed, 1401 insertions, 12 deletions
diff --git a/org/default_paths.org b/org/default_paths.org index 0a4af80..aa166ef 100644 --- a/org/default_paths.org +++ b/org/default_paths.org @@ -1049,6 +1049,7 @@ import sisudoc.ocda.meta.rgx_files; <<template_paths_sqlite_0>> <<template_paths_sqlite_1>> <<template_paths_sqlite_2>> +<<template_paths_markdown_md>> <<template_paths_typst_pdf>> <<template_paths_text>> <<template_paths_skel>> @@ -1747,6 +1748,66 @@ template spinePathsSQLite() { } #+END_SRC +** _markdown_ :markdown:md: + +#+NAME: template_paths_markdown_md +#+BEGIN_SRC d +/+ ↓ where the markdown output goes. + . + <lang>/markdown/<doc>.<lang>.md, which is where the other three document + outputs go: html, epub and text all sit under the language, and markdown is + a document rather than a rendered artefact. (latex, typst and pdf are flat, + being the print path.) Called "markdown" and not "md" because none of its + four siblings is a name shortened to its extension - text is the word where + the extension is .txt, odf the family where it is .odt - and because "md" + beside the language directories reads as an eleventh language. + . + The language stays in the filename, as it does for the epub and the text, + because a .md is the format most likely to be copied out of the tree and the + directory carries the language only while the file stays in it. + . + The images are the shared set at the output root, the ones the html + references, not a copy: two directories down the reference is ../../image/, + exactly as the scroll html's is. ++/ +template spinePathsMarkdown() { + import std.conv; + auto spinePathsMarkdown(M)( + M doc_matters, + ) { + auto out_pth = spineOutPaths!()(doc_matters.output_path, doc_matters.src.language); + struct _PathsStruct { + /+ ↓ named as the epub, the html and the latex are: from the document + filename and the language, not from the document uid. The uid joins the + pod name where the two differ, and every artefact the site links to is + named for the document +/ + string base_filename(string fn_src) { + return fn_src.baseName.stripExtension; + } + string base_pth() { + return (((out_pth.output_base).chainPath("markdown")).asNormalizedPath).array; + } + string markdown_file() { + return ((base_pth.chainPath(base_filename(doc_matters.src.filename) + ~ "." ~ doc_matters.src.language + ~ ".md") + ).asNormalizedPath).array; + } + /+ ↓ the shared image directory at the output root, written by whichever + output gets there first and read by all of them +/ + string images() { + return (((out_pth.output_root).chainPath("image")).asNormalizedPath).array; + } + /+ ↓ how the .md refers to them, from two directories down +/ + string images_rel() { + return "../../image"; + } + } + return _PathsStruct(); + } +} +#+END_SRC + ** _typst_ :typst:pdf: #+NAME: template_paths_typst_pdf diff --git a/org/default_regex.org b/org/default_regex.org index 659c640..4b2178c 100644 --- a/org/default_regex.org +++ b/org/default_regex.org @@ -557,6 +557,11 @@ static br_line_spaced = ctRegex!(`\s*┚\s*`, "mg"); carry their own copy of this in rgx_xhtml may be needed by other writers +/ static line_break = ctRegex!(` [\\]{2}`, "m"); +/+ ↓ the trailing backslash a book index entry ends its line with. The text + and odt writers build this one inline, as regex("\\s*\\\\"), each time they + need it ++/ +static trailing_backslash = ctRegex!(`\s*\\`, "mg"); #+END_SRC #+BEGIN_SRC d diff --git a/org/out_markdown.org b/org/out_markdown.org new file mode 100644 index 0000000..f7f8145 --- /dev/null +++ b/org/out_markdown.org @@ -0,0 +1,608 @@ +-*- mode: org -*- +#+TITLE: sisudoc spine (doc_reform) output xmls +#+DESCRIPTION: documents - structuring, publishing in multiple formats & search +#+FILETAGS: :spine:output:text: +#+AUTHOR: Ralph Amissah +#+EMAIL: [[mailto:ralph.amissah@gmail.com][ralph.amissah@gmail.com]] +#+COPYRIGHT: Copyright (C) 2015 (continuously updated, current 2026) Ralph Amissah +#+LANGUAGE: en +#+STARTUP: content hideblocks hidestars noindent entitiespretty +#+PROPERTY: header-args+ :eval never-export :exports code +#+PROPERTY: header-args+ :noweb yes :padline no +#+PROPERTY: header-args+ :results silent :cache no +#+PROPERTY: header-args+ :mkdirp yes +#+OPTIONS: H:3 num:nil toc:t \n:t ::t |:t ^:nil -:t f:t *:t +- magic single double-quote → " ← FIX changes hilighting behavior (occuring + after it) in org document. INVESTIGATE (org-mode CONFIG?) FIND & FIX + +- [[./doc-reform.org][doc-reform.org]] [[./][org/]] +- [[./output_hub.org][output_hub]] + +* Markdown (Text) +** outputText template + +#+HEADER: :tangle "../src/sisudoc/outputs/io_out/markdown.d" +#+HEADER: :noweb yes +#+BEGIN_SRC d +<<doc_header_including_copyright_and_license>> +module sisudoc.outputs.io_out.markdown; +@safe: +/+ ↓ the markdown output: the document as CommonMark. + . + An alternative to --text, and the better one where a reader will render it: + the text file has the object numbers, the markdown has the object numbers + *and the graph between them* - every citation, every note, every index + entry and every line of the table of contents is a working link, and the + headings are headings. + . + The dialect is **CommonMark**, plus pipe tables, which every renderer worth + using understands. Raw inline html is used in exactly two places, both for + the object number, because markdown has no anchor syntax and nothing else + will do: + . + <a id="12"></a>The text of the paragraph.<sup>[12](#12)</sup> + . + The anchor is at the head of the object so that a link to it lands on the + object and not past it; the number is at the end, as a link to itself, so + that a reader can take the citation out of the rendered page. That is what + the html output does with an object number. + . + **Markdown's own footnote syntax is not used.** It is an extension rather + than CommonMark, and a renderer numbers footnotes itself, in the order it + meets them. The document's note numbers are part of its citation and are + not a renderer's to choose. So a note is written where it is referred to, + with the document's own number and a link back to the object - the same + decision the typst output makes. + . + A port of tools/sisudoc-ocda-writers/dlang/src/write/markdown.d, which is + held over the 36 document reference collection against the html writer's + objects word for word, and against the Gleam writer byte for byte. ++/ +template outputMarkdown() { + import sisudoc.outputs.io_out; + import sisudoc.outputs.io_out.rgx; + import sisudoc.outputs.io_out.paths_output; + import std.algorithm : canFind, max; + import std.array : appender, array, join, replace; + import std.array : asplit = split; + import std.conv : to; + import std.exception : ErrnoException; + import std.file; + import std.regex : matchAll, matchFirst, replaceAll, split; + import std.stdio; + import std.string : strip; + mixin spineRgxOut; + static auto rgx = RgxO(); + enum newline = "\n"; + enum newlines = "\n\n"; + /+ ↓ the markers that protect the markdown this writer emits from the escaping + that follows. + . + The escaping has to come *last*, and this is why: the abstraction's own + markers are built from the characters markdown reserves - "⑆*┨" for + emphasis, "⑆_┨" for an underscore face - so escaping first turns them + into "⑆\*┨" and no face is ever recognised. + +/ + enum keep_open = ""; + enum keep_close = ""; + /+ ↓ a third marker of the same kind, standing for the directory the images are + in. + . + An image reference is written while an object is written, and *where this + file sits* is known only to spinePathsMarkdown. Naming the path here as + well would be two places agreeing about one path until one of them + changed, which is the defect the pdf links had. So the reference carries a + marker and the marker is substituted once, where the document is written, + out of the one place that knows. + +/ + enum image_dir_mark = ""; + string keep(string markdown_source) { + return keep_open ~ markdown_source ~ keep_close; + } + string unprotect(string txt) { + return txt.replace(keep_open, "").replace(keep_close, ""); + } + /+ ↓ the characters markdown reads as syntax, escaped wherever they occur and + only outside a protected run. + . + The same blunt rule the typst output uses, and for the same reason: a + closed set is checkable and a rule about positions is not. The three html + characters become entities rather than backslash escapes, because + markdown passes raw html through and "<b" would open a tag. + +/ + string escape(string txt) { + auto _out = appender!string; + int depth = 0; + foreach (dchar c; txt) { + if (c == keep_open.to!dchar) { depth++; continue; } + if (c == keep_close.to!dchar) { if (depth > 0) { depth--; } continue; } + if (depth > 0) { _out ~= c; continue; } + switch (c) { + case '&': _out ~= "&"; break; + case '<': _out ~= "<"; break; + case '>': _out ~= ">"; break; + case '\\': case '`': case '*': case '_': case '[': case ']': + case '#': case '|': case '~': + _out ~= '\\'; _out ~= c; break; + default: _out ~= c; break; + } + } + return _out.data; + } + /+ ↓ the document's own text, escaped for a protected run: the same rule, run + early, and then flattened so that the result is inert +/ + string inert(string txt) { + return unprotect(escape(txt)); + } + /+ ↓ a "- ", "+ ", "> " or "1. " at the start of a line is markdown's own: a + list item, an enumeration item, a blockquote. The writer uses the first of + them itself, and a line of the document's that begins the same way would + be read as one - which is what a verse or a group can do. Guarded rather + than escaped everywhere, because these are special only in that position. + +/ + string guardLineStarts(string s) { + auto _out = appender!string; + bool at_start = true; + for (size_t i = 0; i < s.length; i++) { + if (at_start) { + size_t j = i; + while (j < s.length && s[j] == ' ') { j++; } + if (j < s.length) { + char c = s[j]; + bool marker = (c == '-' || c == '+' || c == '>') + && j + 1 < s.length && (s[j + 1] == ' ' || s[j + 1] == '\t'); + size_t k = j; + while (k < s.length && s[k] >= '0' && s[k] <= '9') { k++; } + bool enumerated = k > j && k + 1 < s.length + && (s[k] == '.' || s[k] == ')') && s[k + 1] == ' '; + if (marker || enumerated) { + out_put(_out, s[i .. (enumerated ? k : j)]); + _out ~= '\\'; + i = (enumerated ? k : j) - 1; + at_start = false; + continue; + } + } + at_start = false; + } + _out ~= s[i]; + if (s[i] == '\n') { at_start = true; } + } + return _out.data; + } + void out_put(T)(ref T sink, string s) { sink ~= s; } + /+ ↓ a yaml scalar. Quoted always, so that a colon or a leading marker in a + title cannot change the shape of the document's own metadata. +/ + string yaml(string s) { + return "\"" ~ s.replace("\\", "\\\\").replace("\"", "\\\"") + .replace("\n", " ") ~ "\""; + } + /+ ↓ which sections, in which order: the latex sequence, less the endnotes, + because a note is written where it is referred to +/ + string[] markdownSections(M)(M doc_matters) { + string[] _seq; + foreach (part; doc_matters.has.keys_seq.latex) { + if (part == "endnotes") { continue; } + _seq ~= part; + } + return _seq; + } + /+ ↓ the table of contents depth of each indent_hang the document uses. + . + spine's indent_hang is not a depth: it reserves 1 to 3 for levels above + the body, so a toc uses 1 and then jumps to 4, 5, 6. A markdown list + nests by two spaces a level and cannot take a jump, so the values the + document actually uses are ranked. + +/ + size_t[int] tocDepths(D,M)(const D doc_abstraction, M doc_matters) { + bool[int] _seen; + foreach (part; markdownSections!()(doc_matters)) { + if (part != "toc") { continue; } + foreach (obj; doc_abstraction[part]) { + if (obj.metainfo.is_of_type == "comment") { continue; } + _seen[obj.attrib.indent_hang.to!int] = true; + } + } + import std.algorithm : sort; + auto _levels = _seen.keys.sort.array; + size_t[int] _depth; + foreach (i, level; _levels) { _depth[level] = i; } + return _depth; + } + /+ ↓ the document's metadata as yaml front matter, which is how a markdown file + carries metadata and what every static site generator reads +/ + string markdownHead(M)(M doc_matters) { + auto _out = appender!string; + _out ~= "---" ~ newline; + void put(string key, string value) { + if (value.length > 0) { _out ~= key ~ ": " ~ yaml(value) ~ newline; } + } + put("title", doc_matters.conf_make_meta.meta.title_full); + put("author", doc_matters.conf_make_meta.meta.creator_author); + put("date", doc_matters.conf_make_meta.meta.date_published); + put("language", doc_matters.src.language); + put("copyright", doc_matters.conf_make_meta.meta.rights_copyright); + put("license", doc_matters.conf_make_meta.meta.rights_license); + _out ~= "---" ~ newlines; + return _out.data; + } + /+ ↓ an image, with its alt text where the document gave one +/ + string images(string txt) { + return replaceAll!((m) { + string _rest = m["post"].to!string; + string _alt; + if (auto a = _rest.matchFirst(rgx.inline_image_alt)) { + _alt = a["alt"].to!string; + _rest = a.post.to!string; + } + return m["pre"].to!string + ~ keep("") + ~ _rest; + })(txt, rgx.inline_image); + } + /+ ↓ a fragment that names an object of this document +/ + bool isObjectNumber(string s) { + if (s.length == 0 || s == "0") { return false; } + foreach (c; s) { if (c < '0' || c > '9') { return false; } } + return true; + } + /+ ↓ a link target is a url or a fragment and not prose, so the escaping the + text went through comes off it again +/ + string unescapeTarget(string target) { + auto _out = appender!string; + for (size_t i = 0; i < target.length; i++) { + if (target[i] == 0x5c && i + 1 < target.length) { _out ~= target[++i]; continue; } + _out ~= target[i]; + } + return _out.data.replace("&", "&").replace("<", "<").replace(">", ">"); + } + /+ ↓ the font faces. Emphasis and strong are markdown's own; the rest have no + markdown and take the html element markdown itself would produce. + . + The spaces inside a run are dropped: the abstraction's marker takes in the + space that was beside the run, the text before it already ends in one, and + markdown's emphasis will not open on a space - "* lex *" is three words + and two asterisks, not an italic. + +/ + string fontFace(string txt) { + string wrap(string open, string close, string inner) { + size_t a = 0, b = inner.length; + while (a < b && inner[a] == ' ') { a++; } + while (b > a && inner[b - 1] == ' ') { b--; } + return keep(open) ~ inner[a .. b] ~ keep(close); + } + string apply(alias pattern)(string s, string open, string close) { + return replaceAll!((m) => wrap(open, close, m.captures[1].to!string))(s, pattern); + } + txt = apply!(rgx.inline_emphasis)(txt, "**", "**"); + txt = apply!(rgx.inline_bold)(txt, "**", "**"); + txt = apply!(rgx.inline_italics)(txt, "*", "*"); + txt = apply!(rgx.inline_underscore)(txt, "<u>", "</u>"); + txt = apply!(rgx.inline_superscript)(txt, "<sup>", "</sup>"); + txt = apply!(rgx.inline_subscript)(txt, "<sub>", "</sub>"); + txt = apply!(rgx.inline_mono)(txt, "`", "`"); + txt = apply!(rgx.inline_strike)(txt, "~~", "~~"); + txt = apply!(rgx.inline_insert)(txt, "<ins>", "</ins>"); + txt = apply!(rgx.inline_cite)(txt, "<cite>", "</cite>"); + return txt; + } + /+ ↓ a link. An internal target becomes a fragment, which is the object number, + and so resolves against the anchor every object carries. + . + And *only* an object number: those are the anchors this writer defines, so + a target that names something else - the endnotes section, an anchor tag + the markup placed - would be a link to nothing. The text stands on its own + instead, which is what a reader sees anyway. + +/ + string links(O)(string txt, const O obj) { + auto _stow = obj.stow.link; + txt = replaceAll!((m) { + size_t _num = m["num"].to!size_t; + string _url = (_num < _stow.length) ? _stow[_num].to!string : ""; + return m["linked_text"].to!string ~ "┤" ~ _url ~ "├"; + })(txt, rgx.inline_link_number_only); + return replaceAll!((m) { + string _text = m.captures[1].to!string; + string _target = m.captures[2].to!string; + if (auto im = _target.matchFirst(rgx.inline_link_seg_and_hash)) { + _target = "#" ~ im["hash"].to!string; + } + string _shown = inert(fontFace(_text)); + if (_target.length == 0) { return keep(_shown); } + string _clean = unescapeTarget(_target); + if (_clean.length > 0 && _clean[0] == '#' + && !isObjectNumber(_clean[1 .. $])) { + return keep(_shown); + } + return keep("[" ~ _shown ~ "](" ~ _clean ~ ")"); + })(txt, rgx.inline_link); + } + /+ ↓ a note reference: a superscript link to the note, which sits under the + object that refers to it +/ + string noteRefs(string txt) { + return replaceAll!((m) { + string _mark = m["num"].to!string; + return keep("<sup>[" ~ inert(_mark) ~ "](#note-" ~ _mark ~ ")</sup>"); + })(txt, rgx.inline_notes_al_all_note); + } + /+ ↓ the line breaks the markup asked for, and the grouped indents +/ + string breaks(O)(string txt, const O obj) { + if (obj.metainfo.is_a == "code") { return txt; } + if (obj.metainfo.is_a == "group" || obj.metainfo.is_a == "block") { + txt = txt + .replaceAll(rgx.grouped_para_indent_hang, "$2$2") + .replaceAll(rgx.grouped_para_bullet_indent, "$1● ") + .replaceAll(rgx.grouped_para_bullet, "● ") + .replaceAll(rgx.grouped_para_indent, "$1$1"); + } + return txt + .replaceAll(rgx.nbsp_char, " ") + .replaceAll(rgx.br_line, newline) + .replaceAll(rgx.br_line_inline, newline) + .replaceAll(rgx.br_line_spaced, newlines) + .replaceAll(rgx.line_break, newline) + .replaceAll(rgx.mark_internal_site_lnk, ""); + } + /+ ↓ an object's text as markdown. + . + The order is the one every writer here uses: images before links, because + an image sits inside a link; then the note references, the faces, the + anchors and the breaks. The escaping is last. + +/ + string inlineText(O)(string txt, const O obj) { + txt = images(txt); + txt = links!()(txt, obj); + txt = noteRefs(txt); + txt = fontFace(txt); + txt = txt.replaceAll(rgx.inline_link_anchor, ""); + txt = breaks!()(txt, obj); + return guardLineStarts(unprotect(escape(txt))); + } + /+ ↓ six levels of heading, which is all markdown has. A document may be eight + deep; the two deepest take the sixth, and the object number, the anchor and + the table of contents carry the true depth. + +/ + int headingLevel(string level) { + switch (level) { + case "A": return 1; + case "B": return 2; + case "C": return 3; + case "D": return 4; + case "1": return 5; + case "2": case "3": case "4": return 6; + default: return 1; + } + } + /+ ↓ the object's anchor, so that a citation can point at it +/ + string anchor(O)(const O obj) { + return (obj.metainfo.ocn == 0) + ? "" : "<a id=\"" ~ obj.metainfo.ocn.to!string ~ "\"></a>"; + } + /+ ↓ the object's number, at the end of it, as a link to itself +/ + string ocnMark(O)(const O obj) { + if (obj.metainfo.ocn == 0) { return ""; } + string _n = obj.metainfo.ocn.to!string; + return "<sup>[" ~ _n ~ "](#" ~ _n ~ ")</sup>"; + } + /+ ↓ the notes the object referred to, each where it was referred to, with the + document's own number and a link back to the object +/ + string notes(O)(const O obj) { + auto _out = appender!string; + string _back = (obj.metainfo.ocn == 0) + ? "" : " [↩](#" ~ obj.metainfo.ocn.to!string ~ ")"; + foreach (m; obj.text.matchAll(rgx.inline_notes_al_all_note)) { + string _mark = m["num"].to!string; + _out ~= newlines ~ "<a id=\"note-" ~ _mark ~ "\"></a><sup>" + ~ inert(_mark) ~ ".</sup> " + ~ inlineText!()(m["note"].to!string, obj) ~ _back; + } + /+ ↓ and a blank line after the last of them, or the next object begins on the + same line and markdown reads the two as one paragraph +/ + return (_out.data.length > 0) ? _out.data ~ newlines : ""; + } + /+ ↓ the object a table of contents entry points at, where that is an object + number +/ + string internalTarget(O)(const O obj) { + foreach (m; obj.text.matchAll(rgx.any_internal_target)) { + string _frag = m["frag"].to!string; + if (isObjectNumber(_frag)) { return _frag; } + } + return ""; + } + /+ ↓ one object, and then the notes it referred to. + . + The notes are gathered here rather than inside each kind, because a + reference can be anywhere - the markup manual has one in a heading - and a + reference whose note was never written is a link to nothing. + +/ + string object(O,M)(const O obj, M doc_matters, size_t[int] toc_depth) { + return objectBody!()(obj, doc_matters, toc_depth) ~ notes!()(obj); + } + string objectBody(O,M)(const O obj, M doc_matters, size_t[int] toc_depth) { + switch (obj.metainfo.is_a) { + case "heading": + string _hashes; + foreach (_; 0 .. headingLevel(obj.metainfo.marked_up_level.to!string)) { + _hashes ~= "#"; + } + return _hashes ~ " " ~ anchor!()(obj) ~ inlineText!()(obj.text, obj) + ~ ocnMark!()(obj) ~ newlines; + case "toc": + /+ ↓ a list item at its depth, the heading as a link, and the object number + as the locator. The number and not a page: a page is a fact about one + typesetting, and the object number is the same reference in every + output of the document. + +/ + string _indent; + if (auto d = obj.attrib.indent_hang.to!int in toc_depth) { + foreach (_; 0 .. *d) { _indent ~= " "; } + } + string _n = internalTarget!()(obj); + string _locator = (_n.length > 0) + ? (" <sup>[" ~ _n ~ "](#" ~ _n ~ ")</sup>") : ""; + return _indent ~ "- " ~ inlineText!()(obj.text, obj) ~ _locator ~ newline; + case "table": + return table!()(obj, doc_matters); + case "code": + /+ ↓ a fenced block, long enough that the code cannot close it +/ + string _txt = obj.text.replaceAll(rgx.nbsp_char, " "); + string _fence = "```"; + while (_txt.canFind(_fence)) { _fence ~= "`"; } + return anchor!()(obj) ~ newline ~ _fence ~ obj.metainfo.syntax.to!string + ~ newline ~ _txt ~ newline ~ _fence ~ newline ~ ocnMark!()(obj) ~ newlines; + case "quote": + return "> " ~ anchor!()(obj) ~ inlineText!()(obj.text, obj) + ~ ocnMark!()(obj) ~ newlines; + case "verse": case "poem": case "group": case "block": + /+ ↓ every line kept, with CommonMark's own hard line break - a backslash at + the end of a line - rather than the two trailing spaces an editor will + strip +/ + auto _rows = inlineText!()(obj.text, obj).asplit(newline); + auto _out = appender!string; + _out ~= anchor!()(obj); + foreach (i, row; _rows) { + _out ~= row; + if (i + 1 < _rows.length) { _out ~= "\\" ~ newline; } + } + return _out.data ~ ocnMark!()(obj) ~ newlines; + case "bookindex": + /+ ↓ a list item, and every object it names a link +/ + string _txt = obj.text.replaceAll(rgx.trailing_backslash, ""); + return "- " ~ anchor!()(obj) ~ inlineText!()(_txt, obj) + ~ ocnMark!()(obj) ~ newlines; + case "para": case "blurb": case "glossary": case "bibliography": + string _bullet = obj.attrib.bullet ? "- " : ""; + string _indent; + foreach (_; 0 .. obj.attrib.indent_base) { _indent ~= " "; } + return _indent ~ _bullet ~ anchor!()(obj) ~ inlineText!()(obj.text, obj) + ~ ocnMark!()(obj) ~ newlines; + default: + return ""; + } + } + /+ ↓ a pipe table, which needs a header row: markdown has no table without one, + so a table the document did not give a header gets an empty row and keeps + all of its own. + +/ + string table(O,M)(const O obj, M doc_matters) { + auto _rows = obj.text.split(rgx.table_delimiter_row); + int _cols = max(obj.table.number_of_columns.to!int, 1); + string[][] _body; + foreach (row; _rows) { + if (row.strip.length == 0) { continue; } + string[] _cells; + foreach (cell; row.split(rgx.table_delimiter_col)) { + _cells ~= inlineText!()(cell, obj).replace("|", "\\|"); + } + _body ~= _cells; + } + string rule() { + auto r = appender!string; + r ~= "|"; + foreach (i; 0 .. _cols) { + string _a = (i < obj.table.column_aligns.length) + ? obj.table.column_aligns[i].to!string : "l"; + r ~= (_a == "c") ? " :---: |" : (_a == "r") ? " ---: |" : " :--- |"; + } + return r.data ~ newline; + } + string line(string[] cells) { + auto l = appender!string; + l ~= "|"; + foreach (i; 0 .. _cols) { + l ~= " " ~ ((i < cells.length) ? cells[i] : "") ~ " |"; + } + return l.data ~ newline; + } + auto _out = appender!string; + _out ~= anchor!()(obj) ~ newline; + if (_body.length > 0 && obj.table.heading) { + _out ~= line(_body[0]) ~ rule(); + foreach (row; _body[1 .. $]) { _out ~= line(row); } + } else { + _out ~= line([]) ~ rule(); + foreach (row; _body) { _out ~= line(row); } + } + return _out.data ~ ocnMark!()(obj) ~ newlines; + } + string markdownBody(D,M)(const D doc_abstraction, M doc_matters) { + auto _toc_depth = tocDepths!()(doc_abstraction, doc_matters); + auto _out = appender!string; + foreach (part; markdownSections!()(doc_matters)) { + foreach (obj; doc_abstraction[part]) { + if (obj.metainfo.is_of_type == "comment") { continue; } + if (obj.metainfo.dummy_heading) { continue; } + if (obj.text.strip.length == 0) { continue; } + _out ~= object!()(obj, doc_matters, _toc_depth); + } + } + return _out.data; + } + void outputMarkdown(D,M)( + const D doc_abstraction, + M doc_matters, + ) { + auto pth_md = spinePathsMarkdown(doc_matters); + try { + if (!exists(pth_md.base_pth)) { (pth_md.base_pth).mkdirRecurse; } + } catch (ErrnoException ex) { + } + if (doc_matters.opt.action.vox_gt_1) { + writeln(" ", pth_md.markdown_file); + } + { + auto f = File(pth_md.markdown_file, "w"); + f.write(markdownHead!()(doc_matters)); + /+ ↓ the image directory, named once, here, out of the paths module that + decided where this file goes +/ + f.write(markdownBody!()(doc_abstraction, doc_matters) + .replace(image_dir_mark, pth_md.images_rel)); + } + /+ ↓ the images the document names, into the shared image directory at the + output root, because markdown names them by path and does not carry them. + . + The same set the html references, and not a copy of it: a file already + there is left alone, so this costs nothing when the html output has + written them and still works when --markdown is asked for on its own. + +/ + if (doc_matters.srcs.image_list.length > 0) { + try { + if (!exists(pth_md.images)) { (pth_md.images).mkdirRecurse; } + foreach (image; doc_matters.srcs.image_list) { + string _in = doc_matters.src.image_dir_path ~ "/" ~ image; + string _out = pth_md.images ~ "/" ~ image; + if (exists(_in) && !exists(_out)) { _in.copy(_out); } + } + } catch (Exception ex) { + } + } + } +} +#+END_SRC + +* org includes +** spine project VERSION + +#+NAME: spine_version +#+HEADER: :noweb yes +#+BEGIN_SRC emacs-lisp +<<./sisudoc_spine_version_info_and_doc_header_including_copyright_and_license.org:spine_project_version()>> +#+END_SRC + +** year + +#+NAME: year +#+HEADER: :noweb yes +#+BEGIN_SRC emacs-lisp +<<./sisudoc_spine_version_info_and_doc_header_including_copyright_and_license.org:year()>> +#+END_SRC + +** document header including copyright & license + +#+NAME: doc_header_including_copyright_and_license +#+HEADER: :noweb yes +#+BEGIN_SRC emacs-lisp +<<./sisudoc_spine_version_info_and_doc_header_including_copyright_and_license.org:spine_doc_header_including_copyright_and_license()>> +#+END_SRC + +* __END__ diff --git a/org/out_metadata.org b/org/out_metadata.org index 2c210e6..c868bad 100644 --- a/org/out_metadata.org +++ b/org/out_metadata.org @@ -155,6 +155,11 @@ metadata_ ~= "<p class=\"lev1\">● outputs: [ html:& ~ " ※ seg </a>] " ~ "[<a href=\"../../" ~ pth_epub.internal_base ~ "/" ~ doc_matters.src.filename_base ~ "." ~ doc_matters.src.language ~ ".epub\" class=\"lnkicon\">" ~ " ◆ epub </a>] "; +if (doc_matters.opt.action.markdown) { + metadata_ ~= "[<a href=\"../markdown/" ~ doc_matters.src.filename_base + ~ "." ~ doc_matters.src.language ~ ".md\" class=\"lnkicon\">" + ~ " □ markdown </a>] "; +} if ((doc_matters.opt.action.html_link_pdf) || (doc_matters.opt.action.html_link_pdf_a4)) { metadata_ ~= "[ pdf: <a href=\"../../pdf/" ~ doc_matters.src.filename_base diff --git a/org/output_hub.org b/org/output_hub.org index 1349940..ab0424c 100644 --- a/org/output_hub.org +++ b/org/output_hub.org @@ -50,7 +50,7 @@ template outputHub() { mixin spinePo4aConfig; write(po4aConfig(doc.matters)); } - enum outTask { source_or_pod, sqlite, sqlite_multi, latex, typst, odt, epub, html_scroll, html_seg, html_stuff, text, skel } + enum outTask { source_or_pod, sqlite, sqlite_multi, latex, typst, markdown, odt, epub, html_scroll, html_seg, html_stuff, text, skel } void Scheduled(D)(int sched, D doc) { auto msg = Msg!()(doc.matters); <<output_scheduled_task_source_or_pod>> @@ -61,6 +61,7 @@ template outputHub() { <<output_scheduled_task_html_out>> <<output_scheduled_task_latex>> <<output_scheduled_task_typst_pdf>> + <<output_scheduled_task_markdown>> <<output_scheduled_task_text>> <<output_scheduled_task_odt>> <<output_scheduled_task_sqlite>> @@ -264,6 +265,18 @@ if (sched == outTask.typst) { } #+END_SRC +**** text :markdown:md: + +#+NAME: output_scheduled_task_markdown +#+BEGIN_SRC d +if (sched == outTask.markdown) { + msg.v("markdown processing... "); + import sisudoc.outputs.io_out.markdown; + outputMarkdown!()(doc.abstraction, doc.matters); + msg.vv("markdown done"); +} +#+END_SRC + **** text :text:txt: #+NAME: output_scheduled_task_text diff --git a/org/spine.org b/org/spine.org index 13878e5..65bebff 100644 --- a/org/spine.org +++ b/org/spine.org @@ -559,6 +559,8 @@ bool[string] opts = [ "latex-header-sty" : false, "light" : false, "manifest" : false, + "markdown" : false, + "md" : false, "hide-ocn" : false, "no-ocn" : false, "ocda-db" : false, @@ -701,6 +703,8 @@ auto helpInfo = getopt(args, "latex-header-sty", "latex document header sty files", &opts["latex-header-sty"], "light", "default light theme", &opts["light"], "manifest", "process manifest output", &opts["manifest"], + "markdown", "markdown output (.md, CommonMark)", &opts["markdown"], + "md", "markdown output (--markdown)", &opts["md"], "no-ocn", "object cite numbers", &opts["no-ocn"], "ocda-db", "document abstraction (write ocda.db sqlite file)", &opts["ocda-db"], "ocn-off", "object cite numbers", &opts["ocn-off"], @@ -1013,7 +1017,7 @@ if (settings["abstraction-source"].length > 0) { #+NAME: spine_args_get_options_aa2str #+BEGIN_SRC d -enum outTask { source_or_pod, sqlite, sqlite_multi, latex, typst, odt, epub, html_scroll, html_seg, html_stuff, text, skel } +enum outTask { source_or_pod, sqlite, sqlite_multi, latex, typst, markdown, odt, epub, html_scroll, html_seg, html_stuff, text, skel } struct OptActions { @trusted bool allow_downloads() { return opts["allow-downloads"]; @@ -1144,6 +1148,9 @@ struct OptActions { @trusted bool latex() { return opts["latex"]; } + @trusted bool markdown() { + return (opts["markdown"] || opts["md"]) ? true : false; + } @trusted bool typst() { return (opts["typst"] || opts["typ"] || opts["pdf"]) ? true : false; } @@ -1505,6 +1512,7 @@ struct OptActions { if (odt) schedule ~= outTask.odt; if (latex) schedule ~= outTask.latex; if (typst) schedule ~= outTask.typst; + if (markdown) schedule ~= outTask.markdown; if (text) schedule ~= outTask.text; if (skel) schedule ~= outTask.skel; return schedule.sort().uniq; @@ -1523,25 +1531,28 @@ struct OptActions { || odt || latex || typst + || markdown + || text || manifest || sqlite_discrete || sqlite_delete || sqlite_update - || text || skel ) ? true : false; } @trusted bool require_processing_files() { return ( opts["abstraction"] - || epub || curate || html || html_seg || html_scroll + || epub + || odt || latex || typst - || odt + || markdown + || text || manifest || show_abstraction || ocda_db @@ -1552,7 +1563,6 @@ struct OptActions { || source_or_pod || sqlite_discrete || sqlite_update - || text || xhtml || skel ) ? true : false; diff --git a/src/sisudoc/ocda/meta/rgx.d b/src/sisudoc/ocda/meta/rgx.d index f418582..249fd3d 100644 --- a/src/sisudoc/ocda/meta/rgx.d +++ b/src/sisudoc/ocda/meta/rgx.d @@ -230,6 +230,11 @@ static template spineRgxIn() { carry their own copy of this in rgx_xhtml may be needed by other writers +/ static line_break = ctRegex!(` [\\]{2}`, "m"); + /+ ↓ the trailing backslash a book index entry ends its line with. The text + and odt writers build this one inline, as regex("\\s*\\\\"), each time they + need it + +/ + static trailing_backslash = ctRegex!(`\s*\\`, "mg"); /+ inline markup footnotes endnotes +/ static inline_notes_al = ctRegex!(`【(?:[*+]\s+|\s*)(.+?)】`, "mg"); static inline_notes_al_special = ctRegex!(`【(?:[*+]\s+)(.+?)】`, "mg"); // TODO remove match when special footnotes are implemented diff --git a/src/sisudoc/outputs/io_out/hub.d b/src/sisudoc/outputs/io_out/hub.d index 1a0ed70..4b1f016 100644 --- a/src/sisudoc/outputs/io_out/hub.d +++ b/src/sisudoc/outputs/io_out/hub.d @@ -77,7 +77,7 @@ template outputHub() { mixin spinePo4aConfig; write(po4aConfig(doc.matters)); } - enum outTask { source_or_pod, sqlite, sqlite_multi, latex, typst, odt, epub, html_scroll, html_seg, html_stuff, text, skel } + enum outTask { source_or_pod, sqlite, sqlite_multi, latex, typst, markdown, odt, epub, html_scroll, html_seg, html_stuff, text, skel } void Scheduled(D)(int sched, D doc) { auto msg = Msg!()(doc.matters); if (sched == outTask.source_or_pod) { @@ -143,6 +143,12 @@ template outputHub() { } msg.vv("typst done"); } + if (sched == outTask.markdown) { + msg.v("markdown processing... "); + import sisudoc.outputs.io_out.markdown; + outputMarkdown!()(doc.abstraction, doc.matters); + msg.vv("markdown done"); + } if (sched == outTask.text) { msg.v("text processing... "); import sisudoc.outputs.io_out.text; diff --git a/src/sisudoc/outputs/io_out/markdown.d b/src/sisudoc/outputs/io_out/markdown.d new file mode 100644 index 0000000..b575b8c --- /dev/null +++ b/src/sisudoc/outputs/io_out/markdown.d @@ -0,0 +1,602 @@ +/+ +- Name: SisuDoc Spine, Doc Reform [a part of] + - Description: documents, structuring, processing, publishing, search + - static content generator + + - Author: Ralph Amissah + [ralph.amissah@gmail.com] + + - Copyright: (C) 2015 (continuously updated, current 2026) Ralph Amissah, All Rights Reserved. + + - License: AGPL 3 or later: + + Spine (SiSU), a framework for document structuring, publishing and + search + + Copyright (C) Ralph Amissah + + This program is free software: you can redistribute it and/or modify it + under the terms of the GNU AFERO General Public License as published by the + Free Software Foundation, either version 3 of the License, or (at your + option) any later version. + + This program is distributed in the hope that it will be useful, but WITHOUT + ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or + FITNESS FOR A PARTICULAR PURPOSE. See the GNU General Public License for + more details. + + You should have received a copy of the GNU General Public License along with + this program. If not, see [https://www.gnu.org/licenses/]. + + If you have Internet connection, the latest version of the AGPL should be + available at these locations: + [https://www.fsf.org/licensing/licenses/agpl.html] + [https://www.gnu.org/licenses/agpl.html] + + - Spine (by Doc Reform, related to SiSU) uses standard: + - docReform markup syntax + - standard SiSU markup syntax with modified headers and minor modifications + - docReform object numbering + - standard SiSU object citation numbering & system + + - Homepages: + [https://www.sisudoc.org] + [https://www.doc-reform.org] + + - Git + [https://git.sisudoc.org/] + ++/ +module sisudoc.outputs.io_out.markdown; +@safe: +/+ ↓ the markdown output: the document as CommonMark. + . + An alternative to --text, and the better one where a reader will render it: + the text file has the object numbers, the markdown has the object numbers + *and the graph between them* - every citation, every note, every index + entry and every line of the table of contents is a working link, and the + headings are headings. + . + The dialect is **CommonMark**, plus pipe tables, which every renderer worth + using understands. Raw inline html is used in exactly two places, both for + the object number, because markdown has no anchor syntax and nothing else + will do: + . + <a id="12"></a>The text of the paragraph.<sup>[12](#12)</sup> + . + The anchor is at the head of the object so that a link to it lands on the + object and not past it; the number is at the end, as a link to itself, so + that a reader can take the citation out of the rendered page. That is what + the html output does with an object number. + . + **Markdown's own footnote syntax is not used.** It is an extension rather + than CommonMark, and a renderer numbers footnotes itself, in the order it + meets them. The document's note numbers are part of its citation and are + not a renderer's to choose. So a note is written where it is referred to, + with the document's own number and a link back to the object - the same + decision the typst output makes. + . + A port of tools/sisudoc-ocda-writers/dlang/src/write/markdown.d, which is + held over the 36 document reference collection against the html writer's + objects word for word, and against the Gleam writer byte for byte. ++/ +template outputMarkdown() { + import sisudoc.outputs.io_out; + import sisudoc.outputs.io_out.rgx; + import sisudoc.outputs.io_out.paths_output; + import std.algorithm : canFind, max; + import std.array : appender, array, join, replace; + import std.array : asplit = split; + import std.conv : to; + import std.exception : ErrnoException; + import std.file; + import std.regex : matchAll, matchFirst, replaceAll, split; + import std.stdio; + import std.string : strip; + mixin spineRgxOut; + static auto rgx = RgxO(); + enum newline = "\n"; + enum newlines = "\n\n"; + /+ ↓ the markers that protect the markdown this writer emits from the escaping + that follows. + . + The escaping has to come *last*, and this is why: the abstraction's own + markers are built from the characters markdown reserves - "⑆*┨" for + emphasis, "⑆_┨" for an underscore face - so escaping first turns them + into "⑆\*┨" and no face is ever recognised. + +/ + enum keep_open = ""; + enum keep_close = ""; + /+ ↓ a third marker of the same kind, standing for the directory the images are + in. + . + An image reference is written while an object is written, and *where this + file sits* is known only to spinePathsMarkdown. Naming the path here as + well would be two places agreeing about one path until one of them + changed, which is the defect the pdf links had. So the reference carries a + marker and the marker is substituted once, where the document is written, + out of the one place that knows. + +/ + enum image_dir_mark = ""; + string keep(string markdown_source) { + return keep_open ~ markdown_source ~ keep_close; + } + string unprotect(string txt) { + return txt.replace(keep_open, "").replace(keep_close, ""); + } + /+ ↓ the characters markdown reads as syntax, escaped wherever they occur and + only outside a protected run. + . + The same blunt rule the typst output uses, and for the same reason: a + closed set is checkable and a rule about positions is not. The three html + characters become entities rather than backslash escapes, because + markdown passes raw html through and "<b" would open a tag. + +/ + string escape(string txt) { + auto _out = appender!string; + int depth = 0; + foreach (dchar c; txt) { + if (c == keep_open.to!dchar) { depth++; continue; } + if (c == keep_close.to!dchar) { if (depth > 0) { depth--; } continue; } + if (depth > 0) { _out ~= c; continue; } + switch (c) { + case '&': _out ~= "&"; break; + case '<': _out ~= "<"; break; + case '>': _out ~= ">"; break; + case '\\': case '`': case '*': case '_': case '[': case ']': + case '#': case '|': case '~': + _out ~= '\\'; _out ~= c; break; + default: _out ~= c; break; + } + } + return _out.data; + } + /+ ↓ the document's own text, escaped for a protected run: the same rule, run + early, and then flattened so that the result is inert +/ + string inert(string txt) { + return unprotect(escape(txt)); + } + /+ ↓ a "- ", "+ ", "> " or "1. " at the start of a line is markdown's own: a + list item, an enumeration item, a blockquote. The writer uses the first of + them itself, and a line of the document's that begins the same way would + be read as one - which is what a verse or a group can do. Guarded rather + than escaped everywhere, because these are special only in that position. + +/ + string guardLineStarts(string s) { + auto _out = appender!string; + bool at_start = true; + for (size_t i = 0; i < s.length; i++) { + if (at_start) { + size_t j = i; + while (j < s.length && s[j] == ' ') { j++; } + if (j < s.length) { + char c = s[j]; + bool marker = (c == '-' || c == '+' || c == '>') + && j + 1 < s.length && (s[j + 1] == ' ' || s[j + 1] == '\t'); + size_t k = j; + while (k < s.length && s[k] >= '0' && s[k] <= '9') { k++; } + bool enumerated = k > j && k + 1 < s.length + && (s[k] == '.' || s[k] == ')') && s[k + 1] == ' '; + if (marker || enumerated) { + out_put(_out, s[i .. (enumerated ? k : j)]); + _out ~= '\\'; + i = (enumerated ? k : j) - 1; + at_start = false; + continue; + } + } + at_start = false; + } + _out ~= s[i]; + if (s[i] == '\n') { at_start = true; } + } + return _out.data; + } + void out_put(T)(ref T sink, string s) { sink ~= s; } + /+ ↓ a yaml scalar. Quoted always, so that a colon or a leading marker in a + title cannot change the shape of the document's own metadata. +/ + string yaml(string s) { + return "\"" ~ s.replace("\\", "\\\\").replace("\"", "\\\"") + .replace("\n", " ") ~ "\""; + } + /+ ↓ which sections, in which order: the latex sequence, less the endnotes, + because a note is written where it is referred to +/ + string[] markdownSections(M)(M doc_matters) { + string[] _seq; + foreach (part; doc_matters.has.keys_seq.latex) { + if (part == "endnotes") { continue; } + _seq ~= part; + } + return _seq; + } + /+ ↓ the table of contents depth of each indent_hang the document uses. + . + spine's indent_hang is not a depth: it reserves 1 to 3 for levels above + the body, so a toc uses 1 and then jumps to 4, 5, 6. A markdown list + nests by two spaces a level and cannot take a jump, so the values the + document actually uses are ranked. + +/ + size_t[int] tocDepths(D,M)(const D doc_abstraction, M doc_matters) { + bool[int] _seen; + foreach (part; markdownSections!()(doc_matters)) { + if (part != "toc") { continue; } + foreach (obj; doc_abstraction[part]) { + if (obj.metainfo.is_of_type == "comment") { continue; } + _seen[obj.attrib.indent_hang.to!int] = true; + } + } + import std.algorithm : sort; + auto _levels = _seen.keys.sort.array; + size_t[int] _depth; + foreach (i, level; _levels) { _depth[level] = i; } + return _depth; + } + /+ ↓ the document's metadata as yaml front matter, which is how a markdown file + carries metadata and what every static site generator reads +/ + string markdownHead(M)(M doc_matters) { + auto _out = appender!string; + _out ~= "---" ~ newline; + void put(string key, string value) { + if (value.length > 0) { _out ~= key ~ ": " ~ yaml(value) ~ newline; } + } + put("title", doc_matters.conf_make_meta.meta.title_full); + put("author", doc_matters.conf_make_meta.meta.creator_author); + put("date", doc_matters.conf_make_meta.meta.date_published); + put("language", doc_matters.src.language); + put("copyright", doc_matters.conf_make_meta.meta.rights_copyright); + put("license", doc_matters.conf_make_meta.meta.rights_license); + _out ~= "---" ~ newlines; + return _out.data; + } + /+ ↓ an image, with its alt text where the document gave one +/ + string images(string txt) { + return replaceAll!((m) { + string _rest = m["post"].to!string; + string _alt; + if (auto a = _rest.matchFirst(rgx.inline_image_alt)) { + _alt = a["alt"].to!string; + _rest = a.post.to!string; + } + return m["pre"].to!string + ~ keep("") + ~ _rest; + })(txt, rgx.inline_image); + } + /+ ↓ a fragment that names an object of this document +/ + bool isObjectNumber(string s) { + if (s.length == 0 || s == "0") { return false; } + foreach (c; s) { if (c < '0' || c > '9') { return false; } } + return true; + } + /+ ↓ a link target is a url or a fragment and not prose, so the escaping the + text went through comes off it again +/ + string unescapeTarget(string target) { + auto _out = appender!string; + for (size_t i = 0; i < target.length; i++) { + if (target[i] == 0x5c && i + 1 < target.length) { _out ~= target[++i]; continue; } + _out ~= target[i]; + } + return _out.data.replace("&", "&").replace("<", "<").replace(">", ">"); + } + /+ ↓ the font faces. Emphasis and strong are markdown's own; the rest have no + markdown and take the html element markdown itself would produce. + . + The spaces inside a run are dropped: the abstraction's marker takes in the + space that was beside the run, the text before it already ends in one, and + markdown's emphasis will not open on a space - "* lex *" is three words + and two asterisks, not an italic. + +/ + string fontFace(string txt) { + string wrap(string open, string close, string inner) { + size_t a = 0, b = inner.length; + while (a < b && inner[a] == ' ') { a++; } + while (b > a && inner[b - 1] == ' ') { b--; } + return keep(open) ~ inner[a .. b] ~ keep(close); + } + string apply(alias pattern)(string s, string open, string close) { + return replaceAll!((m) => wrap(open, close, m.captures[1].to!string))(s, pattern); + } + txt = apply!(rgx.inline_emphasis)(txt, "**", "**"); + txt = apply!(rgx.inline_bold)(txt, "**", "**"); + txt = apply!(rgx.inline_italics)(txt, "*", "*"); + txt = apply!(rgx.inline_underscore)(txt, "<u>", "</u>"); + txt = apply!(rgx.inline_superscript)(txt, "<sup>", "</sup>"); + txt = apply!(rgx.inline_subscript)(txt, "<sub>", "</sub>"); + txt = apply!(rgx.inline_mono)(txt, "`", "`"); + txt = apply!(rgx.inline_strike)(txt, "~~", "~~"); + txt = apply!(rgx.inline_insert)(txt, "<ins>", "</ins>"); + txt = apply!(rgx.inline_cite)(txt, "<cite>", "</cite>"); + return txt; + } + /+ ↓ a link. An internal target becomes a fragment, which is the object number, + and so resolves against the anchor every object carries. + . + And *only* an object number: those are the anchors this writer defines, so + a target that names something else - the endnotes section, an anchor tag + the markup placed - would be a link to nothing. The text stands on its own + instead, which is what a reader sees anyway. + +/ + string links(O)(string txt, const O obj) { + auto _stow = obj.stow.link; + txt = replaceAll!((m) { + size_t _num = m["num"].to!size_t; + string _url = (_num < _stow.length) ? _stow[_num].to!string : ""; + return m["linked_text"].to!string ~ "┤" ~ _url ~ "├"; + })(txt, rgx.inline_link_number_only); + return replaceAll!((m) { + string _text = m.captures[1].to!string; + string _target = m.captures[2].to!string; + if (auto im = _target.matchFirst(rgx.inline_link_seg_and_hash)) { + _target = "#" ~ im["hash"].to!string; + } + string _shown = inert(fontFace(_text)); + if (_target.length == 0) { return keep(_shown); } + string _clean = unescapeTarget(_target); + if (_clean.length > 0 && _clean[0] == '#' + && !isObjectNumber(_clean[1 .. $])) { + return keep(_shown); + } + return keep("[" ~ _shown ~ "](" ~ _clean ~ ")"); + })(txt, rgx.inline_link); + } + /+ ↓ a note reference: a superscript link to the note, which sits under the + object that refers to it +/ + string noteRefs(string txt) { + return replaceAll!((m) { + string _mark = m["num"].to!string; + return keep("<sup>[" ~ inert(_mark) ~ "](#note-" ~ _mark ~ ")</sup>"); + })(txt, rgx.inline_notes_al_all_note); + } + /+ ↓ the line breaks the markup asked for, and the grouped indents +/ + string breaks(O)(string txt, const O obj) { + if (obj.metainfo.is_a == "code") { return txt; } + if (obj.metainfo.is_a == "group" || obj.metainfo.is_a == "block") { + txt = txt + .replaceAll(rgx.grouped_para_indent_hang, "$2$2") + .replaceAll(rgx.grouped_para_bullet_indent, "$1● ") + .replaceAll(rgx.grouped_para_bullet, "● ") + .replaceAll(rgx.grouped_para_indent, "$1$1"); + } + return txt + .replaceAll(rgx.nbsp_char, " ") + .replaceAll(rgx.br_line, newline) + .replaceAll(rgx.br_line_inline, newline) + .replaceAll(rgx.br_line_spaced, newlines) + .replaceAll(rgx.line_break, newline) + .replaceAll(rgx.mark_internal_site_lnk, ""); + } + /+ ↓ an object's text as markdown. + . + The order is the one every writer here uses: images before links, because + an image sits inside a link; then the note references, the faces, the + anchors and the breaks. The escaping is last. + +/ + string inlineText(O)(string txt, const O obj) { + txt = images(txt); + txt = links!()(txt, obj); + txt = noteRefs(txt); + txt = fontFace(txt); + txt = txt.replaceAll(rgx.inline_link_anchor, ""); + txt = breaks!()(txt, obj); + return guardLineStarts(unprotect(escape(txt))); + } + /+ ↓ six levels of heading, which is all markdown has. A document may be eight + deep; the two deepest take the sixth, and the object number, the anchor and + the table of contents carry the true depth. + +/ + int headingLevel(string level) { + switch (level) { + case "A": return 1; + case "B": return 2; + case "C": return 3; + case "D": return 4; + case "1": return 5; + case "2": case "3": case "4": return 6; + default: return 1; + } + } + /+ ↓ the object's anchor, so that a citation can point at it +/ + string anchor(O)(const O obj) { + return (obj.metainfo.ocn == 0) + ? "" : "<a id=\"" ~ obj.metainfo.ocn.to!string ~ "\"></a>"; + } + /+ ↓ the object's number, at the end of it, as a link to itself +/ + string ocnMark(O)(const O obj) { + if (obj.metainfo.ocn == 0) { return ""; } + string _n = obj.metainfo.ocn.to!string; + return "<sup>[" ~ _n ~ "](#" ~ _n ~ ")</sup>"; + } + /+ ↓ the notes the object referred to, each where it was referred to, with the + document's own number and a link back to the object +/ + string notes(O)(const O obj) { + auto _out = appender!string; + string _back = (obj.metainfo.ocn == 0) + ? "" : " [↩](#" ~ obj.metainfo.ocn.to!string ~ ")"; + foreach (m; obj.text.matchAll(rgx.inline_notes_al_all_note)) { + string _mark = m["num"].to!string; + _out ~= newlines ~ "<a id=\"note-" ~ _mark ~ "\"></a><sup>" + ~ inert(_mark) ~ ".</sup> " + ~ inlineText!()(m["note"].to!string, obj) ~ _back; + } + /+ ↓ and a blank line after the last of them, or the next object begins on the + same line and markdown reads the two as one paragraph +/ + return (_out.data.length > 0) ? _out.data ~ newlines : ""; + } + /+ ↓ the object a table of contents entry points at, where that is an object + number +/ + string internalTarget(O)(const O obj) { + foreach (m; obj.text.matchAll(rgx.any_internal_target)) { + string _frag = m["frag"].to!string; + if (isObjectNumber(_frag)) { return _frag; } + } + return ""; + } + /+ ↓ one object, and then the notes it referred to. + . + The notes are gathered here rather than inside each kind, because a + reference can be anywhere - the markup manual has one in a heading - and a + reference whose note was never written is a link to nothing. + +/ + string object(O,M)(const O obj, M doc_matters, size_t[int] toc_depth) { + return objectBody!()(obj, doc_matters, toc_depth) ~ notes!()(obj); + } + string objectBody(O,M)(const O obj, M doc_matters, size_t[int] toc_depth) { + switch (obj.metainfo.is_a) { + case "heading": + string _hashes; + foreach (_; 0 .. headingLevel(obj.metainfo.marked_up_level.to!string)) { + _hashes ~= "#"; + } + return _hashes ~ " " ~ anchor!()(obj) ~ inlineText!()(obj.text, obj) + ~ ocnMark!()(obj) ~ newlines; + case "toc": + /+ ↓ a list item at its depth, the heading as a link, and the object number + as the locator. The number and not a page: a page is a fact about one + typesetting, and the object number is the same reference in every + output of the document. + +/ + string _indent; + if (auto d = obj.attrib.indent_hang.to!int in toc_depth) { + foreach (_; 0 .. *d) { _indent ~= " "; } + } + string _n = internalTarget!()(obj); + string _locator = (_n.length > 0) + ? (" <sup>[" ~ _n ~ "](#" ~ _n ~ ")</sup>") : ""; + return _indent ~ "- " ~ inlineText!()(obj.text, obj) ~ _locator ~ newline; + case "table": + return table!()(obj, doc_matters); + case "code": + /+ ↓ a fenced block, long enough that the code cannot close it +/ + string _txt = obj.text.replaceAll(rgx.nbsp_char, " "); + string _fence = "```"; + while (_txt.canFind(_fence)) { _fence ~= "`"; } + return anchor!()(obj) ~ newline ~ _fence ~ obj.metainfo.syntax.to!string + ~ newline ~ _txt ~ newline ~ _fence ~ newline ~ ocnMark!()(obj) ~ newlines; + case "quote": + return "> " ~ anchor!()(obj) ~ inlineText!()(obj.text, obj) + ~ ocnMark!()(obj) ~ newlines; + case "verse": case "poem": case "group": case "block": + /+ ↓ every line kept, with CommonMark's own hard line break - a backslash at + the end of a line - rather than the two trailing spaces an editor will + strip +/ + auto _rows = inlineText!()(obj.text, obj).asplit(newline); + auto _out = appender!string; + _out ~= anchor!()(obj); + foreach (i, row; _rows) { + _out ~= row; + if (i + 1 < _rows.length) { _out ~= "\\" ~ newline; } + } + return _out.data ~ ocnMark!()(obj) ~ newlines; + case "bookindex": + /+ ↓ a list item, and every object it names a link +/ + string _txt = obj.text.replaceAll(rgx.trailing_backslash, ""); + return "- " ~ anchor!()(obj) ~ inlineText!()(_txt, obj) + ~ ocnMark!()(obj) ~ newlines; + case "para": case "blurb": case "glossary": case "bibliography": + string _bullet = obj.attrib.bullet ? "- " : ""; + string _indent; + foreach (_; 0 .. obj.attrib.indent_base) { _indent ~= " "; } + return _indent ~ _bullet ~ anchor!()(obj) ~ inlineText!()(obj.text, obj) + ~ ocnMark!()(obj) ~ newlines; + default: + return ""; + } + } + /+ ↓ a pipe table, which needs a header row: markdown has no table without one, + so a table the document did not give a header gets an empty row and keeps + all of its own. + +/ + string table(O,M)(const O obj, M doc_matters) { + auto _rows = obj.text.split(rgx.table_delimiter_row); + int _cols = max(obj.table.number_of_columns.to!int, 1); + string[][] _body; + foreach (row; _rows) { + if (row.strip.length == 0) { continue; } + string[] _cells; + foreach (cell; row.split(rgx.table_delimiter_col)) { + _cells ~= inlineText!()(cell, obj).replace("|", "\\|"); + } + _body ~= _cells; + } + string rule() { + auto r = appender!string; + r ~= "|"; + foreach (i; 0 .. _cols) { + string _a = (i < obj.table.column_aligns.length) + ? obj.table.column_aligns[i].to!string : "l"; + r ~= (_a == "c") ? " :---: |" : (_a == "r") ? " ---: |" : " :--- |"; + } + return r.data ~ newline; + } + string line(string[] cells) { + auto l = appender!string; + l ~= "|"; + foreach (i; 0 .. _cols) { + l ~= " " ~ ((i < cells.length) ? cells[i] : "") ~ " |"; + } + return l.data ~ newline; + } + auto _out = appender!string; + _out ~= anchor!()(obj) ~ newline; + if (_body.length > 0 && obj.table.heading) { + _out ~= line(_body[0]) ~ rule(); + foreach (row; _body[1 .. $]) { _out ~= line(row); } + } else { + _out ~= line([]) ~ rule(); + foreach (row; _body) { _out ~= line(row); } + } + return _out.data ~ ocnMark!()(obj) ~ newlines; + } + string markdownBody(D,M)(const D doc_abstraction, M doc_matters) { + auto _toc_depth = tocDepths!()(doc_abstraction, doc_matters); + auto _out = appender!string; + foreach (part; markdownSections!()(doc_matters)) { + foreach (obj; doc_abstraction[part]) { + if (obj.metainfo.is_of_type == "comment") { continue; } + if (obj.metainfo.dummy_heading) { continue; } + if (obj.text.strip.length == 0) { continue; } + _out ~= object!()(obj, doc_matters, _toc_depth); + } + } + return _out.data; + } + void outputMarkdown(D,M)( + const D doc_abstraction, + M doc_matters, + ) { + auto pth_md = spinePathsMarkdown(doc_matters); + try { + if (!exists(pth_md.base_pth)) { (pth_md.base_pth).mkdirRecurse; } + } catch (ErrnoException ex) { + } + if (doc_matters.opt.action.vox_gt_1) { + writeln(" ", pth_md.markdown_file); + } + { + auto f = File(pth_md.markdown_file, "w"); + f.write(markdownHead!()(doc_matters)); + /+ ↓ the image directory, named once, here, out of the paths module that + decided where this file goes +/ + f.write(markdownBody!()(doc_abstraction, doc_matters) + .replace(image_dir_mark, pth_md.images_rel)); + } + /+ ↓ the images the document names, into the shared image directory at the + output root, because markdown names them by path and does not carry them. + . + The same set the html references, and not a copy of it: a file already + there is left alone, so this costs nothing when the html output has + written them and still works when --markdown is asked for on its own. + +/ + if (doc_matters.srcs.image_list.length > 0) { + try { + if (!exists(pth_md.images)) { (pth_md.images).mkdirRecurse; } + foreach (image; doc_matters.srcs.image_list) { + string _in = doc_matters.src.image_dir_path ~ "/" ~ image; + string _out = pth_md.images ~ "/" ~ image; + if (exists(_in) && !exists(_out)) { _in.copy(_out); } + } + } catch (Exception ex) { + } + } + } +} diff --git a/src/sisudoc/outputs/io_out/metadata.d b/src/sisudoc/outputs/io_out/metadata.d index e90ea8f..6b08bd7 100644 --- a/src/sisudoc/outputs/io_out/metadata.d +++ b/src/sisudoc/outputs/io_out/metadata.d @@ -489,6 +489,11 @@ string theme_light_1 = format(q"┃ ~ " ※ seg </a>] " ~ "[<a href=\"../../" ~ pth_epub.internal_base ~ "/" ~ doc_matters.src.filename_base ~ "." ~ doc_matters.src.language ~ ".epub\" class=\"lnkicon\">" ~ " ◆ epub </a>] "; + if (doc_matters.opt.action.markdown) { + metadata_ ~= "[<a href=\"../markdown/" ~ doc_matters.src.filename_base + ~ "." ~ doc_matters.src.language ~ ".md\" class=\"lnkicon\">" + ~ " □ markdown </a>] "; + } if ((doc_matters.opt.action.html_link_pdf) || (doc_matters.opt.action.html_link_pdf_a4)) { metadata_ ~= "[ pdf: <a href=\"../../pdf/" ~ doc_matters.src.filename_base diff --git a/src/sisudoc/outputs/io_out/paths_output.d b/src/sisudoc/outputs/io_out/paths_output.d index df0583c..8c6a573 100644 --- a/src/sisudoc/outputs/io_out/paths_output.d +++ b/src/sisudoc/outputs/io_out/paths_output.d @@ -669,6 +669,60 @@ template spinePathsSQLite() { } } +/+ ↓ where the markdown output goes. + . + <lang>/markdown/<doc>.<lang>.md, which is where the other three document + outputs go: html, epub and text all sit under the language, and markdown is + a document rather than a rendered artefact. (latex, typst and pdf are flat, + being the print path.) Called "markdown" and not "md" because none of its + four siblings is a name shortened to its extension - text is the word where + the extension is .txt, odf the family where it is .odt - and because "md" + beside the language directories reads as an eleventh language. + . + The language stays in the filename, as it does for the epub and the text, + because a .md is the format most likely to be copied out of the tree and the + directory carries the language only while the file stays in it. + . + The images are the shared set at the output root, the ones the html + references, not a copy: two directories down the reference is ../../image/, + exactly as the scroll html's is. ++/ +template spinePathsMarkdown() { + import std.conv; + auto spinePathsMarkdown(M)( + M doc_matters, + ) { + auto out_pth = spineOutPaths!()(doc_matters.output_path, doc_matters.src.language); + struct _PathsStruct { + /+ ↓ named as the epub, the html and the latex are: from the document + filename and the language, not from the document uid. The uid joins the + pod name where the two differ, and every artefact the site links to is + named for the document +/ + string base_filename(string fn_src) { + return fn_src.baseName.stripExtension; + } + string base_pth() { + return (((out_pth.output_base).chainPath("markdown")).asNormalizedPath).array; + } + string markdown_file() { + return ((base_pth.chainPath(base_filename(doc_matters.src.filename) + ~ "." ~ doc_matters.src.language + ~ ".md") + ).asNormalizedPath).array; + } + /+ ↓ the shared image directory at the output root, written by whichever + output gets there first and read by all of them +/ + string images() { + return (((out_pth.output_root).chainPath("image")).asNormalizedPath).array; + } + /+ ↓ how the .md refers to them, from two directories down +/ + string images_rel() { + return "../../image"; + } + } + return _PathsStruct(); + } +} /+ ↓ where the typst output goes. One .typ per document-language, because the paper is an input to the compile rather than a thing compiled in, and the images beside it because a diff --git a/src/sisudoc/outputs/io_out/rgx.d b/src/sisudoc/outputs/io_out/rgx.d index bafd8b6..60e169b 100644 --- a/src/sisudoc/outputs/io_out/rgx.d +++ b/src/sisudoc/outputs/io_out/rgx.d @@ -85,6 +85,11 @@ static template spineRgxOut() { carry their own copy of this in rgx_xhtml may be needed by other writers +/ static line_break = ctRegex!(` [\\]{2}`, "m"); + /+ ↓ the trailing backslash a book index entry ends its line with. The text + and odt writers build this one inline, as regex("\\s*\\\\"), each time they + need it + +/ + static trailing_backslash = ctRegex!(`\s*\\`, "mg"); /+ quotation marks +/ static quotes_open_and_close = ctRegex!(`[“”]`, "mg"); /+ inline markup footnotes endnotes +/ diff --git a/src/sisudoc/spine.d b/src/sisudoc/spine.d index e42a5f1..db6c8ac 100644 --- a/src/sisudoc/spine.d +++ b/src/sisudoc/spine.d @@ -168,6 +168,8 @@ string program_name = "spine"; "latex-header-sty" : false, "light" : false, "manifest" : false, + "markdown" : false, + "md" : false, "hide-ocn" : false, "no-ocn" : false, "ocda-db" : false, @@ -296,6 +298,8 @@ string program_name = "spine"; "latex-header-sty", "latex document header sty files", &opts["latex-header-sty"], "light", "default light theme", &opts["light"], "manifest", "process manifest output", &opts["manifest"], + "markdown", "markdown output (.md, CommonMark)", &opts["markdown"], + "md", "markdown output (--markdown)", &opts["md"], "no-ocn", "object cite numbers", &opts["no-ocn"], "ocda-db", "document abstraction (write ocda.db sqlite file)", &opts["ocda-db"], "ocn-off", "object cite numbers", &opts["ocn-off"], @@ -582,7 +586,7 @@ Verifying: stdout.flush; exit(_loaded.loaded ? 0 : 1); } - enum outTask { source_or_pod, sqlite, sqlite_multi, latex, typst, odt, epub, html_scroll, html_seg, html_stuff, text, skel } + enum outTask { source_or_pod, sqlite, sqlite_multi, latex, typst, markdown, odt, epub, html_scroll, html_seg, html_stuff, text, skel } struct OptActions { @trusted bool allow_downloads() { return opts["allow-downloads"]; @@ -713,6 +717,9 @@ Verifying: @trusted bool latex() { return opts["latex"]; } + @trusted bool markdown() { + return (opts["markdown"] || opts["md"]) ? true : false; + } @trusted bool typst() { return (opts["typst"] || opts["typ"] || opts["pdf"]) ? true : false; } @@ -1074,6 +1081,7 @@ Verifying: if (odt) schedule ~= outTask.odt; if (latex) schedule ~= outTask.latex; if (typst) schedule ~= outTask.typst; + if (markdown) schedule ~= outTask.markdown; if (text) schedule ~= outTask.text; if (skel) schedule ~= outTask.skel; return schedule.sort().uniq; @@ -1092,25 +1100,28 @@ Verifying: || odt || latex || typst + || markdown + || text || manifest || sqlite_discrete || sqlite_delete || sqlite_update - || text || skel ) ? true : false; } @trusted bool require_processing_files() { return ( opts["abstraction"] - || epub || curate || html || html_seg || html_scroll + || epub + || odt || latex || typst - || odt + || markdown + || text || manifest || show_abstraction || ocda_db @@ -1121,7 +1132,6 @@ Verifying: || source_or_pod || sqlite_discrete || sqlite_update - || text || xhtml || skel ) ? true : false; |
