/+ - Name: SisuDoc Spine, Doc Reform [a part of] - Description: documents, structuring, processing, publishing, search - static content generator - Author: Ralph Amissah [ralph.amissah@gmail.com] - Copyright: (C) 2015 (continuously updated, current 2026) Ralph Amissah, All Rights Reserved. - License: AGPL 3 or later: Spine (SiSU), a framework for document structuring, publishing and search Copyright (C) Ralph Amissah This program is free software: you can redistribute it and/or modify it under the terms of the GNU AFERO General Public License as published by the Free Software Foundation, either version 3 of the License, or (at your option) any later version. This program is distributed in the hope that it will be useful, but WITHOUT ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU General Public License for more details. You should have received a copy of the GNU General Public License along with this program. If not, see [https://www.gnu.org/licenses/]. If you have Internet connection, the latest version of the AGPL should be available at these locations: [https://www.fsf.org/licensing/licenses/agpl.html] [https://www.gnu.org/licenses/agpl.html] - Spine (by Doc Reform, related to SiSU) uses standard: - docReform markup syntax - standard SiSU markup syntax with modified headers and minor modifications - docReform object numbering - standard SiSU object citation numbering & system - Homepages: [https://www.sisudoc.org] [https://www.doc-reform.org] - Git [https://git.sisudoc.org/] +/ module sisudoc.sisu_document_parser; /++ name "spine" description "A SiSU inspired document parser written in D." homepage "https://sisudoc.org" +/ @safe: import std.algorithm; import std.datetime; import std.getopt; import std.file; import std.path; import std.process; import sisudoc.outputs.conf.compile_time_info; import sisudoc.ocda.meta; import sisudoc.ocda.meta.metadoc; import sisudoc.outputs.io_out.curate.metadoc_curate; import sisudoc.outputs.io_out.curate.metadoc_curate_authors; import sisudoc.outputs.io_out.curate.metadoc_curate_topics; import sisudoc.ocda.meta.metadoc_from_src; import sisudoc.ocda.meta.conf_make_meta_structs; import sisudoc.ocda.meta.conf_make_meta_json; import sisudoc.ocda.meta.defaults; import sisudoc.ocda.meta.doc_debugs; import sisudoc.ocda.meta.ocn_align; import sisudoc.ocda.meta.rgx; import sisudoc.ocda.meta.rgx_yaml; import sisudoc.ocda.meta.rgx_files; import sisudoc.ocda.io_in.paths_source; import sisudoc.ocda.io_in.read_config_files; import sisudoc.ocda.io_in.read_source_files; import sisudoc.ocda.io_in.read_zip_pod; import sisudoc.outputs.io_out.hub; mixin(import("version.txt")); mixin(import("configuration.txt")); mixin CompileTimeInfo; string project_name = "spine"; string program_name = "spine"; @system void main(string[] args) { mixin spineRgxIn; mixin spineRgxYamlTags; mixin spineRgxFiles; mixin spineBiblio; mixin outputHub; auto hvst = spineCurateMetadata!(); string flag_action; string arg_unrecognized; enum dAM { abstraction, matters } static auto rgx = RgxI(); static auto rgx_y = RgxYaml(); static auto rgx_files = RgxFiles(); /+ ↓ an action whose output is a file's contents, printed to stdout, turns this off: the banner would otherwise be part of what it printed. The round trips do the same by leaving before scope(success) runs, which they can because they leave early; --po4a-cfg prints from inside output processing and so says so here instead. Set where the options are read, below: the banner is declared before they exist. +/ bool _run_banner = true; scope(success) { if (_run_banner) { writefln( "~ run complete, ok ~ (%s-%s.%s.%s, %s D:%s, %s %s)", program_name, _ver.major, _ver.minor, _ver.patch, __VENDOR__, __VERSION__, bits, os, ); } } scope(failure) { debug(checkdoc) { stderr.writefln( "run failure", ); } } bool[string] opts = [ "abstraction" : false, "abstraction-db" : false, "allow-downloads" : false, "assertions" : false, "concordance" : false, "dark" : false, "debug" : false, "debug-curate" : false, "debug-curate-authors" : false, "debug-curate-topics" : false, "debug-epub" : false, "debug-harvest" : false, "debug-html" : false, "debug-latex" : false, "debug-manifest" : false, "debug-metadata" : false, "debug-pod" : false, "debug-sqlite" : false, "debug-stages" : false, "digest" : false, "epub" : false, "generated-by" : false, "curate" : false, "curate-authors" : false, "curate-topics" : false, "html" : false, "html-link-curate" : false, "html-link-markup" : false, "html-link-pdf" : false, "html-link-pdf-a4" : false, "html-link-pdf-letter" : false, "html-link-search" : false, "html-link-text" : false, "html-seg" : false, "html-scroll" : false, "latex" : false, "latex-color-links" : false, "latex-init" : false, "latex-header-sty" : false, "light" : false, "manifest" : false, "hide-ocn" : false, "no-ocn" : false, "ocda-db" : false, "ocn-off" : false, "odf" : false, "odt" : false, "parallel" : false, "parallel-subprocesses" : false, "pdf" : false, "pdf-color-links" : false, "pdf-init" : false, "po4a-cfg" : false, "pod" : false, "pod2" : false, "serial" : false, "show-abstraction" : false, "show-config" : false, "show-curate" : false, "show-curate-authors" : false, "show-curate-topics" : false, "show-epub" : false, "show-html" : false, "show-latex" : false, "show-make" : false, "show-manifest" : false, "show-metadata" : false, "show-pod" : false, "show-sqlite" : false, "show-summary" : false, "skel" : false, "source" : false, "sqlite-discrete" : false, "sqlite-db-create" : false, "sqlite-db-drop" : false, "sqlite-db-recreate" : false, "sqlite-delete" : false, "sqlite-insert" : false, "sqlite-update" : false, "text" : false, "vox_is0" : false, // silent "vox_is1" : false, // quiet "vox_is2" : false, // default (unset) "vox_is3" : false, // verbose "vox_is4" : false, // very verbose "xhtml" : false, "section_toc" : true, "section_body" : true, "section_endnotes" : true, "section_glossary" : true, "section_biblio" : true, "section_bookindex" : true, "section_blurb" : true, "backmatter" : true, "no-verify" : false, "skip-output" : false, "strict" : false, "theme-dark" : false, "theme-light" : false, "workon" : false, ]; string[string] settings = [ "output" : "", "pod-compression" : "", "ssp-round-trip" : "", "db-round-trip" : "", "ocda-verify" : "", "abstraction-source" : "", "www-http" : "", "www-host" : "", "www-host-doc-root" : "", "www-url-doc-root" : "", "cgi-http" : "", "cgi-host" : "", "cgi-bin-root" : "", "cgi-sqlite-search-filename" : "", "cgi-url-root" : "", "cgi-url-action" : "", "cgi-search-title" : "", "config" : "", "lang" : "all", "set-papersize" : "", "set-textwrap" : "", "set-digest" : "", "sqlite-db-path" : "", "sqlite-db-filename" : "", ]; auto helpInfo = getopt(args, std.getopt.config.passThrough, "abstraction", "document abstraction", &opts["abstraction"], "abstraction-db", "document abstraction (write ocda.db sqlite file)", &opts["abstraction-db"], "allow-downloads", "allow a url argument to be fetched (zip or db)", &opts["allow-downloads"], "assert", "set optional assertions on", &opts["assertions"], "cgi-bin-root", "path to cgi-bin directory", &settings["cgi-bin-root"], "cgi-url-root", "url to cgi-bin (to find cgi-bin)", &settings["cgi-url-root"], "cgi-url-action", "url to post to cgi-bin search form", &settings["cgi-url-action"], "cgi-search-title", "if generating a cgi search form the title to use for it", &settings["cgi-search-title"], "cgi-sqlite-search-filename", "=[filename] default is spine-search", &settings["cgi-sqlite-search-filename"], "concordance", "file for document", &opts["concordance"], "curate", "extract info on authors & topics from document header metadata", &opts["curate"], "curate-authors", "extract info on authors from document header metadata", &opts["curate-authors"], "curate-topics", "extract info on topics from document header metadata", &opts["curate-topics"], "dark", "alternative dark theme", &opts["dark"], "digest", "hash digest for each object", &opts["digest"], "epub", "process epub output", &opts["epub"], "generated-by", "generated by headers (software version & time)", &opts["generated-by"], "hide-ocn", "object cite numbers", &opts["hide-ocn"], "html", "process html output", &opts["html"], "html-link-curate", "place links back to curate in segmented html", &opts["html-link-curate"], "html-link-markup", "provide html link to markup source, shared optionally", &opts["html-link-markup"], "html-link-pdf", "provide html link to pdf a4 & letter output", &opts["html-link-pdf"], "html-link-pdf-a4", "provide html link to pdf a4 output", &opts["html-link-pdf-a4"], "html-link-pdf-letter", "provide html link to pdf letter size output", &opts["html-link-pdf-letter"], "html-link-text", "provide html link to text output", &opts["html-link-text"], "html-link-search", "html embedded search submission", &opts["html-link-search"], "html-seg", "process html output", &opts["html-seg"], "html-scroll", "process html output", &opts["html-scroll"], "lang", "=[lang code e.g. =en or =en,es]", &settings["lang"], "latex", "latex output (for pdfs)", &opts["latex"], "latex-color-links", "mono or color links for pdfs", &opts["latex-color-links"], "latex-init", "initialise latex shared files (see latex-header-sty)", &opts["latex-init"], "latex-header-sty", "latex document header sty files", &opts["latex-header-sty"], "light", "default light theme", &opts["light"], "manifest", "process manifest output", &opts["manifest"], "no-ocn", "object cite numbers", &opts["no-ocn"], "ocda-db", "document abstraction (write ocda.db sqlite file)", &opts["ocda-db"], "ocn-off", "object cite numbers", &opts["ocn-off"], "odf", "open document format text (--odt)", &opts["odf"], "odt", "open document format text", &opts["odt"], "output", "=/path/to/output/dir specify where to place output", &settings["output"], "ssp-round-trip", "=/path/to/file.ssp read a .ssp back and re-emit it on stdout", &settings["ssp-round-trip"], "db-round-trip", "=/path/to/file.ocda.db read it back and emit .ssp on stdout", &settings["db-round-trip"], "ocda-verify", "=/path/to/file.ocda.db does this spine still make this abstraction from the markup it carries", &settings["ocda-verify"], "no-verify", "do not check carried markup against the abstraction it was built with", &opts["no-verify"], "abstraction-source", "=/path/to/(.sst|pod|.ssp|.ocda.db) identify it, and load it if it is an abstraction", &settings["abstraction-source"], "parallel", "parallelise document processing (opt-in; slower than serial)", &opts["parallel"], "parallel-subprocesses", "nested parallelisation", &opts["parallel-subprocesses"], "pdf", "latex output for pdfs", &opts["pdf"], "pdf-color-links", "mono or color links for pdfs", &opts["pdf-color-links"], "pdf-init", "initialise latex shared files (see latex-header-sty)", &opts["pdf-init"], "po4a-cfg", "print the po4a configuration for this document", &opts["po4a-cfg"], "pod", "spine (doc reform) pod source content bundled", &opts["pod"], "pod2", "pod with document abstraction (.ssp) bundled", &opts["pod2"], "quiet|q", "output to terminal", &opts["vox_is1"], "section-backmatter", "document backmatter (default)" , &opts["backmatter"], "section-biblio", "document biblio (default)", &opts["section_biblio"], "section-blurb", "document blurb (default)", &opts["section_blurb"], "section-body", "document body (default)", &opts["section_body"], "section-bookindex", "document bookindex (default)", &opts["section_bookindex"], "section-endnotes", "document endnotes (default)", &opts["section_endnotes"], "section-glossary", "document glossary (default)", &opts["section_glossary"], "section-toc", "table of contents (default)", &opts["section_toc"], "serial", "serial document processing (default)", &opts["serial"], "skip-output", "skip output", &opts["skip-output"], "strict", "document checks that warn become failures", &opts["strict"], "show-abstraction", "show document abstraction (write .ssp file)", &opts["show-abstraction"], "show-config", "show config", &opts["show-config"], "show-curate", "show curate", &opts["show-curate"], "show-curate-authors", "show curate authors", &opts["show-curate-authors"], "show-curate-topics", "show curate topics", &opts["show-curate-topics"], "show-epub", "show epub", &opts["show-epub"], "show-html", "show html", &opts["show-html"], "show-latex", "show latex", &opts["show-latex"], "show-make", "show make", &opts["show-make"], "show-manifest", "show manifest", &opts["show-manifest"], "show-metadata", "show metadata", &opts["show-metadata"], "show-pod", "show pod", &opts["show-pod"], "show-sqlite", "show sqlite", &opts["show-sqlite"], "show-summary", "show summary", &opts["show-summary"], "source", "document markup source", &opts["source"], "silent", "output to terminal", &opts["vox_is0"], "set-digest", "default hash digest type (e.g. sha256)", &settings["set-digest"], "pod-compression", "pod archive compression level, zstd 1 to 19 (default 19)", &settings["pod-compression"], "set-papersize", "default papersize (latex pdf eg. a4 or a5 or b4 or letter)", &settings["set-papersize"], "set-textwrap", "default textwrap (e.g. 80 (characters)", &settings["set-textwrap"], "skel", "skel (dummy outline)", &opts["skel"], "sqlite-discrete", "process discrete sqlite output", &opts["sqlite-discrete"], "sqlite-db-create", "create db, create tables", &opts["sqlite-db-create"], "sqlite-db-drop", "drop tables & db", &opts["sqlite-db-drop"], "sqlite-db-filename", "sqlite db to create, populate & make available for search", &settings["sqlite-db-filename"], "sqlite-db-path", "sqlite db path", &settings["sqlite-db-path"], "sqlite-db-recreate", "create db, create tables", &opts["sqlite-db-recreate"], "sqlite-delete", "sqlite output", &opts["sqlite-delete"], "sqlite-insert", "sqlite output", &opts["sqlite-insert"], "sqlite-update", "sqlite output", &opts["sqlite-update"], "text", "text output", &opts["text"], "txt", "text output", &opts["text"], "www-http", "http or https", &settings["www-http"], "www-host", "web server host (domain) name", &settings["www-host"], "www-host-doc-root", "web host host (domain) name with path to doc root", &settings["www-host-doc-root"], "www-url-doc-root", "e.g. http://localhost", &settings["www-url-doc-root"], "theme-dark", "alternative dark theme", &opts["theme-dark"], "theme-light", "default light theme", &opts["theme-light"], "verbose|v", "output to terminal", &opts["vox_is3"], "very-verbose", "output to terminal", &opts["vox_is4"], "workon", "(reserved for some matters under development & testing)", &opts["workon"], "xhtml", "xhtml output", &opts["xhtml"], "config", "=/path/to/config/file/including/filename", &settings["config"], "debug", "debug", &opts["debug"], "debug-curate", "debug curate", &opts["debug-curate"], "debug-curate-authors", "debug curate authors", &opts["debug-curate-authors"], "debug-curate-topics", "debug curate topics", &opts["debug-curate-topics"], "debug-epub", "debug epub", &opts["debug-epub"], "debug-harvest", "debug harvest", &opts["debug-harvest"], "debug-html", "debug html", &opts["debug-html"], "debug-latex", "debug latex", &opts["debug-latex"], "debug-manifest", "debug manifest", &opts["debug-manifest"], "debug-metadata", "debug metadata", &opts["debug-metadata"], "debug-pod", "debug pod", &opts["debug-pod"], "debug-sqlite", "debug sqlite", &opts["debug-sqlite"], "debug-stages", "debug stages", &opts["debug-stages"], // "sqlite-db-filename", "=[filename].sql.db", &settings["sqlite-db-filename"], ); /+ ↓ --po4a-cfg prints a file's contents; see the banner above +/ if (opts["po4a-cfg"]) { _run_banner = false; } if (helpInfo.helpWanted) { defaultGetoptPrinter( "spine: structure, parse, publish and search document collections.\n" ~ "\n" ~ " spine [options] ...\n" ~ "\n" ~ "A is markup, or an artefact made from it:\n" ~ "\n" ~ " .sst, .ssm markup, one document\n" ~ " / a pod: markup, images, conf, manifest\n" ~ " .sisupod the same, zipped (.zip also read)\n" ~ " .ssp the abstraction, as text\n" ~ " .ocda.db the abstraction, as sqlite\n" ~ "\n" ~ "A url is fetched only with --allow-downloads.\n", helpInfo.options ); /+ ↓ the contract, after the options: what an artefact carries and what can be done with it. It belongs in --help because it is the answer to "I have this file, what can spine do with it", which is the question somebody holding one actually has. +/ writeln(q"┃ The artefacts, and what they carry: .ssp the abstraction as text, one file per language. Describes its images by name, size and digest; does not carry them, and does not carry the markup. .ocda.db the abstraction as sqlite, one file per document holding EVERY language of it, and carrying the markup it was built from, the images, conf/document_make, pod.manifest and the translation catalogues. A document source in its own right: a pod can be written back out of it. Both record the digest of the markup they were built from, so what they came from is checkable. A .ocda.db also records the digest of the .ssp, so the chain .sst -> .ssp -> .ocda.db is checkable end to end. Actions that need the markup rather than only the abstraction: --source --pod --pod2 --show-abstraction --ocda-db Given a .ocda.db that carries markup, these write the pod back out and build from it, so the document is made by the same code over the same bytes as the original. Given a .ssp, which carries no markup, they are refused and the rest of the run goes on. Verifying: --ocda-verify= re-parse the markup the file carries and ask whether it still makes the abstraction the file holds. Writes nothing; the exit status is the answer. --no-verify build anyway when it does not, with a warning. The check is otherwise automatic and runs before anything is written. --strict document checks that warn become failures, ocn alignment between languages among them. ┃"); /+ ↓ --help is a question answered on stdout, not a run: the scope(success) banner would be the last line of the answer +/ _run_banner = false; } /+ ↓ read a .ssp back into the abstraction and emit it again, on stdout. the two should be byte identical: that is the check that the reader is faithful, and it is made against the writer's own record definition, not against a second description of the format +/ if (settings["ssp-round-trip"].length > 0) { import sisudoc.ocda.abstraction.ssp; import sisudoc.ocda.abstraction.ssp_in; mixin spineAbstractionRead; mixin sspObjectRecord; auto _doc = sspReadFile(settings["ssp-round-trip"]); string[] _out; if (_doc.format.length > 0) { _out ~= _doc.format; } if (_doc.source.length > 0) { _out ~= "% Source: " ~ _doc.source; } _out ~= ""; void _block(string _name, string[] _keys, string[string] _kv) { _out ~= "@" ~ _name ~ " {"; foreach (k; _keys) { if (k in _kv) { _out ~= " " ~ k ~ ": " ~ _kv[k]; } } _out ~= "}"; _out ~= ""; } _block("source", _doc.source_info_order, _doc.source_info); _block("meta", _doc.meta_order, _doc.meta); _block("make", _doc.make_order, _doc.make); _block("doc_has", _doc.doc_has_order, _doc.doc_has); foreach (section; _doc.section_order) { _out ~= "@" ~ section ~ " {"; _out ~= ""; foreach (obj; _doc.abstraction[section]) { _out ~= sspObjectRecord(obj, section); } _out ~= "}"; _out ~= ""; } foreach (line; _out) { writeln(line); } /+ ↓ the document goes to stdout, so the run-complete banner that scope(success) prints would be part of it; leave without it +/ import core.stdc.stdlib : exit; stdout.flush; exit(0); } /+ ↓ read a .ocda.db back into the abstraction and emit it as .ssp on stdout. compared against the .ssp for the same document, this says whether the two artefacts really do carry the same thing +/ if (settings["db-round-trip"].length > 0) { import sisudoc.ocda.abstraction.ssp; import sisudoc.ocda.abstraction.db_in; mixin spineAbstractionDbRead; mixin sspObjectRecord; /+ ↓ a database now holds every language of its document, so a language has to be named. --lang says which; with none named dbReadFile if there is only one, takes the only document there is, and where there is a choice reports the languages and stops rather than picking one. A .ssp emitted for a language nobody asked for would match nothing, and this is what the round trip test compares byte for byte. +/ string _rt_lang = ""; { auto _langs = settings["lang"].split(","); if (_langs.length > 0 && _langs[0] != "all") { _rt_lang = _langs[0]; } } auto _doc = dbReadFile(settings["db-round-trip"], _rt_lang); if (_doc.format.length == 0 && _doc.section_order.length == 0) { import core.stdc.stdlib : exit; stdout.flush; exit(1); } string[] _out; if (_doc.format.length > 0) { _out ~= _doc.format; } if (_doc.source.length > 0) { _out ~= "% Source: " ~ _doc.source; } _out ~= ""; void _blockdb(string _name, string[] _keys, string[string] _kv) { _out ~= "@" ~ _name ~ " {"; foreach (k; _keys) { if (k in _kv) { _out ~= " " ~ k ~ ": " ~ _kv[k]; } } _out ~= "}"; _out ~= ""; } _blockdb("source", _doc.source_info_order, _doc.source_info); _blockdb("meta", _doc.meta_order, _doc.meta); _blockdb("make", _doc.make_order, _doc.make); _blockdb("doc_has", _doc.doc_has_order, _doc.doc_has); foreach (section; _doc.section_order) { _out ~= "@" ~ section ~ " {"; _out ~= ""; foreach (obj; _doc.abstraction[section]) { _out ~= sspObjectRecord(obj, section); } _out ~= "}"; _out ~= ""; } foreach (line; _out) { writeln(line); } import core.stdc.stdlib : exit; stdout.flush; exit(0); } /+ ↓ say what a path is and, where it is an abstraction artefact, load it and report what came back. one way in for any of the sources spine knows, which is what the loader package is for. exit 0 when an abstraction was loaded, 1 when it was not, with the reason given: an unknown path, or a source form that goes through the parser rather than through a file read +/ if (settings["abstraction-source"].length > 0) { import sisudoc.ocda.abstraction.load; mixin spineAbstractionLoad; auto _loaded = abstractionLoad(settings["abstraction-source"]); foreach (line; abstractionLoadSummary(_loaded)) { writeln(line); } import core.stdc.stdlib : exit; stdout.flush; exit(_loaded.loaded ? 0 : 1); } enum outTask { source_or_pod, sqlite, sqlite_multi, latex, odt, epub, html_scroll, html_seg, html_stuff, text, skel } struct OptActions { @trusted bool allow_downloads() { return opts["allow-downloads"]; } @trusted bool assertions() { return opts["assertions"]; } @trusted bool concordance() { return opts["concordance"]; } @trusted string config_path_set() { return settings["config"]; } @trusted bool css_theme_default() { bool _is_light; if (opts["light"] || opts["theme-light"]) { _is_light = true; } else if (opts["dark"] || opts["theme-dark"]) { _is_light = false; } else { _is_light = true; } return _is_light; } @trusted bool debug_do() { bool _dbg; if (opts["debug"]) { _dbg = true; } else { _dbg = false; } return _dbg; } @trusted bool debug_do_curate() { return (opts["debug"] || opts["debug-curate"]) ? true : false; } @trusted bool debug_do_curate_authors() { return (opts["debug"] || opts["debug-curate"] || opts["debug-curate-authors"]) ? true : false; } @trusted bool debug_do_curate_topics() { return (opts["debug"] || opts["debug-curate"] || opts["debug-curate-topics"]) ? true : false; } @trusted bool debug_do_epub() { return (opts["debug"] || opts["debug-epub"]) ? true : false; } @trusted bool debug_do_harvest() { return (opts["debug"] || opts["debug-harvest"]) ? true : false; } @trusted bool debug_do_html() { return (opts["debug"] || opts["debug-html"]) ? true : false; } @trusted bool debug_do_latex() { return (opts["debug"] || opts["debug-latex"]) ? true : false; } @trusted bool debug_do_manifest() { return (opts["debug"] || opts["debug-manifest"]) ? true : false; } @trusted bool debug_do_metadata() { return (opts["debug"] || opts["debug-metadata"]) ? true : false; } @trusted bool debug_do_pod() { return (opts["debug"] || opts["debug-pod"]) ? true : false; } @trusted bool debug_do_sqlite() { return (opts["debug"] || opts["debug-sqlite"]) ? true : false; } @trusted bool debug_do_stages() { return (opts["debug"] || opts["debug-stages"]) ? true : false; } @trusted bool debug_do_xmls() { return (opts["debug"] || opts["debug-html"] || opts["debug-epub"]) ? true : false; } @trusted bool curate() { return (opts["curate"] || opts["curate-authors"] || opts["curate-topics"]) ? true : false; } @trusted bool curate_authors() { return (opts["curate"] || opts["curate-authors"]) ? true : false; } @trusted bool curate_topics() { return (opts["curate"] || opts["curate-topics"]) ? true : false; } @trusted bool digest() { return opts["digest"]; } @trusted bool epub() { return opts["epub"]; } @trusted bool generated_by() { return opts["generated-by"]; } @trusted bool html_link_curate() { return (opts["html-link-curate"]) ? true : false; } @trusted bool html_link_markup_source() { return (opts["html-link-markup"]) ? true : false; } @trusted bool html_link_pdf() { return (opts["html-link-pdf"]) ? true : false; } @trusted bool html_link_pdf_a4() { return (opts["html-link-pdf-a4"]) ? true : false; } @trusted bool html_link_pdf_letter() { return (opts["html-link-pdf-letter"]) ? true : false; } @trusted bool html_link_text() { return (opts["html-link-text"]) ? true : false; } @trusted bool text_link_curate() { return (opts["text-link-curate"]) ? true : false; } @trusted bool html_link_search() { return (opts["html-link-search"]) ? true : false; } @trusted bool html() { return (opts["html"] || opts["html-seg"] || opts["html-scroll"]) ? true : false; } @trusted bool html_seg() { return (opts["html"] || opts["html-seg"]) ? true : false; } @trusted bool html_scroll() { return (opts["html"] || opts["html-scroll"]) ? true : false; } @trusted bool html_stuff() { return (opts["html"] || opts["html-scroll"] || opts["html-seg"]) ? true : false; } @trusted bool latex() { return (opts["latex"] || opts["pdf"]) ? true : false; } @trusted bool latex_color_links() { return (opts["latex-color-links"] || opts["pdf-color-links"]) ? true : false; } @trusted bool latex_document_header_sty() { return (opts["latex-init"] || opts["latex-header-sty"] || opts["pdf-init"]) ? true : false; } @trusted bool odt() { return (opts["odf"] || opts["odt"]) ? true : false; } @trusted bool manifest() { return opts["manifest"]; } @trusted bool ocda_db() { return (opts["abstraction-db"] || opts["ocda-db"]) ? true : false; } @trusted bool ocn_hidden() { return opts["hide-ocn"]; } @trusted bool ocn_off() { return ((opts["ocn-off"]) || (opts["no-ocn"])) ? true : false; } @trusted bool po4a_cfg() { return opts["po4a-cfg"]; } @trusted bool pod() { return (opts["pod"] || opts["pod2"]) ? true : false; } @trusted bool pod2() { return opts["pod2"]; } @trusted bool show_config() { return opts["show-config"]; } @trusted bool show_abstraction() { return (opts["show-abstraction"] || pod2) ? true : false; } @trusted bool show_curate() { return opts["show-curate"]; } @trusted bool show_curate_authors() { return (opts["show-curate"] || opts["show-curate-authors"] || vox_gt_2) ? true : false; } @trusted bool show_curate_topics() { return (opts["show-curate"] || opts["show-curate-topics"] || vox_gt_3) ? true : false; } @trusted bool show_epub() { return opts["show-epub"]; } @trusted bool show_html() { return opts["show-html"]; } @trusted bool show_latex() { return opts["show-latex"]; } @trusted bool show_make() { return opts["show-make"]; } @trusted bool show_manifest() { return opts["show-manifest"]; } @trusted bool show_metadata() { return opts["show-metadata"]; } @trusted bool show_pod() { return opts["show-pod"]; } @trusted bool show_sqlite() { return (opts["show-sqlite"] || vox_gt_3) ? true : false; } @trusted bool show_summary() { return (opts["show-summary"] || vox_gt_2) ? true : false; } @trusted bool source() { return opts["source"]; } @trusted bool source_or_pod() { return (opts["pod"] || opts["pod2"] || opts["source"]) ? true : false; } /+ ↓ the actions that need the markup source and not only the abstraction. . --source, --pod and --pod2 write the markup out. --show-abstraction and --ocda-db write an artefact, and an artefact is a statement about the source: if this were derived from a loaded abstraction it would say only that the loader is consistent with itself, where derived from the markup it says what this spine makes of that document today, which is the question posited by --ocda-verify. . Given a database that carries source, these route the argument through the pod: it is written back out and parsed. Given one that does not, or a .ssp, which carries no markup at all, they are refused, and any rendering asked for in the same run still happens from the abstraction. +/ @trusted bool needs_markup_source() { return (source_or_pod || show_abstraction || ocda_db || ocda_verify) ? true : false; } /+ ↓ verify: does this spine still make this abstraction from the markup the artefact carries? . A .ssp or a database states an abstraction and, since a .ocda.db carries the markup it was built from; re-parsing the one and comparing against the other is the only check that the two still agree, and it is the check that says whether a document travels through time as well as through a file. . --ocda-verify names a file and asks only that. It is an action in its own right: no output is written, and the run's exit status is the answer. +/ @trusted bool ocda_verify() { return (settings["ocda-verify"].length > 0); } @trusted string ocda_verify_file() { return settings["ocda-verify"]; } /+ ↓ the escape, for the case it is actually wanted: a newer spine reading an older artefact, where the abstraction is expected to differ and the output is wanted anyway. It warns. +/ @trusted bool no_verify() { return opts["no-verify"]; } @trusted bool sqlite_discrete() { return opts["sqlite-discrete"]; } @trusted bool sqlite_db_drop() { return (opts["sqlite-db-recreate"] || opts["sqlite-db-drop"]) ? true : false; } @trusted bool sqlite_db_create() { return (opts["sqlite-db-recreate"] || opts["sqlite-db-create"]) ? true : false; } @trusted bool sqlite_delete() { return opts["sqlite-delete"]; } @trusted bool sqlite_update() { return (opts["sqlite-update"] || opts["sqlite-insert"]) ? true : false; } @trusted bool sqlite_shared_db_action() { return ( opts["sqlite-db-recreate"] || opts["sqlite-db-create"] || opts["sqlite-delete"] || opts["sqlite-insert"] || opts["sqlite-update"] ) ? true : false; } @trusted bool skel() { return opts["skel"]; } @trusted bool text() { return opts["text"]; } @trusted bool vox_0() { // --silent return opts["vox_is0"]; } @trusted bool vox_1() { // --quiet -q return opts["vox_is1"]; } @trusted bool vox_2() { // normal, minimal, without flag bool _vox_default = true; if (opts["vox_is0"] || opts["vox_is1"] || opts["vox_is3"] || opts["vox_is4"]) { _vox_default = false; } else { _vox_default = true; } return _vox_default; } @trusted bool vox_3() { // --verbose -v return opts["vox_is3"]; } @trusted bool vox_4() { // --very-verbose return opts["vox_is4"]; } @trusted bool vox_gt_0() { // --quiet -q and above return ( vox_1 || vox_2 || vox_3 || vox_4) ? true : false; } @trusted bool vox_gt_1() { // normal, and above return (vox_2 || vox_3 || vox_4) ? true : false; } @trusted bool vox_gt_2() { // --verbose -v and above return ( vox_3 || vox_4) ? true : false; } @trusted bool vox_gt_3() { // --very-verbose return (vox_4) ? true : false; } @trusted bool vox_silent() { return vox_0; } // --silent @trusted bool vox_quiet() { return vox_gt_0; } // --quiet -q & above @trusted bool vox_default() { return vox_gt_1; } // defalt, & above @trusted bool vox_verbose() { return vox_gt_2; } // --verbose -v & above @trusted bool vox_very_verbose() { return vox_gt_3; } // --very-verbose @trusted bool xhtml() { return opts["xhtml"]; } @trusted bool section_toc() { return opts["section_toc"]; } @trusted bool section_body() { return opts["section_body"]; } @trusted bool section_endnotes() { return opts["section_endnotes"]; } @trusted bool section_glossary() { return opts["section_glossary"]; } @trusted bool section_biblio() { return opts["section_biblio"]; } @trusted bool section_bookindex() { return opts["section_bookindex"]; } @trusted bool section_blurb() { return opts["section_blurb"]; } @trusted bool backmatter() { return opts["backmatter"]; } @trusted bool skip_output() { /+ ↓ --ocda-verify answers a question and writes nothing: the run's exit status is the answer, and a check that left output behind could not be run against a published tree +/ return (opts["skip-output"] || ocda_verify) ? true : false; } @trusted bool strict() { return opts["strict"]; } @trusted bool workon() { return opts["workon"]; } @trusted string[] languages_set() { return settings["lang"].split(","); } @trusted string output_dir_set() { return settings["output"]; } @trusted string sqliteDB_filename() { return settings["sqlite-db-filename"]; } @trusted string sqliteDB_path() { return settings["sqlite-db-path"]; } @trusted string cgi_bin_root() { return settings["cgi-bin-root"]; } @trusted string cgi_search_title() { return settings["cgi-search-title"]; } @trusted string cgi_sqlite_search_filename() { return settings["cgi-sqlite-search-filename"]; } @trusted string cgi_sqlite_search_filename_d() { return (settings["cgi-sqlite-search-filename"].length > 0) ? (settings["cgi-sqlite-search-filename"].translate(['-' : "_"]) ~ ".d") : ""; } @trusted string cgi_url_root() { return settings["cgi-url-root"]; } @trusted string cgi_url_action() { return settings["cgi-url-action"]; } @trusted string hash_digest_type() { return settings["set-digest"]; } @trusted string text_wrap() { return settings["set-textwrap"]; } /+ ↓ the zstd level a pod archive is written at. 19 by default: a pod is written once and fetched many times, and there is a significant gain (live-manual 675970 bytes at 19 against 786683 at 9). Lower it where build time matters more than the bytes do. +/ @trusted int pod_compression() { import std.conv : to; if (settings["pod-compression"].length > 0) { try { return settings["pod-compression"].to!int; } catch (Exception) { writeln("WARNING --pod-compression is not a number, using 19: ", settings["pod-compression"]); } } return 19; } @trusted string latex_papersize() { return settings["set-papersize"]; } @trusted string webserver_host_name() { return settings["www-host"]; } @trusted string webserver_host_doc_root() { return settings["www-host-doc-root"]; } @trusted string webserver_url_doc_root() { return settings["www-url-doc-root"]; } @trusted string webserver_http() { return settings["www-http"]; } /+ ↓ serial is the behaviour, --parallel is the option. . In-process parallelism costs rather than pays. Measured on 35 sample documents, 16 cores, --text --html --epub --latex: 14.42s wall and 154s cpu in parallel against 10.57s wall and 12.9s cpu serially. The abstraction stage on its own is the same shape, 4.8s against 4.0s for eleven times the cpu. The cost tracks the number of cores made available rather than the work done - 3.7s cpu pinned to one core, 46.8s on sixteen, for identical output - which is threads burning time without progressing. Separate processes over the same documents scale as expected (4.2s to 1.3s, 16 processes, output byte identical), so it is not the work and not the thread count: it is what the threads share, most likely the GC allocation lock, which is spun rather than slept on. Not confirmed with a profiler. . So --parallel no longer follows from asking for output. It is kept, not removed: test-abstraction-ssp.sh compares a parallel run against a serial one, and that comparison is what guards the .ssp write race. An instrument, until the sharing is understood and fixed. +/ @trusted bool parallelise() { bool _is; if (opts["serial"] == true) { _is = false; } else if ( /+ ↓ these cannot run in parallel however asked: --curate aggregates across documents, and the shared sqlite db has the one writer . --ocda-db joins them because one database now holds every language of a document, written a language at a time. Several languages at once against one file, one of them clearing it and each opening its own transaction, is a race, and the losing language would be missing from the artefact. +/ sqlite_shared_db_action || ocda_db || source_or_pod ) { _is = false; } else if (opts["parallel"] == true) { _is = true; } else { _is = false; } return _is; } @trusted bool parallelise_subprocesses() { return opts["parallel-subprocesses"]; } auto output_task_scheduler() { int[] schedule; if (source_or_pod) schedule ~= outTask.source_or_pod; if (sqlite_discrete) schedule ~= outTask.sqlite; if (epub) schedule ~= outTask.epub; if (html_scroll) schedule ~= outTask.html_scroll; if (html_seg) schedule ~= outTask.html_seg; if (html_stuff) schedule ~= outTask.html_stuff; if (odt) schedule ~= outTask.odt; if (latex) schedule ~= outTask.latex; if (text) schedule ~= outTask.text; if (skel) schedule ~= outTask.skel; return schedule.sort().uniq; } @trusted bool abstraction() { return ( opts["abstraction"] || show_abstraction || ocda_db || ocda_verify || concordance || source_or_pod || curate || html || epub || odt || latex || manifest || sqlite_discrete || sqlite_delete || sqlite_update || text || skel ) ? true : false; } @trusted bool require_processing_files() { return ( opts["abstraction"] || epub || curate || html || html_seg || html_scroll || latex || odt || manifest || show_abstraction || ocda_db || ocda_verify || show_make || show_metadata || show_summary || source_or_pod || sqlite_discrete || sqlite_update || text || xhtml || skel ) ? true : false; } @trusted bool meta_processing_general() { return ( opts["abstraction"] || show_abstraction || ocda_db || curate || html || epub || odt || latex || sqlite_discrete || sqlite_update || text || skel ) ? true :false; } /+ ↓ the dom structure tags, which the .ssp records and which are only computed when something needs them. . --ocda-verify belongs on this list and the omission cost an hour: verify re-parses the carried markup and compares the .ssp it would write against the one the artefact holds, and every artefact is written by a run that has this on (show_abstraction and ocda_db both turn it on). Without it, verify built an abstraction missing the dom tags, and reported ten sound languages as ten failures. . Worth knowing more generally: the abstraction is not one thing regardless of what was asked for. --text alone builds one without these tags. That is invisible in output, since only html, epub, odt and the databases use them, but anything comparing abstractions has to compare like with like. +/ @trusted bool meta_processing_xml_dom() { return ( opts["abstraction"] || show_abstraction || ocda_db || ocda_verify || html || epub || odt || sqlite_discrete || sqlite_update ) ? true : false; } } OptActions _opt_action = OptActions(); auto program_info() { struct ProgramInfo { string project() { return project_name; } string name() { return program_name; } string ver() { return format("%s.%s.%s", _ver.major, _ver.minor, _ver.patch, ); } string compiler() { return format ("%s D:%s, %s %s", __VENDOR__, __VERSION__, bits, os, ); } @trusted string name_and_version() { return format("%s-%s", name, ver); } @trusted string name_version_and_compiler() { return format("%s-%s (%s)", name, ver, compiler); } auto time_output_generated() { auto _st = Clock.currTime(UTC()); auto _t = TimeOfDay(_st.hour, _st.minute, _st.second); auto _time = _st.year.to!string ~ "-" ~ _st.month.to!int.to!string // prefer as month number ~ "-" ~ _st.day.to!string ~ " [" ~ _st.isoWeek.to!string ~ "/" ~ _st.dayOfWeek.to!int.to!string ~ "]" ~ " - " ~ _t.toISOExtString // ~ " " ~ _st.hour.to!string ~ ":" ~ _st.minute.to!string ~ ":" ~ _st.second.to!string ~ " UTC"; return _time; // return _st.toISOExtString(); } } return ProgramInfo(); } auto _env = [ "pwd" : environment["PWD"], "home" : environment["HOME"], ]; auto _manifested = PathMatters!()(_opt_action, _env, ""); auto _manifests = [ _manifested ]; string[] _artefact_args; /+ .ssp and .ocda.db given on the command line +/ auto _conf_file_details = configFilePaths!()(_manifested, _env, _opt_action.config_path_set); /+ ↓ track extracted zip pod temp directories for cleanup +/ mixin spineExtractZipPod; ZipPodResult[] _zip_pod_extractions; DownloadResult[] _url_downloads; /+ ↓ pre-process args: resolve URL arguments to local temp files +/ string[] _resolved_args; foreach (arg; args[1..$]) { if (isUrl(arg)) { /+ ↓ --allow-downloads makes fetching an argument opt in. the flag has existed since downloads were added but was not read; every url argument was fetched. A refused url is dropped from the arguments, which is what a failed download already does. +/ if (!(_opt_action.allow_downloads)) { writeln("ERROR >> Refused to fetch: ", arg, " - fetching a url source requires --allow-downloads"); continue; } auto _dlr = downloadSourceUrl(arg); if (_dlr.ok) { _url_downloads ~= _dlr; _resolved_args ~= _dlr.local_path; if (_opt_action.vox_gt_1) { writeln("downloaded: ", arg, " -> ", _dlr.local_path); } } else { writeln("ERROR >> Download failed: ", arg, " - ", _dlr.error_msg); } } else { _resolved_args ~= arg; } } /+ ↓ a database asked for as a source: write the pod back out of it and carry on with the pod. . The actions that need the source markup are --source, --pod, --pod2, --show-abstraction and --ocda-db, which are included in a 2.0 database. The pod is written to a directory of this run's making and takes the place of the argument, so everything after this point is handling a pod, and the document is built by the same code over the same bytes as the original. That is what makes "identical output" a consequence. . One route per argument, never a mix. An argument that materialises is a pod for the whole run: every output asked for comes from the markup, including the html and epub that could have been rendered from the loaded abstraction instead. Two routes for one argument would mean two answers to "what is this document", and the run could not say which it had given. . Before the config discovery below, because that walks the argument looking for a .dr/ above it, and the argument it should walk is the pod rather than the database. . A database without source markup still refuses, as it has to: no artefact written by an older spine carries any. +/ /+ ↓ --ocda-verify names its file rather than taking it as an argument, so that the question is unmistakable in the command line. It joins the arguments here and is materialised and parsed like any other database; what makes it a verification rather than a build is that nothing is written and the comparison below decides the exit status. +/ if (_opt_action.ocda_verify) { _resolved_args ~= _opt_action.ocda_verify_file; } string[] _pod_materialisations; string[string] _pod_came_from_db; // pod directory -> the database if (_opt_action.needs_markup_source) { import sisudoc.ocda.abstraction.pod_from_db; import std.process : thisProcessID; mixin spinePodFromDb; string[] _args_after; foreach (arg; _resolved_args) { if (!(arg.endsWith(".ocda.db") || arg.endsWith(".db")) || arg.match(rgx.flag_action) ) { _args_after ~= arg; continue; } string _root = (tempDir.chainPath("spine-pod-" ~ arg.baseName ~ "-" ~ thisProcessID.to!string).array).to!string; auto _mat = podFromDb!()(arg, _root, _opt_action); if (_mat.ok) { _pod_materialisations ~= _root; _pod_came_from_db[_mat.pod_dir] = arg; _args_after ~= _mat.pod_dir; /+ ↓ a document rebuilt from a database gets the whole abstraction built, dom structure tags and all. . Because that is what it will be compared against: the artefact recorded its abstraction from a run that had these on, every run that writes one having them on. Parsed without them, the same markup makes a different .ssp, and verify would report ten sound languages as ten failures. It did, for an hour. +/ opts["abstraction"] = true; if (_opt_action.vox_gt_1) { writeln("pod from database: ", arg.baseName, " -> ", _mat.pod_dir); } } else { /+ ↓ the argument stays, and is read as an abstraction below: what was asked for that needs the markup cannot be done, and what does not (need markup) still can. Which actions those are is specified once, by the artefact loop this argument now falls to, rather than twice here as well. +/ _args_after ~= arg; stderr.writeln("WARNING: ", arg.baseName, " ", _mat.note); } } _resolved_args = _args_after; } ConfComposite _siteConfig; if ( _opt_action.require_processing_files && _opt_action.config_path_set.empty ) { foreach(arg; _resolved_args) { if (!(arg.match(rgx.flag_action))) { /+ cli markup source path +/ // get first input markup source file names for processing string _config_arg = arg; /+ ↓ if first non-flag arg is a zip, extract for config discovery +/ if (arg.match(rgx_files.src_pth_zip)) { auto _zpr = extractZipPod(arg); if (_zpr.ok) { _zip_pod_extractions ~= _zpr; _config_arg = _zpr.pod_dir; } } _manifested = PathMatters!()(_opt_action, _env, _config_arg); { /+ local site config +/ _conf_file_details = configFilePaths!()(_manifested, _env, _opt_action.config_path_set); auto _config_local_site_struct = readConfigSite!()(_conf_file_details, _opt_action, _cfg); import sisudoc.ocda.meta.conf_make_meta_yaml; _siteConfig = _config_local_site_struct.configParseYAMLreturnSpineStruct!()(_siteConfig, _manifested, _opt_action, _cfg); // - get local site config break; } } } } else { /+ local site config +/ auto _config_local_site_struct = readConfigSite!()(_conf_file_details, _opt_action, _cfg); import sisudoc.ocda.meta.conf_make_meta_yaml; _siteConfig = _config_local_site_struct.configParseYAMLreturnSpineStruct!()(_siteConfig, _manifested, _opt_action, _cfg); // - get local site config } if (_opt_action.show_config) { import sisudoc.ocda.meta.metadoc_show_config; spineShowSiteConfig!()(_opt_action, _siteConfig); } if (!(_opt_action.skip_output)) { if ((_opt_action.debug_do) || (_opt_action.debug_do_stages) ) { writeln("step0 commence → (without processing files)"); } outputHubOp!()(_env, _opt_action, _siteConfig); if ((_opt_action.debug_do) || (_opt_action.debug_do_stages) ) { writeln("- step0 complete"); } } ConfComposite _make_and_meta_struct = _siteConfig; destroy(_siteConfig); foreach(arg; _resolved_args) { if (arg.match(rgx.flag_action)) { /+ cli instruction, flag do +/ flag_action ~= " " ~ arg; // flags not taken by getopt } else if (_opt_action.require_processing_files) { /+ cli, assumed to be path to source files +/ auto _manifest_start = PodManifest!()(_opt_action, arg); if ( /+ pod files +/ !(arg.match(rgx_files.src_pth_sst_or_ssm)) && _manifest_start.pod_manifest_file_with_path && _opt_action.abstraction ) { string pod_manifest_root_content_paths_to_markup_location_raw_; string markup_contents_location_; string sisudoc_txt_ = _manifest_start.pod_manifest_file_with_path; enforce( exists(sisudoc_txt_)!=0, "file not found: «" ~ sisudoc_txt_ ~ "»" ); if (exists(sisudoc_txt_)) { try { import dyaml; Node pod_manifest_yaml; try { pod_manifest_yaml = Loader.fromFile(sisudoc_txt_).load(); } catch (ErrnoException ex) { } catch (FileException ex) { writeln("ERROR failed to read config file"); } catch (Throwable) { writeln("ERROR failed to read config file content, not parsed as yaml"); } if ("doc" in pod_manifest_yaml) { if (pod_manifest_yaml["doc"].type.mapping && pod_manifest_yaml["doc"].tag.match(rgx_y.yaml_tag_is_map) ) { if ("path" in pod_manifest_yaml["doc"]) { if (pod_manifest_yaml["doc"]["path"].tag.match(rgx_y.yaml_tag_is_seq)) { foreach (string _path; pod_manifest_yaml["doc"]["path"]) { markup_contents_location_ ~= _path ~ "\n"; pod_manifest_root_content_paths_to_markup_location_raw_ ~= _path ~ "\n"; } } else if ( pod_manifest_yaml["doc"]["path"].type.string && pod_manifest_yaml["doc"]["path"].tag.match(rgx_y.yaml_tag_is_str) ) { markup_contents_location_ = pod_manifest_yaml["doc"]["path"].get!string; pod_manifest_root_content_paths_to_markup_location_raw_ = pod_manifest_yaml["doc"]["path"].get!string; } } if ("filename" in pod_manifest_yaml["doc"]) { if (pod_manifest_yaml["doc"]["filename"].tag.match(rgx_y.yaml_tag_is_seq)) { foreach (string _filename; pod_manifest_yaml["doc"]["filename"]) { if ("language" in pod_manifest_yaml["doc"]) { if (pod_manifest_yaml["doc"]["language"].tag.match(rgx_y.yaml_tag_is_seq)) { foreach (string _lang; pod_manifest_yaml["doc"]["language"]) { markup_contents_location_ ~= "media/text/" ~ _lang ~ "/" ~ _filename ~ "\n"; } } else if (pod_manifest_yaml["doc"]["language"].tag.match(rgx_y.yaml_tag_is_str) ) { markup_contents_location_ = "media/text/" ~ pod_manifest_yaml["doc"]["language"].get!string ~ "/" ~ _filename ~ "\n"; } else { string _lang_default = "en"; markup_contents_location_ ~= "media/text/" ~ _lang_default ~ "/" ~ pod_manifest_yaml["doc"]["filename"].get!string ~ "\n"; } } else { string _lang_default = "en"; markup_contents_location_ ~= "media/text/" ~ _lang_default ~ "/" ~ pod_manifest_yaml["doc"]["filename"].get!string ~ "\n"; } } } else if ( pod_manifest_yaml["doc"]["filename"].type.string && pod_manifest_yaml["doc"]["filename"].tag.match(rgx_y.yaml_tag_is_str) ) { if ("language" in pod_manifest_yaml["doc"]) { if (pod_manifest_yaml["doc"]["language"].tag.match(rgx_y.yaml_tag_is_seq)) { foreach (string _lang; pod_manifest_yaml["doc"]["language"]) { markup_contents_location_ ~= "media/text/" ~ _lang ~ "/" ~ pod_manifest_yaml["doc"]["filename"].get!string ~ "\n"; } } else if (pod_manifest_yaml["doc"]["language"].tag.match(rgx_y.yaml_tag_is_str)) { markup_contents_location_ = "media/text/" ~ pod_manifest_yaml["doc"]["language"].get!string ~ "/" ~ pod_manifest_yaml["doc"]["filename"].get!string ~ "\n"; } else { string _lang_default = "en"; markup_contents_location_ ~= "media/text/" ~ _lang_default ~ "/" ~ pod_manifest_yaml["doc"]["filename"].get!string ~ "\n"; } } else { string _lang_default = "en"; markup_contents_location_ ~= "media/text/" ~ _lang_default ~ "/" ~ pod_manifest_yaml["doc"]["filename"].get!string ~ "\n"; } } } } } } catch (ErrnoException ex) { } catch (FileException ex) { // Handle errors } } else { writeln("manifest not found: ", sisudoc_txt_); } auto markup_contents_locations_arr = (cast(char[]) markup_contents_location_).split; auto tmp_dir_ = (sisudoc_txt_).dirName.array; foreach (markup_contents_location; markup_contents_locations_arr) { assert(markup_contents_location.match(rgx_files.src_pth_sst_or_ssm), "not a recognised file: «" ~ markup_contents_location ~ "»" ); auto markup_contents_location_pth_ = (markup_contents_location).to!string; Regex!(char) lang_rgx_ = regex(r"/(" ~ _opt_action.languages_set.join("|") ~ ")/"); if (_opt_action.languages_set[0] == "all" || (markup_contents_location_pth_).match(lang_rgx_) ) { auto _fns = (((tmp_dir_).chainPath(markup_contents_location_pth_)).array).to!string; _manifested = PathMatters!()(_opt_action, _env, arg, _fns, markup_contents_locations_arr); _manifests ~= _manifested; } } } else if (arg.match(rgx_files.src_pth_sst_or_ssm)) { /+ markup txt files +/ if (exists(arg)==0) { writeln("ERROR >> Processing Skipped! File not found: ", arg); } else { _manifested = PathMatters!()(_opt_action, _env, arg, arg); _manifests ~= _manifested; } } else if (arg.match(rgx_files.src_pth_zip)) { /+ ↓ zip pod archive: extract to temp dir, process as pod +/ /+ check if this zip was already extracted during config discovery +/ string _zip_pod_dir; foreach (ref _zpr; _zip_pod_extractions) { if (_zpr.ok && _zpr.pod_dir.length > 0 && _zpr.pod_dir.baseName == arg.baseName.stripExtension) { _zip_pod_dir = _zpr.pod_dir; break; } } if (_zip_pod_dir.length == 0) { auto _zpr = extractZipPod(arg); if (!_zpr.ok) { writeln("ERROR >> Processing Skipped! Zip extraction failed: ", arg, " - ", _zpr.error_msg); } else { _zip_pod_extractions ~= _zpr; _zip_pod_dir = _zpr.pod_dir; } } if (_zip_pod_dir.length > 0) { /+ process extracted pod directory same as regular pod +/ auto _zip_manifest = PodManifest!()(_opt_action, _zip_pod_dir); if (_zip_manifest.pod_manifest_file_with_path && _opt_action.abstraction ) { string pod_manifest_root_content_paths_to_markup_location_raw_; string markup_contents_location_; string sisudoc_txt_ = _zip_manifest.pod_manifest_file_with_path; enforce( exists(sisudoc_txt_)!=0, "file not found: <<" ~ sisudoc_txt_ ~ ">>" ); if (exists(sisudoc_txt_)) { try { import dyaml; Node pod_manifest_yaml; try { pod_manifest_yaml = Loader.fromFile(sisudoc_txt_).load(); } catch (ErrnoException ex) { } catch (FileException ex) { writeln("ERROR failed to read config file"); } catch (Throwable) { writeln("ERROR failed to read config file content, not parsed as yaml"); } if ("doc" in pod_manifest_yaml) { if (pod_manifest_yaml["doc"].type.mapping && pod_manifest_yaml["doc"].tag.match(rgx_y.yaml_tag_is_map) ) { if ("path" in pod_manifest_yaml["doc"]) { if (pod_manifest_yaml["doc"]["path"].tag.match(rgx_y.yaml_tag_is_seq)) { foreach (string _path; pod_manifest_yaml["doc"]["path"]) { markup_contents_location_ ~= _path ~ "\n"; pod_manifest_root_content_paths_to_markup_location_raw_ ~= _path ~ "\n"; } } else if ( pod_manifest_yaml["doc"]["path"].type.string && pod_manifest_yaml["doc"]["path"].tag.match(rgx_y.yaml_tag_is_str) ) { markup_contents_location_ = pod_manifest_yaml["doc"]["path"].get!string; pod_manifest_root_content_paths_to_markup_location_raw_ = pod_manifest_yaml["doc"]["path"].get!string; } } if ("filename" in pod_manifest_yaml["doc"]) { if (pod_manifest_yaml["doc"]["filename"].tag.match(rgx_y.yaml_tag_is_seq)) { foreach (string _filename; pod_manifest_yaml["doc"]["filename"]) { if ("language" in pod_manifest_yaml["doc"]) { if (pod_manifest_yaml["doc"]["language"].tag.match(rgx_y.yaml_tag_is_seq)) { foreach (string _lang; pod_manifest_yaml["doc"]["language"]) { markup_contents_location_ ~= "media/text/" ~ _lang ~ "/" ~ _filename ~ "\n"; } } else if (pod_manifest_yaml["doc"]["language"].tag.match(rgx_y.yaml_tag_is_str) ) { markup_contents_location_ = "media/text/" ~ pod_manifest_yaml["doc"]["language"].get!string ~ "/" ~ _filename ~ "\n"; } else { string _lang_default = "en"; markup_contents_location_ ~= "media/text/" ~ _lang_default ~ "/" ~ pod_manifest_yaml["doc"]["filename"].get!string ~ "\n"; } } else { string _lang_default = "en"; markup_contents_location_ ~= "media/text/" ~ _lang_default ~ "/" ~ pod_manifest_yaml["doc"]["filename"].get!string ~ "\n"; } } } else if ( pod_manifest_yaml["doc"]["filename"].type.string && pod_manifest_yaml["doc"]["filename"].tag.match(rgx_y.yaml_tag_is_str) ) { if ("language" in pod_manifest_yaml["doc"]) { if (pod_manifest_yaml["doc"]["language"].tag.match(rgx_y.yaml_tag_is_seq)) { foreach (string _lang; pod_manifest_yaml["doc"]["language"]) { markup_contents_location_ ~= "media/text/" ~ _lang ~ "/" ~ pod_manifest_yaml["doc"]["filename"].get!string ~ "\n"; } } else if (pod_manifest_yaml["doc"]["language"].tag.match(rgx_y.yaml_tag_is_str)) { markup_contents_location_ = "media/text/" ~ pod_manifest_yaml["doc"]["language"].get!string ~ "/" ~ pod_manifest_yaml["doc"]["filename"].get!string ~ "\n"; } else { string _lang_default = "en"; markup_contents_location_ ~= "media/text/" ~ _lang_default ~ "/" ~ pod_manifest_yaml["doc"]["filename"].get!string ~ "\n"; } } else { string _lang_default = "en"; markup_contents_location_ ~= "media/text/" ~ _lang_default ~ "/" ~ pod_manifest_yaml["doc"]["filename"].get!string ~ "\n"; } } } } } } catch (ErrnoException ex) { } catch (FileException ex) { // Handle errors } } else { writeln("manifest not found: ", sisudoc_txt_); } auto markup_contents_locations_arr = (cast(char[]) markup_contents_location_).split; auto tmp_dir_ = (sisudoc_txt_).dirName.array; foreach (markup_contents_location; markup_contents_locations_arr) { assert(markup_contents_location.match(rgx_files.src_pth_sst_or_ssm), "not a recognised file: <<" ~ markup_contents_location ~ ">>" ); auto markup_contents_location_pth_ = (markup_contents_location).to!string; Regex!(char) lang_rgx_ = regex(r"/(" ~ _opt_action.languages_set.join("|") ~ ")/"); if (_opt_action.languages_set[0] == "all" || (markup_contents_location_pth_).match(lang_rgx_) ) { auto _fns = (((tmp_dir_).chainPath(markup_contents_location_pth_)).array).to!string; _manifested = PathMatters!()(_opt_action, _env, _zip_pod_dir, _fns, markup_contents_locations_arr); _manifests ~= _manifested; } } } } } else if (arg.endsWith(".ssp") || arg.endsWith(".ocda.db")) { /+ ↓ an abstraction artefact: the document is read rather than parsed, and joins the run as a document like any other +/ _artefact_args ~= arg; } else { // anything remaining, unused arg_unrecognized ~= " " ~ arg; } } } /+ ↓ the ocn alignment profiles, one per language, compared after the loop. A named mixin, so that what the template declares stays in a scope of its own rather than joining main's. +/ mixin spineOcnAlign _ocna; _ocna.ST_OcnProfile[] _ocn_profiles; /+ ↓ what verify found, gathered as the documents go by +/ size_t _verify_checked = 0; size_t _verify_failed = 0; string[] _verify_notes; if (_manifests.length > 1 // _manifests[0] initialized dummy element && _opt_action.abstraction) { /+ ↓ output hub +/ if (!(_opt_action.skip_output)) { outputHubInitialize!()(_opt_action, program_info); } if (_opt_action.parallelise) { // see else import std.parallelism; foreach(manifest; parallel(_manifests[1..$])) { if (!empty(manifest.src.filename)) { scope(success) { if (_opt_action.vox_gt_1) { writeln("-- ~ document complete, ok ~ ------------------------------------"); } } scope(failure) { debug(checkdoc) { stderr.writefln( "~ document run failure ~ (%s v%s)\n\t%s\n%s", __VENDOR__, __VERSION__, manifest.src.filename, "------------------------------------------------------------------", ); } } enforce( manifest.src.filename.match(rgx_files.src_pth_types), "not a sisu markup filename: «" ~ manifest.src.filename ~ "»" ); if ((_opt_action.debug_do) || (_opt_action.debug_do_stages) ) { writeln("--->\nstepX commence → (document abstraction) [", manifest.src.filename, "]"); } auto doc = spineAbstraction!()(_env, program_info, _opt_action, _cfg, manifest, _make_and_meta_struct); if ((doc.matters.opt.action.debug_do) || (_opt_action.debug_do_stages) ) { writeln("- stepX complete for [", manifest.src.filename, "]"); } /+ ↓ verify: the markup the database carried, re-parsed, against the abstraction the database holds. . Here and not later because this is the last moment before anything is written from it. The document has been parsed and nothing has been produced yet, so a mismatch can still stop the writing rather than be discovered in it. . Only for a document that came from a database: a pod given on the command line is the source, and there is nothing to hold it against. +/ bool _verified = true; if (_pod_came_from_db.length > 0) { import sisudoc.ocda.abstraction.ssp_in; import sisudoc.ocda.abstraction.db_in : spineDbSspDigests; mixin spineAbstractionRead _vfy; string _pod_key = (doc.matters.pod.manifest_path.asNormalizedPath).array.to!string; if (auto _db = _pod_key in _pod_came_from_db) { string _made = _vfy.sspRoundTripDocument(doc).ssp_digest; auto _recorded = spineDbSspDigests!()(*_db); string _lang = doc.matters.src.language; string _was = (_lang in _recorded) ? _recorded[_lang] : ""; synchronized { _verify_checked += 1; if (_was.length == 0) { _verify_notes ~= (*_db).baseName ~ " [" ~ _lang ~ "] records no abstraction digest to check against"; _verify_failed += 1; _verified = false; } else if (_was != _made) { _verify_notes ~= (*_db).baseName ~ " [" ~ _lang ~ "] the markup it carries no longer makes the" ~ " abstraction it holds"; _verify_notes ~= " recorded " ~ _was; _verify_notes ~= " made now " ~ _made; _verify_failed += 1; _verified = false; } } } } if (!_verified) { if (_opt_action.no_verify) { stderr.writeln("WARNING: --no-verify, so this is built from", " markup that no longer makes the abstraction recorded with", " it: ", manifest.src.filename, " [", manifest.src.language, "]"); } else if (_opt_action.ocda_verify) { /+ ↓ verify writes nothing, so the rest of the languages are worth checking: the answer wanted is how much of the file is sound, not which language failed first +/ continue; } else { /+ ↓ a build stops here, and does not skip the language and carry on. . Skipping one language of a pod leaves the pod writer carrying markup for a language whose directory was never made, which threw a FileException out of the run: a half written pod, and an exception instead of an explanation. A document whose carried markup no longer makes its recorded abstraction is not one to write a pod from at all, so the run ends here and says so. +/ foreach (_n; _verify_notes) { stderr.writeln(" ", _n); } stderr.writeln("~ run FAILED ~ verify: ", manifest.src.filename, " [", manifest.src.language, "] is built from markup that no", " longer makes the abstraction recorded with it."); stderr.writeln(" Anything already written is incomplete.", " --ocda-verify= reports on every language, and", " --no-verify builds anyway."); import core.stdc.stdlib : exit; stdout.flush; exit(1); } } /+ ↓ debugs +/ if (doc.matters.opt.action.show_summary) { import sisudoc.ocda.meta.metadoc_show_summary; spineMetaDocSummary!()(doc); } /+ ↓ debugs +/ if (doc.matters.opt.action.show_metadata) { import sisudoc.ocda.meta.metadoc_show_metadata; spineShowMetaData!()(doc.matters); } /+ ↓ debugs +/ if (doc.matters.opt.action.show_make) { import sisudoc.ocda.meta.metadoc_show_make; spineShowMake!()(doc.matters); } /+ ↓ debugs +/ if (doc.matters.opt.action.show_config) { import sisudoc.ocda.meta.metadoc_show_config; spineShowConfig!()(doc.matters); } /+ ↓ document abstraction text representation +/ if (doc.matters.opt.action.show_abstraction) { import sisudoc.ocda.abstraction.ssp; spineAbstractionTxt!()(doc); } /+ ↓ document abstraction sqlite database, built from the .ssp: the lines are emitted and read straight back, and the objects that come out are what is written, so the database cannot carry anything the .ssp does not +/ if (doc.matters.opt.action.ocda_db) { import sisudoc.ocda.abstraction.ssp_in; import sisudoc.outputs.io_out.sqlite_ocda_db; mixin spineAbstractionRead; spineAbstractionDb!()(doc, sspRoundTripDocument(doc)); } /+ ↓ the ocn alignment profile of this language, kept for the check that runs once every language of the document has been abstracted. Taken whatever the outputs are: the languages of a document share their object numbering whether or not anything is being written. +/ { auto _ocn_p = _ocna.ocnProfile(doc); if (doc.matters.opt.action.ocda_db) { import sisudoc.outputs.io_out.paths_output; auto _pths_pod_ocn = spinePathsPods!()(doc.matters); _ocn_p.db_file = _pths_pod_ocn.pod_dir_() ~ "/" ~ doc.matters.src.doc_uid_out_no_lang ~ ".ocda.db"; } synchronized { _ocn_profiles ~= _ocn_p; } } if (doc.matters.opt.action.curate) { auto _hvst = spineMetaDocCurate!()(doc.matters, hvst); if ( _hvst.title.length > 0 && _hvst.author_surname_fn.length > 0 ) { hvst.curates ~= _hvst; } else { if ((doc.matters.opt.action.debug_do) || (_opt_action.debug_do_curate) || (doc.matters.opt.action.vox_gt_3) ) { writeln("WARNING curate: document header yaml does not contain information related to: title or author: ", _hvst.path_html_segtoc); } } } /+ ↓ debugs +/ if (doc.matters.opt.action.debug_do) { spineDebugs!()(doc.abstraction, doc.matters); } /+ ↓ output hub +/ if (!(doc.matters.opt.action.skip_output)) { if ((_opt_action.debug_do) || (_opt_action.debug_do_stages)) { writeln("step5 commence → (process outputs) [", manifest.src.filename, "]"); } doc.outputHub!(); if ((_opt_action.debug_do) || (_opt_action.debug_do_stages)) { writeln("- step5 complete for [", manifest.src.filename, "]"); } } scope(exit) { if (_opt_action.vox_gt_1) { writefln( "processed file: %s [%s]", manifest.src.filename, manifest.src.language ); } destroy(manifest); } } else { /+ no recognized filename provided +/ writeln("no recognized filename"); break; // terminate, stop } } } else { // note cannot parallelise sqlite shared db foreach(manifest; _manifests[1..$]) { if (_opt_action.vox_gt_3) { writeln("parallelisation off: actions include sqlite shared db"); } if (!empty(manifest.src.filename)) { scope(success) { if (_opt_action.vox_gt_1) { writeln("-- ~ document complete, ok ~ ------------------------------------"); } } scope(failure) { debug(checkdoc) { stderr.writefln( "~ document run failure ~ (%s v%s)\n\t%s\n%s", __VENDOR__, __VERSION__, manifest.src.filename, "------------------------------------------------------------------", ); } } enforce( manifest.src.filename.match(rgx_files.src_pth_types), "not a sisu markup filename: «" ~ manifest.src.filename ~ "»" ); if ((_opt_action.debug_do) || (_opt_action.debug_do_stages) ) { writeln("--->\nstepX commence → (document abstraction) [", manifest.src.filename, "]"); } auto doc = spineAbstraction!()(_env, program_info, _opt_action, _cfg, manifest, _make_and_meta_struct); if ((doc.matters.opt.action.debug_do) || (_opt_action.debug_do_stages) ) { writeln("- stepX complete for [", manifest.src.filename, "]"); } /+ ↓ verify: the markup the database carried, re-parsed, against the abstraction the database holds. . Here and not later because this is the last moment before anything is written from it. The document has been parsed and nothing has been produced yet, so a mismatch can still stop the writing rather than be discovered in it. . Only for a document that came from a database: a pod given on the command line is the source, and there is nothing to hold it against. +/ bool _verified = true; if (_pod_came_from_db.length > 0) { import sisudoc.ocda.abstraction.ssp_in; import sisudoc.ocda.abstraction.db_in : spineDbSspDigests; mixin spineAbstractionRead _vfy; string _pod_key = (doc.matters.pod.manifest_path.asNormalizedPath).array.to!string; if (auto _db = _pod_key in _pod_came_from_db) { string _made = _vfy.sspRoundTripDocument(doc).ssp_digest; auto _recorded = spineDbSspDigests!()(*_db); string _lang = doc.matters.src.language; string _was = (_lang in _recorded) ? _recorded[_lang] : ""; synchronized { _verify_checked += 1; if (_was.length == 0) { _verify_notes ~= (*_db).baseName ~ " [" ~ _lang ~ "] records no abstraction digest to check against"; _verify_failed += 1; _verified = false; } else if (_was != _made) { _verify_notes ~= (*_db).baseName ~ " [" ~ _lang ~ "] the markup it carries no longer makes the" ~ " abstraction it holds"; _verify_notes ~= " recorded " ~ _was; _verify_notes ~= " made now " ~ _made; _verify_failed += 1; _verified = false; } } } } if (!_verified) { if (_opt_action.no_verify) { stderr.writeln("WARNING: --no-verify, so this is built from", " markup that no longer makes the abstraction recorded with", " it: ", manifest.src.filename, " [", manifest.src.language, "]"); } else if (_opt_action.ocda_verify) { /+ ↓ verify writes nothing, so the rest of the languages are worth checking: the answer wanted is how much of the file is sound, not which language failed first +/ continue; } else { /+ ↓ a build stops here, and does not skip the language and carry on. . Skipping one language of a pod leaves the pod writer carrying markup for a language whose directory was never made, which threw a FileException out of the run: a half written pod, and an exception instead of an explanation. A document whose carried markup no longer makes its recorded abstraction is not one to write a pod from at all, so the run ends here and says so. +/ foreach (_n; _verify_notes) { stderr.writeln(" ", _n); } stderr.writeln("~ run FAILED ~ verify: ", manifest.src.filename, " [", manifest.src.language, "] is built from markup that no", " longer makes the abstraction recorded with it."); stderr.writeln(" Anything already written is incomplete.", " --ocda-verify= reports on every language, and", " --no-verify builds anyway."); import core.stdc.stdlib : exit; stdout.flush; exit(1); } } /+ ↓ debugs +/ if (doc.matters.opt.action.show_summary) { import sisudoc.ocda.meta.metadoc_show_summary; spineMetaDocSummary!()(doc); } /+ ↓ debugs +/ if (doc.matters.opt.action.show_metadata) { import sisudoc.ocda.meta.metadoc_show_metadata; spineShowMetaData!()(doc.matters); } /+ ↓ debugs +/ if (doc.matters.opt.action.show_make) { import sisudoc.ocda.meta.metadoc_show_make; spineShowMake!()(doc.matters); } /+ ↓ debugs +/ if (doc.matters.opt.action.show_config) { import sisudoc.ocda.meta.metadoc_show_config; spineShowConfig!()(doc.matters); } /+ ↓ document abstraction text representation +/ if (doc.matters.opt.action.show_abstraction) { import sisudoc.ocda.abstraction.ssp; spineAbstractionTxt!()(doc); } /+ ↓ document abstraction sqlite database, built from the .ssp: the lines are emitted and read straight back, and the objects that come out are what is written, so the database cannot carry anything the .ssp does not +/ if (doc.matters.opt.action.ocda_db) { import sisudoc.ocda.abstraction.ssp_in; import sisudoc.outputs.io_out.sqlite_ocda_db; mixin spineAbstractionRead; spineAbstractionDb!()(doc, sspRoundTripDocument(doc)); } /+ ↓ the ocn alignment profile of this language, as (for parallel processing) above +/ { auto _ocn_p = _ocna.ocnProfile(doc); if (doc.matters.opt.action.ocda_db) { import sisudoc.outputs.io_out.paths_output; auto _pths_pod_ocn = spinePathsPods!()(doc.matters); _ocn_p.db_file = _pths_pod_ocn.pod_dir_() ~ "/" ~ doc.matters.src.doc_uid_out_no_lang ~ ".ocda.db"; } _ocn_profiles ~= _ocn_p; } if (doc.matters.opt.action.curate) { auto _hvst = spineMetaDocCurate!()(doc.matters, hvst); if ( _hvst.title.length > 0 && _hvst.author_surname_fn.length > 0 ) { hvst.curates ~= _hvst; } else { if ((doc.matters.opt.action.debug_do) || (_opt_action.debug_do_curate) || (doc.matters.opt.action.vox_gt_3) ) { writeln("WARNING curate: document header yaml does not contain information related to: title or author: ", _hvst.path_html_segtoc); } } } /+ ↓ debugs +/ if (doc.matters.opt.action.debug_do) { spineDebugs!()(doc.abstraction, doc.matters); } /+ ↓ output hub +/ if (!(doc.matters.opt.action.skip_output)) { if ((_opt_action.debug_do) || (_opt_action.debug_do_stages)) { writeln("step5 commence → (process outputs) [", manifest.src.filename, "]"); } doc.outputHub!(); if ((_opt_action.debug_do) || (_opt_action.debug_do_stages)) { writeln("- step5 complete for [", manifest.src.filename, "]"); } } scope(exit) { if (_opt_action.vox_gt_1) { writefln( "processed file: %s [%s]", manifest.src.filename, manifest.src.language ); } destroy(manifest); } } else { /+ no recognized filename provided +/ writeln("no recognized filename"); break; // terminate, stop } } } } /+ ↓ what verify found, said once at the end. A run that was only asked to verify says so plainly and its exit status is the answer. A run that was building says it too, because a document whose markup no longer makes its abstraction was skipped and the user has to know why their output is short. +/ if (_verify_checked > 0) { foreach (_n; _verify_notes) { stderr.writeln(" ", _n); } if (_opt_action.ocda_verify || _opt_action.vox_gt_1 || _verify_failed > 0) { writeln("verify: ", _verify_checked - _verify_failed, " of ", _verify_checked, " re-parse to the abstraction recorded with them"); } } else if (_opt_action.ocda_verify) { stderr.writeln("verify: nothing was checked. ", _opt_action.ocda_verify_file.baseName, " carries no markup, or none that could be read"); _verify_failed += 1; } /+ ↓ the languages of a document, held against each other. . Here because it is the first point at which it can be done: a language cannot know whether it numbers its objects the same way as its siblings until they have all been abstracted. A warning, so that a document with a diverging translation still builds; --strict turns it into a failure for the run that is meant to be publishable. . The outcome is noted per language in the database, where there is one, under translation.ocn_aligned. Only for documents that have more than one language: a single language document has nothing to align with, and "true" there would be an answer to a question nobody asked. +/ bool _ocn_align_failed = false; if (_ocn_profiles.length > 1) { auto _ocn_results = _ocna.ocnAlignCompare(_ocn_profiles); foreach (_line; _ocna.ocnAlignReportLines(_ocn_results)) { stderr.writeln(_line); } bool[string][string] _aligned; // doc_key -> lang -> aligned foreach (_r; _ocn_results) { _aligned[_r.doc_key][_r.lang] = _r.aligned; _aligned[_r.doc_key][_r.reference_lang] = true; if (!_r.aligned) { _ocn_align_failed = true; } } foreach (_p; _ocn_profiles) { if (_p.db_file.length == 0) { continue; } if (_p.doc_key !in _aligned) { continue; } if (_p.lang !in _aligned[_p.doc_key]) { continue; } import sisudoc.outputs.io_out.sqlite_ocda_db : spineOcdaDbDocNote; spineOcdaDbDocNote!()(_p.db_file, _p.lang, "translation.ocn_aligned", _aligned[_p.doc_key][_p.lang] ? "true" : "false"); } } /+ ↓ documents given as an abstraction artefact rather than as markup. .ssp and .ocda.db are read, not parsed, and what comes back is the doc the output writers already take, so this loop is the markup one with its first step replaced. It is serial: the artefacts are read one at a time and there is nothing here that parallelising would help with, the reading being the cheap half. --source and --pod2 are refused rather than half done: no artefact carries the markup, so a pod cannot be rebuilt from one. +/ if (_artefact_args.length > 0 && _opt_action.abstraction) { import sisudoc.ocda.abstraction.doc_from_artefact; if (!(_opt_action.skip_output)) { outputHubInitialize!()(_opt_action, program_info); } /+ ↓ what is left here needs the source markup and could not get it. A .ocda.db that carries source never reaches this loop: it was written back out as a pod above and is being parsed. So an artefact here is a .ssp, which carries no markup by construction, or a database that has none, which has already said so by name. Either way the actions that need the source are not available for it, and the rest of the run goes on. +/ if (_opt_action.needs_markup_source) { string[] _not_available; if (_opt_action.source_or_pod) { _not_available ~= "--source/--pod"; } if (_opt_action.show_abstraction) { _not_available ~= "--show-abstraction"; } if (_opt_action.ocda_db) { _not_available ~= "--ocda-db"; } writeln("WARNING: an abstraction alone does not carry markup, so these", " are skipped: ", _not_available.join(", ")); writeln(" for: ", _artefact_args.join(", ")); } foreach (_artefact; _artefact_args) { /+ ↓ a .ocda.db holds every language of its document, so one artefact can be several documents to build. A .ssp is one, and comes back as a single unnamed language. +/ string[] _artefact_langs = spineArtefactLanguages!()(_artefact, _opt_action.languages_set); if (_artefact_langs.length == 0) { stderr.writeln("WARNING: ", _artefact.baseName, " holds no document in the language(s) asked for: ", _opt_action.languages_set.join(" ")); continue; } foreach (_artefact_lang; _artefact_langs) { auto doc = spineDocFromArtefact!()( _artefact, program_info, _opt_action, _env, _make_and_meta_struct, _artefact_lang, ); if (!doc.loaded) { stderr.writeln("ERROR: could not load ", _artefact, (_artefact_lang.length > 0) ? " [" ~ _artefact_lang ~ "]" : "", (doc.note.length > 0) ? ": " ~ doc.note : ""); continue; } if (_opt_action.vox_gt_1) { writeln("-- ~ document from ", doc.source_name, " ~ ------------------------------------"); writeln(" ", _artefact); } if (doc.matters.opt.action.curate) { auto _hvst = spineMetaDocCurate!()(doc.matters, hvst); if (_hvst.title.length > 0 && _hvst.author_surname_fn.length > 0) { hvst.curates ~= _hvst; } } if (!(doc.matters.opt.action.skip_output)) { doc.outputHub!(); } if (_opt_action.vox_gt_1) { writeln("processed artefact: ", _artefact.baseName, " [", doc.matters.src.language, "]"); } /+ ↓ the images a database carried, written out for this run only +/ if (doc.images_tmp.length > 0) { try { if (doc.images_tmp.exists) { doc.images_tmp.rmdirRecurse; } } catch (Exception ex) { stderr.writeln("WARNING: could not remove ", doc.images_tmp, ": ", ex.msg); } } } } } if (hvst.curates.length > 0) { if (_opt_action.curate_topics) { spineMetaDocCuratesTopics!()(hvst, _make_and_meta_struct, _opt_action); } if (_opt_action.curate_authors) { spineMetaDocCuratesAuthors!()(hvst.curates, _make_and_meta_struct, _opt_action); } if (_opt_action.vox_gt_1) { import sisudoc.outputs.io_out.paths_output; auto out_pth = spinePathsHTML!()(_make_and_meta_struct.conf.output_path, ""); if (_opt_action.curate_authors) { writeln("- ", out_pth.curate("authors.html")); } if (_opt_action.curate_topics) { writeln("- ", out_pth.curate("topics.html")); } } } // else { writeln("NO METADATA CURATED"); } /+ ↓ clean up any extracted zip pod temp directories +/ foreach (ref _zpr; _zip_pod_extractions) { cleanupZipPod(_zpr); } /+ ↓ clean up any downloaded temp files +/ foreach (ref _dlr; _url_downloads) { cleanupDownload(_dlr); } /+ ↓ and any pod written out of a database for this run +/ foreach (_root; _pod_materialisations) { try { if (_root.exists) { _root.rmdirRecurse; } } catch (Exception ex) { stderr.writeln("WARNING: could not remove ", _root, ": ", ex.msg); } } /+ ↓ --strict and a document whose languages diverged: fatal to the run. At the end rather than where the check runs, so that the outputs of the run are complete and can be looked at. The warning has already been printed with the detail; this is the exit status. +/ if (_ocn_align_failed && _opt_action.strict) { stderr.writeln( "~ run FAILED ~ --strict: ocn alignment differs between languages" ); import core.stdc.stdlib : exit; exit(1); } /+ ↓ a verification that failed is a failed run, whatever else went well. --no-verify has already said its piece per document and does not reach here: it turns the failure into a warning and the output is produced deliberately. +/ if (_verify_failed > 0 && !(_opt_action.no_verify)) { stderr.writefln( "~ run FAILED ~ verify: %s of %s did not re-parse to the abstraction" ~ " recorded with them", _verify_failed, _verify_checked, ); import core.stdc.stdlib : exit; exit(1); } /+ ↓ sqlite statements that failed are fatal to the run, report and exit non-zero +/ { import sisudoc.outputs.io_out.sqlite_collection_db : sqliteFailureTally; if (sqliteFailureTally() > 0) { stderr.writefln( "~ run FAILED ~ %s sqlite statement(s) failed, db not written as expected", sqliteFailureTally(), ); import core.stdc.stdlib : exit; exit(1); } } }