-*- mode: org -*- #+TITLE: sisudoc spine (doc_reform) markup source raw #+DESCRIPTION: documents - structuring, publishing in multiple formats & search #+FILETAGS: :spine:sourcefile:read: #+AUTHOR: Ralph Amissah #+EMAIL: [[mailto:ralph.amissah@gmail.com][ralph.amissah@gmail.com]] #+COPYRIGHT: Copyright (C) 2015 (continuously updated, current 2026) Ralph Amissah #+LANGUAGE: en #+STARTUP: content hideblocks hidestars noindent entitiespretty #+PROPERTY: header-args+ :eval never-export :exports code #+PROPERTY: header-args+ :noweb yes :padline no #+PROPERTY: header-args+ :results silent :cache no #+PROPERTY: header-args+ :mkdirp yes #+OPTIONS: H:3 num:nil toc:t \n:t ::t |:t ^:nil -:t f:t *:t - magic single double-quote → " ← FIX changes hilighting behavior (occuring after it) in org document. INVESTIGATE (org-mode CONFIG?) FIND & FIX - [[./doc-reform.org][doc-reform.org]] [[./][org/]] * A. load abstraction - sisudoc.ocda.abstraction.load #+HEADER: :tangle "../src/sisudoc/ocda/abstraction/load.d" #+HEADER: :noweb yes #+BEGIN_SRC d <> module sisudoc.ocda.abstraction.load; @safe: /+ ↓ one way in, whatever the document is being read from spine's pipeline is markup -> abstraction -> output. once the abstraction is serialised, there is more than one thing an abstraction can be read from, and a consumer should not have to know which it was handed: . .sst / .ssm + images the markup source pod (dir) + images the same, bundled pod .zip the same, zipped .ssp + images the abstraction, as text .ocda.db the abstraction, sqlite, images inside . A source may also be remote. A URL ending in .zip or in .ocda.db is downloaded to a temp file (downloadSourceUrl in ocda/io_in/read_zip_pod.d, guarded by rgx_url_source and by --allow-downloads) and the local path is processed in its place. Those two travel because each is self-sufficient: a pod holds its source and images, a .ocda.db holds its abstraction and images. A .ssp does not, describing its images by name and digest without carrying them, so it would arrive without them and is not fetched. . This module names those sources, tells them apart, and loads the two that are self-describing artefacts, returning the value the parser produces. . The three source forms are deliberately *not* loaded here. They are the parser's job (sisudoc.ocda.meta.metadoc, spineAbstraction), and it needs the environment, the options and the configuration that spine.d assembles, none of which belongs in a loader. What this module gives that case is the dispatch and a plain statement of where to go. +/ template spineAbstractionLoad() { import std.algorithm : endsWith; import std.conv : to; import std.file; import std.path; import std.stdio; import std.string; import sisudoc.ocda.abstraction.db_in; mixin spineAbstractionDbRead; // brings the .ssp reader with it enum AbstractionSource { unknown, markup, // .sst or .ssm, with its images beside it pod, // a directory holding pod.manifest pod_zip, // that directory, zipped ssp, // .ssp, the abstraction as text, images beside it ocda_db, // .ocda.db, the abstraction as sqlite, images inside it } string abstractionSourceName(AbstractionSource _s) { final switch (_s) { case AbstractionSource.unknown: return "unknown"; case AbstractionSource.markup: return "markup source (.sst/.ssm)"; case AbstractionSource.pod: return "pod directory"; case AbstractionSource.pod_zip: return "pod zip"; case AbstractionSource.ssp: return ".ssp (abstraction as text)"; case AbstractionSource.ocda_db: return ".ocda.db (abstraction as sqlite)"; } } /+ ↓ what is this? by name, and for a directory by what it holds +/ AbstractionSource abstractionSourceOf(string _path) { if (_path.length == 0) { return AbstractionSource.unknown; } if (_path.isValidPath && _path.exists && _path.isDir) { return (_path.chainPath("pod.manifest").array.exists) ? AbstractionSource.pod : AbstractionSource.unknown; } if (_path.endsWith(".ocda.db")) { return AbstractionSource.ocda_db; } if (_path.endsWith(".ssp")) { return AbstractionSource.ssp; } if (_path.endsWith(".sst") || _path.endsWith(".ssm")) { return AbstractionSource.markup; } if (_path.endsWith(".zip")) { return AbstractionSource.pod_zip; } if (_path.endsWith(".db")) { return AbstractionSource.ocda_db; } return AbstractionSource.unknown; } struct LoadedAbstraction { AbstractionSource source; string path; bool loaded; // is .doc filled string note; // why not, when it is not SSPdocument doc; } /+ ↓ load what can be loaded from the path alone. .ssp and .ocda.db come back filled. the three source forms come back with loaded = false and a note saying where they are handled, because reading them needs the manifest, environment and configuration that spine.d builds, not a file path. +/ LoadedAbstraction abstractionLoad(string _path) { LoadedAbstraction _out; _out.path = _path; _out.source = abstractionSourceOf(_path); if (!_path.exists) { _out.note = "no such file or directory"; return _out; } final switch (_out.source) { case AbstractionSource.ssp: _out.doc = sspReadFile(_path); _out.loaded = (_out.doc.section_order.length > 0); if (!_out.loaded) { _out.note = "no object sections found"; } break; case AbstractionSource.ocda_db: _out.doc = dbReadFile(_path); _out.loaded = (_out.doc.section_order.length > 0); if (!_out.loaded) { _out.note = "no object sections found"; } break; case AbstractionSource.markup: case AbstractionSource.pod: case AbstractionSource.pod_zip: _out.note = "a source form: read by the parser," ~ " sisudoc.ocda.meta.metadoc spineAbstraction, which needs the" ~ " manifest, environment and configuration spine assembles"; break; case AbstractionSource.unknown: _out.note = "not a document source spine knows"; break; } return _out; } /+ ↓ what was loaded, in a few lines: for the eye, and for a check that a given artefact really does hold what it should +/ string[] abstractionLoadSummary(LoadedAbstraction _l) { string[] _out; _out ~= "source: " ~ abstractionSourceName(_l.source); _out ~= "path: " ~ _l.path; if (!_l.loaded) { _out ~= "loaded: no"; if (_l.note.length > 0) { _out ~= "note: " ~ _l.note; } return _out; } _out ~= "loaded: yes"; if (_l.doc.source.length > 0) { _out ~= "document: " ~ _l.doc.source; } if ("title.main" in _l.doc.meta) { _out ~= "title: " ~ _l.doc.meta["title.main"]; } if ("creator.author" in _l.doc.meta) { _out ~= "author: " ~ _l.doc.meta["creator.author"]; } _out ~= "header: " ~ _l.doc.meta.length.to!string ~ " meta, " ~ _l.doc.make.length.to!string ~ " make, " ~ _l.doc.doc_has.length.to!string ~ " doc_has"; int _objs, _citable, _headings; string[] _sections; foreach (section; _l.doc.section_order) { int _n; foreach (obj; _l.doc.abstraction[section]) { ++_objs; ++_n; if (obj.metainfo.ocn > 0) { ++_citable; } if (obj.metainfo.is_a == "heading") { ++_headings; } } _sections ~= section ~ " " ~ _n.to!string; } _out ~= "objects: " ~ _objs.to!string ~ " (" ~ _citable.to!string ~ " citable, " ~ _headings.to!string ~ " headings)"; _out ~= "sections: " ~ _sections.join(", "); return _out; } } #+END_SRC * B. read a .ssp file back into the document abstraction - sisudoc.ocda.abstraction.ssp_in #+HEADER: :tangle "../src/sisudoc/ocda/abstraction/ssp_in.d" #+HEADER: :noweb yes #+BEGIN_SRC d <> module sisudoc.ocda.abstraction.ssp_in; @safe: /+ ↓ read a .ssp file back into the document abstraction . the reverse of sisudoc.ocda.abstraction.ssp. what it returns is the same value the parser produces, ObjGenericComposite[][string], so anything that consumes the abstraction can be fed from a .ssp instead of from markup. . the check that it is faithful is a round trip: load a .ssp, emit each object again with sspObjectRecord (the writer's own definition of a record) and require the result to be byte identical to the file read. +/ template spineAbstractionRead() { import std.algorithm : startsWith, findSplit; import std.array; import std.conv : to; import std.file; import std.json : JSONValue; import std.stdio; import std.string; import sisudoc.ocda.meta.metadoc_object_setter; import sisudoc.ocda.abstraction.ssp : sspEscape; mixin sspEscape; mixin ObjectSetter; /+ ↓ what a .ssp file holds: the three header blocks as key/value, and the object sections in the order the file gives them +/ struct SSPdocument { string format; // "% SiSU Document Abstraction v<>" string source; // "% Source: ..." string[string] source_info; // the @source block string[] source_info_order; string ssp_digest; // sha256 of the .ssp read, hex string[string] meta; string[] meta_order; string[string] make; string[] make_order; string[string] doc_has; string[] doc_has_order; string[] section_order; // as found in the file ObjGenericComposite[][string] abstraction; } private int _levOfMarkedUp(string lev) { switch (lev) { case "A": return 0; case "B": return 1; case "C": return 2; case "D": return 3; case "1": return 4; case "2": return 5; case "3": return 6; case "4": return 7; default: return 9; } } private int[] _ints(string val) { int[] _out; foreach (f; val.split(" ")) { if (f.length > 0) { _out ~= f.to!int; } } return _out; } /+ ↓ the eight slot arrays are int[] with a literal default, so every default constructed object shares one array; write into a fresh one +/ private int[8] _eight(string val) { int[8] _out; foreach (i, v; _ints(val)) { if (i < _out.length) { _out[i] = v; } } return _out; } private ubyte[32] _hexToBytes(string hex) { ubyte[32] _out; if (hex.length < 64) { return _out; } foreach (i; 0..32) { _out[i] = hex[i*2 .. i*2+2].to!ubyte(16); } return _out; } SSPdocument sspRead(string[] lines) { SSPdocument doc; string section; // the @section block currently open string block; // "meta", "make", "doc_has", or "" for objects bool in_object; string[] text_lines; ObjGenericComposite obj; void _closeObject() { if (!in_object) { return; } obj.text = text_lines.join("\n"); doc.abstraction[section] ~= obj; obj = ObjGenericComposite(); text_lines = []; in_object = false; } foreach (line; lines) { if (line.length == 0) { continue; } if (line.startsWith("% SiSU Document Abstraction")) { doc.format = line; continue; } if (line.startsWith("% Source: ")) { doc.source = line["% Source: ".length .. $]; continue; } if (line.startsWith("@") && line.endsWith(" {")) { _closeObject(); string _name = line[1 .. $-2]; switch (_name) { case "source": case "meta": case "make": case "doc_has": block = _name; section = ""; break; default: block = ""; section = _name; doc.section_order ~= _name; doc.abstraction[_name] = []; break; } continue; } if (line == "}") { _closeObject(); block = ""; section = ""; continue; } if (block.length > 0) { // a header block key: value line auto _kv = line.stripLeft.findSplit(": "); // leading only: a value may end in a space if (_kv[1].length > 0) { final switch (block) { case "source": doc.source_info[_kv[0].to!string] = _kv[2].to!string; doc.source_info_order ~= _kv[0].to!string; break; case "meta": doc.meta[_kv[0].to!string] = _kv[2].to!string; doc.meta_order ~= _kv[0].to!string; break; case "make": doc.make[_kv[0].to!string] = _kv[2].to!string; doc.make_order ~= _kv[0].to!string; break; case "doc_has": doc.doc_has[_kv[0].to!string] = _kv[2].to!string; doc.doc_has_order ~= _kv[0].to!string; break; } } continue; } if (line.startsWith("[")) { // object declaration line _closeObject(); in_object = true; auto _close = line.indexOf("] "); obj.metainfo.ocn = line[1 .. _close].to!int; string _rest = line[_close + 2 .. $]; if (_rest.startsWith("heading :")) { obj.metainfo.is_a = "heading"; string _tail = _rest["heading :".length .. $]; auto _sp = _tail.indexOf(" "); if (_sp > 0) { obj.metainfo.heading_lev_markup = _levOfMarkedUp(_tail[0 .. _sp]); obj.metainfo.identifier = _tail[_sp + 1 .. $]; } else { obj.metainfo.heading_lev_markup = _levOfMarkedUp(_tail); /+ ↓ omitted means it equals the ocn, and for an object with no ocn it was empty: "a"~N identifiers are always written +/ obj.metainfo.identifier = (obj.metainfo.ocn != 0) ? obj.metainfo.ocn.to!string : ""; } } else { obj.metainfo.is_a = _rest; obj.metainfo.identifier = (obj.metainfo.ocn != 0) ? obj.metainfo.ocn.to!string : ""; } obj.metainfo.is_of_section = section; continue; } if (line.startsWith("| ")) { text_lines ~= line[2 .. $]; continue; } if (!line.startsWith(".")) { continue; } auto _p = line[1 .. $].findSplit(": "); string _key = (_p[1].length > 0) ? _p[0].to!string : line[1 .. $].to!string; string _val = (_p[1].length > 0) ? _p[2].to!string : ""; switch (_key) { case "identifier": obj.metainfo.identifier = _val; break; case "part": obj.metainfo.is_of_part = _val; break; case "section": obj.metainfo.is_of_section = _val; break; case "parent": obj.metainfo.parent_ocn = _val.to!int; break; case "last_descendant": obj.metainfo.last_descendant_ocn = _val.to!int; break; case "children": obj.metainfo.children_headings = _ints(_val); break; case "ancestors": obj.metainfo.markedup_ancestors = _eight(_val); break; case "ancestors_collapsed": obj.metainfo.collapsed_ancestors = _eight(_val); break; case "dom_status": obj.metainfo.dom_structure_markedup_tags_status = _eight(_val); break; case "dom_status_collapsed": obj.metainfo.dom_structure_collapsed_tags_status = _eight(_val); break; case "heading_lev_collapsed": obj.metainfo.heading_lev_collapsed = _val.to!int; break; case "parent_lev": obj.metainfo.parent_lev_markup = _val.to!int; break; case "dummy": obj.metainfo.dummy_heading = true; break; case "ocn_off": obj.metainfo.object_number_off = true; break; case "is_of_type": obj.metainfo.is_of_type = _val; break; case "attrib": obj.metainfo.attrib = _val; break; case "meta_lang": obj.metainfo.lang = _val; break; case "meta_syntax": obj.metainfo.syntax = _val; break; case "sha256": obj.metainfo.sha256.text = _hexToBytes(_val); break; case "image": { ST_file_name_hash_size_ _img; auto _fields = _val.split(" "); if (_fields.length > 0) { _img.fileName = _fields[0].to!string; } foreach (f; _fields[1 .. $]) { if (f == "missing:true") { _img.fileMissing = true; } else if (f.startsWith("sha256:")) { _img.fileHash_sha256 = _hexToBytes(f[7 .. $].to!string); } else if (f.startsWith("bytes:")) { _img.fileSize = f[6 .. $].to!ulong; } else if (f.startsWith("px:")) { auto _wh = f[3 .. $].findSplit("x"); if (_wh[1].length > 0) { _img.imageWidth = _wh[0].to!int; _img.imageHeight = _wh[2].to!int; } } } obj.metainfo.sha256.images ~= _img; } break; case "indent": { auto _i = _ints(_val); if (_i.length > 0) { obj.attrib.indent_base = _i[0]; } if (_i.length > 1) { obj.attrib.indent_hang = _i[1]; } } break; case "bullet": obj.attrib.bullet = true; break; case "lang": obj.attrib.language = _val; break; case "has": foreach (f; _val.split(" ")) { switch (f) { case "links": obj.has.inline_links = true; break; case "notes_reg": obj.has.inline_notes_reg = true; break; case "notes_star": obj.has.inline_notes_star = true; break; case "images": obj.has.images = true; break; case "images_no_dim": obj.has.image_without_dimensions = true; break; default: break; } } break; case "table_cols": obj.table.number_of_columns = _val.to!int; break; case "table_widths": foreach (w; _val.split(" ")) { obj.table.column_widths ~= w.to!double; } break; case "table_aligns": foreach (a; _val.split(" ")) { obj.table.column_aligns ~= a.to!string; } break; case "table_header": obj.table.heading = true; break; case "code_linenumbers": obj.code_block.linenumbers = true; break; case "stow_link": obj.stow.link ~= _val; break; case "segment": obj.tags.in_segment_html = _val; break; case "anchor": obj.tags.anchor_tag_html = _val; break; case "segment_prev": obj.tags.segname_prev = _val; break; case "segment_next": obj.tags.segname_next = _val; break; case "heading_lev_anchor": obj.tags.heading_lev_anchor_tag = _val; break; case "segment_epub": obj.tags.segment_anchor_tag_epub = _val; break; case "segment_html_is": obj.tags.html_segment_anchor_tag_is = _val; break; case "segment_epub_is": obj.tags.epub_segment_anchor_tag_is = _val; break; case "segment_lv4_is": obj.tags.segment_lv4_is = _val; break; case "heading_ancestors_text": { string[8] _h; foreach (i, h; sspUnescapeSplit(_val)) { if (i < _h.length) { _h[i] = h; } } obj.tags.heading_ancestors_text = _h; } break; case "lev4_subtoc": obj.tags.lev4_subtoc ~= _val; break; case "anchor_tag": obj.tags.anchor_tags ~= _val; break; default: break; } } _closeObject(); /+ ↓ .anchor is written only when it differs from .segment, and .section only when it differs from the enclosing block; restore both +/ foreach (sect; doc.section_order) { foreach (ref o; doc.abstraction[sect]) { if (o.tags.anchor_tag_html.length == 0 && o.tags.in_segment_html.length > 0) { o.tags.anchor_tag_html = o.tags.in_segment_html; } } } return doc; } /+ ↓ the whole document as it survives a trip through the .ssp text: emitted as lines by the writer, read straight back. what a consumer built from this cannot see, the .ssp does not carry, which is what makes a second artefact built from it unable to drift. the header blocks come back with it, so the database is filled from these and not from a second list read off doc_matters +/ SSPdocument sspRoundTripDocument(D)(D doc) { import sisudoc.ocda.abstraction.ssp; mixin spineAbstractionTxt; string[] _lines = sspDocumentLines(doc); auto _out = sspRead(_lines); /+ ↓ the writer emits one line per entry, each with its newline, so this is the digest of the file that is written beside the pod +/ _out.ssp_digest = sspDigestOfText(_lines.join("\n") ~ "\n"); return _out; } /+ ↓ sha256 of a .ssp, upper case hex, as digests.txt gives it +/ string sspDigestOfText(string _text) { import std.digest : toHexString; import std.digest.sha : sha256Of; return _text.sha256Of.toHexString.to!string; } auto sspRoundTripAbstraction(D)(D doc) { return sspRoundTripDocument(doc).abstraction; } /+ ↓ the version a reader will accept. the major part must match: a format whose properties have changed meaning is not readable by guesswork, and loading it half populated in silence is the failure this check exists to stop. a newer minor version is accepted, since a minor bump only adds properties and an unknown property line is ignored. +/ bool sspFormatVersionOk(string _format_line, string _what) { import sisudoc.ocda.abstraction.ssp : ssp_format_version, ssp_format_version_major; enum _prefix = "% SiSU Document Abstraction v"; if (_format_line.length == 0) { writeln("ERROR: ", _what, ": no format line, not a .ssp"); return false; } if (!_format_line.startsWith(_prefix)) { writeln("ERROR: ", _what, ": unrecognised format line: ", _format_line); return false; } string _ver = _format_line[_prefix.length .. $].strip; string _major = _ver.findSplit(".")[0]; if (_major != ssp_format_version_major) { writeln("ERROR: ", _what, ": .ssp format v", _ver, ", this spine reads v", ssp_format_version_major, ".x (v", ssp_format_version, "). Regenerate the abstraction from its markup."); return false; } return true; } /+ ↓ with_digest is off by default: hashing the file is only wanted where the digest is going to be recorded, and on the largest document in the sample set it is a tenth of what reading the file costs +/ @trusted SSPdocument sspReadFile(string file_path, bool with_digest = false) { string _text; string[] _lines; try { _text = (cast(char[]) file_path.read).to!string; _lines = _text.split("\n"); } catch (Exception ex) { writeln("ERROR: could not read ", file_path, ": ", ex.msg); return SSPdocument(); } auto _doc = sspRead(_lines); if (!sspFormatVersionOk(_doc.format, file_path)) { return SSPdocument(); } if (with_digest) { _doc.ssp_digest = sspDigestOfText(_text); } return _doc; } } #+END_SRC * C. read a .db file into the document abstraction - sisudoc.ocda.abstraction.db_in #+HEADER: :tangle "../src/sisudoc/ocda/abstraction/db_in.d" #+HEADER: :noweb yes #+BEGIN_SRC d <> module sisudoc.ocda.abstraction.db_in; @safe: /+ ↓ read a .ocda.db back into the document abstraction the counterpart of sisudoc.outputs.io_out.create_ocda, and the sibling of sisudoc.ocda.abstraction.ssp_in: it returns the same value, so a consumer does not have to know which of the two artefacts it was handed. the check is the same one: read a database, emit it as .ssp with the writer's own record definition, and require the result to equal the .ssp written from the same document. +/ template spineAbstractionDbRead() { import std.conv : to; import std.file; import std.stdio; import std.string; import std.array; import std.json; import d2sqlite3; import sisudoc.ocda.abstraction.ssp_in : spineAbstractionRead; mixin spineAbstractionRead; // for SSPdocument, and the shared unescape @trusted private int[8] _jsonEight(string _s) { int[8] _out; if (_s.length == 0) { return _out; } try { auto _j = parseJSON(_s); foreach (i, v; _j.array) { if (i < _out.length) { _out[i] = v.integer.to!int; } } } catch (Exception ex) {} return _out; } @trusted private int[] _jsonInts(string _s) { int[] _out; if (_s.length == 0) { return _out; } try { auto _j = parseJSON(_s); foreach (v; _j.array) { _out ~= v.integer.to!int; } } catch (Exception ex) {} return _out; } @trusted private string[] _jsonStrs(string _s) { string[] _out; if (_s.length == 0) { return _out; } try { auto _j = parseJSON(_s); foreach (v; _j.array) { _out ~= v.str; } } catch (Exception ex) {} return _out; } @trusted private double[] _jsonDoubles(string _s) { double[] _out; if (_s.length == 0) { return _out; } try { auto _j = parseJSON(_s); foreach (v; _j.array) { _out ~= (v.type == JSONType.integer) ? v.integer.to!double : v.floating; } } catch (Exception ex) {} return _out; } private ubyte[32] _hex32(string hex) { ubyte[32] _out; if (hex.length < 64) { return _out; } foreach (i; 0..32) { _out[i] = hex[i*2 .. i*2+2].to!ubyte(16); } return _out; } private string _levMarkedUp(int lev) { switch (lev) { case 0: return "A"; case 1: return "B"; case 2: return "C"; case 3: return "D"; case 4: return "1"; case 5: return "2"; case 6: return "3"; case 7: return "4"; default: return ""; } } @trusted SSPdocument dbReadFile(string db_file) { SSPdocument doc; if (!db_file.exists) { writeln("ERROR: no such file: ", db_file); return doc; } auto db = Database(db_file, SQLITE_OPEN_READONLY); /+ ↓ the four header blocks, from the metadata table. the key prefix says which block a row belongs to, and insertion order is kept +/ foreach (row; db.execute("SELECT key, value FROM metadata ORDER BY rowid")) { string _k = row["key"].as!string; string _v = row["value"].as!string; if (_k.startsWith("make.")) { string _kk = _k["make.".length .. $]; doc.make[_kk] = _v; doc.make_order ~= _kk; } else if (_k.startsWith("doc_has.")) { string _kk = _k["doc_has.".length .. $]; doc.doc_has[_kk] = _v; doc.doc_has_order ~= _kk; } else if (_k.startsWith("schema.")) { if (_k == "schema.version") { doc.format = "% SiSU Document Abstraction v" ~ _v; } } else if (_k == "source.filename") { doc.source = _v; } else if (_k == "source.ssp_digest") { /+ ↓ this database's own note of the .ssp it was built from. it is not part of the .ssp, so it does not go back into the block +/ doc.ssp_digest = _v; } else if (_k.startsWith("source.")) { string _kk = _k["source.".length .. $]; doc.source_info[_kk] = _v; doc.source_info_order ~= _kk; } else { doc.meta[_k] = _v; doc.meta_order ~= _k; } } if (!sspFormatVersionOk(doc.format, db_file)) { return SSPdocument(); } /+ ↓ the per object lists, gathered first, one sweep of each table. these four tables hold what a single object row cannot: the images an object carries, the links stowed off it, its anchor tags and its lev4 subtoc entries. They are read here in four ordered passes and kept by object id, rather than queried per object as the objects are built. The cost is then what the tables hold, not what the document holds: war and peace has 12,135 objects and 138 rows across all four, which as four statements per object was 48,540 statement preparations and made reading a .ocda.db slower than parsing the markup it came from. +/ ST_file_name_hash_size_[][long] _images_by_id; string[][long] _links_by_id; string[][long] _anchors_by_id; string[][long] _subtoc_by_id; foreach (r; db.execute( "SELECT object_id, name, bytes, sha256, width, height, missing" ~ " FROM object_images ORDER BY object_id, seq") ) { ST_file_name_hash_size_ _img; _img.fileName = r["name"].as!string; _img.fileSize = r["bytes"].as!ulong; _img.fileHash_sha256 = _hex32(r["sha256"].as!string); _img.imageWidth = r["width"].as!int; _img.imageHeight = r["height"].as!int; _img.fileMissing = (r["missing"].as!int == 1); _images_by_id[r["object_id"].as!long] ~= _img; } foreach (r; db.execute( "SELECT object_id, url FROM object_links ORDER BY object_id, seq") ) { _links_by_id[r["object_id"].as!long] ~= r["url"].as!string; } foreach (r; db.execute( "SELECT object_id, anchor FROM object_anchors ORDER BY object_id, seq") ) { _anchors_by_id[r["object_id"].as!long] ~= r["anchor"].as!string; } foreach (r; db.execute( "SELECT object_id, entry FROM object_subtoc ORDER BY object_id, seq") ) { _subtoc_by_id[r["object_id"].as!long] ~= r["entry"].as!string; } /+ ↓ the objects, section by section, in the order they were written +/ string[] _sections; foreach (row; db.execute( "SELECT section FROM objects GROUP BY section ORDER BY MIN(id)") ) { _sections ~= row["section"].as!string; } foreach (section; _sections) { doc.section_order ~= section; doc.abstraction[section] = []; /+ ↓ column name to index, resolved once for the statement. d2sqlite3's row["name"] is indexForName, a linear scan over the columns that calls sqlite3_column_name and allocates a D string for each one it passes. With some fifty named reads per object over forty columns that is around a thousand of those per object, and it was the larger half of what made reading a database slow. Resolved once here, the reads below are an integer index. +/ int[string] _col; foreach (row; db.execute( "SELECT * FROM objects WHERE section = '" ~ section ~ "' ORDER BY seq") ) { if (_col.length == 0) { foreach (_i; 0 .. row.length) { _col[row.columnName(_i)] = _i.to!int; } } ObjGenericComposite obj; long _id = row[_col["id"]].as!long; obj.metainfo.ocn = row[_col["ocn"]].as!int; obj.metainfo.is_a = row[_col["is_a"]].as!string; obj.metainfo.is_of_part = row[_col["is_of_part"]].as!string; obj.metainfo.is_of_section = (row[_col["is_of_section"]].as!string.length > 0) ? row[_col["is_of_section"]].as!string : section; obj.metainfo.is_of_type = row[_col["is_of_type"]].as!string; obj.metainfo.identifier = row[_col["identifier"]].as!string; obj.metainfo.heading_lev_markup = (obj.metainfo.is_a == "heading") ? row[_col["heading_level"]].as!int : 9; obj.metainfo.heading_lev_collapsed = (row[_col["heading_lev_collapsed"]].as!string.length > 0) ? row[_col["heading_lev_collapsed"]].as!int : 9; obj.metainfo.parent_ocn = row[_col["parent_ocn"]].as!int; obj.metainfo.parent_lev_markup = row[_col["parent_lev"]].as!int; obj.metainfo.last_descendant_ocn = row[_col["last_descendant_ocn"]].as!int; obj.metainfo.children_headings = _jsonInts(row[_col["children"]].as!string); obj.metainfo.markedup_ancestors = _jsonEight(row[_col["ancestors"]].as!string); obj.metainfo.collapsed_ancestors = _jsonEight(row[_col["ancestors_collapsed"]].as!string); obj.metainfo.dom_structure_markedup_tags_status = _jsonEight(row[_col["dom_status"]].as!string); obj.metainfo.dom_structure_collapsed_tags_status = _jsonEight(row[_col["dom_status_collapsed"]].as!string); obj.metainfo.dummy_heading = (row[_col["dummy_heading"]].as!int == 1); obj.metainfo.object_number_off = (row[_col["object_number_off"]].as!int == 1); obj.metainfo.attrib = row[_col["attrib"]].as!string; obj.metainfo.lang = row[_col["meta_lang"]].as!string; obj.metainfo.syntax = row[_col["meta_syntax"]].as!string; obj.metainfo.sha256.text = _hex32(row[_col["sha256"]].as!string); obj.attrib.indent_base = row[_col["indent_base"]].as!int; obj.attrib.indent_hang = row[_col["indent_hang"]].as!int; obj.attrib.bullet = (row[_col["bullet"]].as!int == 1); obj.attrib.language = row[_col["lang"]].as!string; obj.has.inline_links = (row[_col["has_links"]].as!int == 1); obj.has.inline_notes_reg = (row[_col["has_notes_reg"]].as!int == 1); obj.has.inline_notes_star = (row[_col["has_notes_star"]].as!int == 1); obj.has.images = (row[_col["has_images"]].as!int == 1); obj.has.image_without_dimensions = (row[_col["has_images_no_dim"]].as!int == 1); obj.tags.in_segment_html = row[_col["segment"]].as!string; obj.tags.segname_prev = row[_col["segment_prev"]].as!string; obj.tags.segname_next = row[_col["segment_next"]].as!string; obj.tags.segment_anchor_tag_epub = row[_col["segment_epub"]].as!string; obj.tags.html_segment_anchor_tag_is = row[_col["segment_html_is"]].as!string; obj.tags.epub_segment_anchor_tag_is = row[_col["segment_epub_is"]].as!string; obj.tags.segment_lv4_is = row[_col["segment_lv4_is"]].as!string; obj.tags.anchor_tag_html = row[_col["anchor"]].as!string; obj.tags.heading_lev_anchor_tag = row[_col["heading_lev_anchor"]].as!string; { string[8] _h; foreach (i, h; _jsonStrs(row[_col["heading_ancestors_text"]].as!string)) { if (i < _h.length) { _h[i] = h; } } obj.tags.heading_ancestors_text = _h; } if (obj.metainfo.is_a == "table") { obj.table.number_of_columns = row[_col["table_cols"]].as!int; obj.table.column_widths = _jsonDoubles(row[_col["table_widths"]].as!string); obj.table.column_aligns = _jsonStrs(row[_col["table_aligns"]].as!string); obj.table.heading = (row[_col["table_header"]].as!int == 1); } obj.code_block.linenumbers = (row[_col["code_linenumbers"]].as!int == 1); obj.text = row[_col["text"]].as!string; /+ ↓ the per object lists, from the sweeps above +/ if (auto _v = _id in _images_by_id) { obj.metainfo.sha256.images = *_v; } if (auto _v = _id in _links_by_id) { obj.stow.link = *_v; } if (auto _v = _id in _anchors_by_id) { obj.tags.anchor_tags = *_v; } if (auto _v = _id in _subtoc_by_id) { obj.tags.lev4_subtoc = *_v; } doc.abstraction[section] ~= obj; } } return doc; } /+ ↓ the files a database carries. a .ocda.db holds its images as blobs, which is what lets it travel on its own where a .ssp cannot: a .ssp describes its images by name, size, pixels and sha256, but does not carry them. The role column says what a file is; today only 'image' is written. +/ struct ST_ArtefactFile { string role; string name; ulong bytes; string sha256; ubyte[] data; } @trusted ST_ArtefactFile[] dbReadFiles(string db_file, string role = "image") { ST_ArtefactFile[] _out; if (!db_file.exists) { return _out; } auto db = Database(db_file, SQLITE_OPEN_READONLY); foreach (row; db.execute( "SELECT role, name, bytes, sha256, data FROM files WHERE role = '" ~ role ~ "' ORDER BY name") ) { ST_ArtefactFile _f; _f.role = row["role"].as!string; _f.name = row["name"].as!string; _f.bytes = row["bytes"].as!ulong; _f.sha256 = row["sha256"].as!string; _f.data = row["data"].as!(ubyte[]); _out ~= _f; } return _out; } } #+END_SRC * D. get doc from artefact - sisudoc.ocda.abstraction.doc_from_artefact #+HEADER: :tangle "../src/sisudoc/ocda/abstraction/doc_from_artefact.d" #+HEADER: :noweb yes #+BEGIN_SRC d <> module sisudoc.ocda.abstraction.doc_from_artefact; @safe: /+ ↓ a document, from an artefact rather than from markup spineAbstraction() parses markup and hands the output writers a doc: an abstraction and a doc_matters. This does the same from a .ssp or a .ocda.db, filling the same docMattersMake() arguments, so that what comes out is the value the writers already take and they never learn which of the two made it. . Three of the seven arguments come from the run and are the same either way: program_info, opt_action, and the conf half of conf_make_meta, which is site and run scoped (urls, papersize, the search database) and has no business in a document's own artefact. The other four come from the artefact: . conf_make_meta meta and make from the @meta and @make blocks, over the site config the run assembled doc_has docHasFromAbstraction, over the objects doc_digest the markup digest from @source manifest synthesised from the artefact's path and the document's own name and language . What is not here, deliberately: insert_file_list, which is read only by source_pod.d for --source and --pod2. No artefact carries the markup, so a pod cannot be rebuilt from one, and the empty list is the truthful answer rather than a gap. +/ template spineDocFromArtefact() { import std.algorithm : endsWith; import std.array; import std.conv : to; import std.path; import std.process : thisProcessID; import std.regex; import std.stdio; import std.string; import sisudoc.ocda.meta.conf_make_meta_structs; import sisudoc.ocda.meta.doc_matters; import sisudoc.ocda.io_in.paths_source; import sisudoc.ocda.io_in.carried_names; import sisudoc.ocda.abstraction.doc_has; import sisudoc.ocda.abstraction.load; import sisudoc.ocda.meta.topic_register; mixin spineDocMatters; mixin spineDocHasFromAbstraction; mixin spineAbstractionLoad; mixin spineTopicRegister; mixin spineCarriedNames; /+ ↓ a .ssp header key from a field name: the first underscore becomes the dot that separates the group, so title_main is title.main and rights_copyright_text is rights.copyright_text. The writer's keys are formed the same way, which is what lets this be read off the struct rather than kept as a second list to fall out of step with it. +/ private string _sspKeyOf(string _member) { foreach (i, c; _member) { if (c == '_') { return _member[0 .. i] ~ "." ~ _member[i + 1 .. $]; } } return _member; } /+ ↓ the document's own header, over the site config the run assembled. every string field of MetaComposite is looked for under its own key, so a field added there and emitted by the writer arrives here without anything being added. The arr fields are split from the string they come from, as agreed: that is parsing a value, not deriving a structure. +/ ConfComposite confMakeMetaFromHeader(S)(S _ssp, ConfComposite _site_conf) { ConfComposite _out = _site_conf; // conf: site and run scoped, kept static foreach (_m; __traits(allMembers, MetaComposite)) { static if (is(typeof(__traits(getMember, _out.meta, _m)) == string)) {{ enum string _key = _sspKeyOf(_m); if (auto _v = _key in _ssp.meta) { __traits(getMember, _out.meta, _m) = *_v; } }} } /+ ↓ title_sub is a copy of title_subtitle, made where the header is read, and is what epub3 puts in dc:title id="subtitle". It is not emitted, being the same string twice; it is remade here the way the yaml reader makes it +/ _out.meta.title_sub = _out.meta.title_subtitle; /+ ↓ creator.author is the author list joined with ", ", so splitting on it gives the list back +/ if (_out.meta.creator_author.length > 0) { foreach (_a; _out.meta.creator_author.split(", ")) { if (_a.strip.length > 0) { _out.meta.creator_author_arr ~= _a.strip; } } } /+ ↓ the topic register arrays, by the one rule the yaml reader uses: the split is not a plain one and deriving it twice would be two answers to the same question +/ { auto _tr = topicRegisterArrays(_out.meta.classify_topic_register); _out.meta.classify_topic_register_arr = _tr.arr; _out.meta.classify_topic_register_expanded_arr = _tr.expanded; } /+ ↓ make: the keys are the field names, with no group to separate +/ if (auto _v = "doc_type" in _ssp.make) { _out.make.doc_type = *_v; _out.make_str.doc_type = *_v; } if (auto _v = "auto_num_top_at_level" in _ssp.make) { _out.make.auto_num_top_at_level = *_v; } if (auto _v = "auto_num_top_lv" in _ssp.make) { try { _out.make.auto_num_top_lv = (*_v).to!int; } catch (Exception ex) {} } if (auto _v = "auto_num_depth" in _ssp.make) { try { _out.make.auto_num_depth = (*_v).to!int; } catch (Exception ex) {} } if (auto _v = "breaks" in _ssp.make) { _out.make.breaks = *_v; _out.make_str.breaks = *_v; } if (auto _v = "home_button_text" in _ssp.make) { _out.make.home_button_text = *_v; _out.make_str.home_button_text = *_v; } /+ ↓ footer is a list, joined with the separator the format escapes +/ if (auto _v = "footer" in _ssp.make) { _out.make.footer = sspUnescapeSplit(*_v); _out.make_str.footer = _out.make.footer; } return _out; } /+ ↓ where a document loaded from an artefact says it is. an artefact names its document (% Source:) and its language (@source language) but not where the pod holding its images sits, so that is taken from where the artefact itself was found: . pod//media/abstraction/.ssp -> pod// pod/.ocda.db -> pod// . which is where --pod2 puts them both. The images beside a .ssp are then found where html.d already looks; a .ocda.db carries its own and does not need the directory to exist. +/ struct ST_ArtefactPaths { string pod_dir; string src_file_with_path; } ST_ArtefactPaths artefactPaths(S)(string _artefact, S _ssp) { ST_ArtefactPaths _out; string _lang = ("language" in _ssp.source_info) ? _ssp.source_info["language"] : "en"; string _dir = _artefact.dirName; if (_artefact.endsWith(".ssp")) { /+ ↓ /media/abstraction/.ssp: two directories up is the pod +/ _out.pod_dir = (_dir.baseName == "abstraction") ? _dir.dirName.dirName : _dir; } else { /+ ↓ /.ocda.db, the pod tree beside it named for the document +/ string _uid = _artefact.baseName; foreach (_sfx; [".ocda.db", ".db"]) { if (_uid.endsWith(_sfx)) { _uid = _uid[0 .. $ - _sfx.length]; break; } } if (_uid.endsWith("." ~ _lang)) { _uid = _uid[0 .. $ - _lang.length - 1]; } _out.pod_dir = (_dir.chainPath(_uid).array).to!string; } _out.src_file_with_path = (_out.pod_dir .chainPath("media").chainPath("text").chainPath(_lang) .chainPath(_ssp.source).array).to!string; return _out; } /+ ↓ a database's images, written where the writers look for them. returns the pod directory to use in place of the one beside the artefact, or "" when the database carries no images and the pod beside it should stand. Each blob is checked against the sha256 the database recorded with it: the two were written together, so a mismatch means the file is damaged, and saying so is the point of having recorded the digest at all. +/ @trusted private string _imagesExtract(O)(string _artefact, O _opt_action) { import std.digest : toHexString; import std.digest.sha : sha256Of; import std.file : mkdirRecurse, write, tempDir; import sisudoc.ocda.abstraction.db_in : spineAbstractionDbRead; mixin spineAbstractionDbRead _dbr; auto _files = _dbr.dbReadFiles(_artefact, "image"); if (_files.length == 0) { return ""; } /+ ↓ every name checked before anything is created or written. the name comes out of the database and a database can be downloaded, so it is attacker-controlled: it can climb with "..", and chainPath drops everything before an absolute segment, so "/x" would not land under the image directory at all. One bad name means the artefact cannot be trusted for the rest of them, so this refuses the lot rather than skipping one, which is what the zip reader does with a zip. +/ foreach (_f; _files) { string _bad = validateCarriedFileName(_f.name); if (_bad.length > 0) { stderr.writeln("WARNING: ", _artefact.baseName, " carries an image spine will not write: ", _bad, "; no image is taken from this artefact"); return ""; } } string _root = (tempDir.chainPath("spine-ocda-" ~ _artefact.baseName ~ "-" ~ thisProcessID.to!string).array).to!string; string _img_dir = (_root.chainPath("media").chainPath("image").array).to!string; try { _img_dir.mkdirRecurse; } catch (Exception ex) { stderr.writeln("WARNING: could not make an image directory for ", _artefact, ": ", ex.msg); return ""; } foreach (_f; _files) { string _got = _f.data.sha256Of.toHexString.to!string; if (_f.sha256.length > 0 && _got != _f.sha256) { stderr.writeln("WARNING: image ", _f.name, " in ", _artefact.baseName, " does not match the digest recorded with it (", _f.sha256, " expected, ", _got, " found); it is written out as it stands"); } string _out_path = (_img_dir.chainPath(_f.name).array).to!string; /+ ↓ the check on the check: the name rules above already forbid a directory part, so this can only fire if they were loosened +/ if (!(carriedPathIsWithin(_img_dir, _out_path))) { stderr.writeln("WARNING: ", _f.name, " in ", _artefact.baseName, " resolves outside the image directory; no image is taken from", " this artefact"); return ""; } try { _out_path.write(_f.data); } catch (Exception ex) { stderr.writeln("WARNING: could not write ", _f.name, ": ", ex.msg); } } if (_opt_action.vox_gt_1) { writeln(" ", _files.length, " image(s) from the database, in ", _img_dir); } return _root; } /+ ↓ the images a .ssp describes, against the ones beside it. a .ssp carries no image bytes, so what it can do is say whether the files in the pod it sits in are the ones the abstraction was built from. Nothing checked that before: a changed or wrong image in the directory paired silently. +/ @trusted private void _imagesVerify(S,O)(string _pod_dir, S _ssp, O _opt_action) { import std.digest : toHexString; import std.digest.sha : sha256Of; import std.file : exists, read; string _img_dir = (_pod_dir.chainPath("media").chainPath("image").array).to!string; int _checked, _missing, _wrong; foreach (_section; _ssp.section_order) { foreach (obj; _ssp.abstraction[_section]) { foreach (_img; obj.metainfo.sha256.images) { if (_img.fileName.length == 0) { continue; } string _pth = (_img_dir.chainPath(_img.fileName).array).to!string; if (!_pth.exists) { ++_missing; continue; } ++_checked; string _want = _img.fileHash_sha256.toHexString.to!string; string _got = (cast(ubyte[]) _pth.read).sha256Of.toHexString.to!string; if (_want != _got && _want != "00000000000000000000000000000000" ~ "00000000000000000000000000000000") { ++_wrong; stderr.writeln("WARNING: image ", _img.fileName, " beside ", _ssp.source, " is not the one the abstraction was built from"); } } } } if (_missing > 0) { stderr.writeln("WARNING: ", _missing, " image(s) the abstraction names are", " not in ", _img_dir); } if (_opt_action.vox_gt_1 && _checked > 0) { writeln(" ", _checked, " image(s) checked against the abstraction's digests", (_wrong == 0) ? ", all matching" : ""); } } /+ ↓ the doc the output writers take, from an artefact. the reading is done here rather than by the caller, and that is not only tidiness: main() is @system, so a template mixed in there infers @system for everything it declares, and the output writers are @safe and will not take objects of such a type. Doing it inside this module, which is @safe:, is what makes the loaded objects the same kind of thing the parser's are. Called as spineDocFromArtefact!()(...), the way spineAbstraction!() is, rather than mixed in. +/ auto spineDocFromArtefact(P,O,E)( string _artefact, P program_info, O _opt_action, E _env, ConfComposite _site_conf, ) { auto _loaded = abstractionLoad(_artefact); auto _ssp = _loaded.doc; auto _pths = artefactPaths(_artefact, _ssp); /+ ↓ the images. a .ocda.db carries its own, which is what lets it travel on its own, so they are taken from it and written where the output writers look for images: a directory of this run's making, not the pod beside the artefact, which may be somebody's published tree. A .ssp only describes its images, so those are the ones in the pod it sits in, and are checked against the digests it recorded. +/ string _images_tmp; if (_loaded.loaded) { if (_artefact.endsWith(".ssp")) { _imagesVerify(_pths.pod_dir, _ssp, _opt_action); } else { string _extracted = _imagesExtract(_artefact, _opt_action); if (_extracted.length > 0) { /+ ↓ the source path has to move with it: image_dir_path is reached from the document's own file, not from the pod +/ _images_tmp = _extracted; _pths.pod_dir = _extracted; _pths.src_file_with_path = (_extracted .chainPath("media").chainPath("text") .chainPath(("language" in _ssp.source_info) ? _ssp.source_info["language"] : "en") .chainPath(_ssp.source).array).to!string; } } } auto _manifest = PathMatters!()(_opt_action, _env, _pths.pod_dir, _pths.src_file_with_path); auto _cmm = confMakeMetaFromHeader(_ssp, _site_conf); auto _has = docHasFromAbstraction(_ssp.abstraction, _ssp.doc_has, _opt_action); /+ ↓ the digests: the markup digest is what @source carries, and it is the one that says which source this abstraction was made from. The header and text digests are of parts an artefact does not keep separately, and are left unset rather than guessed at +/ ST_ArtefactDigest _dig; if (auto _v = "digest" in _ssp.source_info) { _dig.markup_doc = _hexToDigest(*_v); } auto _abstraction = _ssp.abstraction; auto _matters = docMattersMake( program_info, _opt_action, _manifest, _cmm, _has, _dig, string[].init, ); /+ ↓ the same shape spineAbstraction() returns, built the same way: a struct of accessors over the locals, with whether the artefact could be read at all, which the caller has to know +/ auto theDOC() { struct ST_DOC { const auto abstraction() { return _abstraction; } auto matters() { return _matters; } bool loaded() { return _loaded.loaded; } string images_tmp() { return _images_tmp; } string note() { return _loaded.note; } string source_name() { return abstractionSourceName(_loaded.source); } } return ST_DOC(); } return theDOC(); } /+ ↓ the digests doc_matters carries. Only markup_doc can be filled from an artefact, @source having recorded it; the header and text digests are of parts no artefact keeps separately and are left unset rather than guessed at. Shaped to match ST_doc_digest, which is declared inside spineRawMarkupContent and reachable only by parsing a file. +/ struct ST_ArtefactDigest { ubyte[32] markup_doc; ubyte[32] header; ubyte[32] text; } private ubyte[32] _hexToDigest(string _hex) { ubyte[32] _out; if (_hex.length < 64) { return _out; } foreach (i; 0..32) { try { _out[i] = _hex[i*2 .. i*2+2].to!ubyte(16); } catch (Exception ex) {} } return _out; } } #+END_SRC * E. Get OCD get abstracted objects for downstream processing #+HEADER: :tangle "../src/sisudoc/ocda/abstraction/doc_has.d" #+HEADER: :noweb yes #+BEGIN_SRC d <> module sisudoc.ocda.abstraction.doc_has; @safe: /+ ↓ ST_DocHas from a loaded abstraction the parser builds ST_DocHas as it goes, with the markup in front of it. A document loaded from a .ssp or a .ocda.db has to arrive at the same value from what the artefact carries, and this is where that is done. Every one of these is an *index over properties the artefact already holds*, not a re-derivation of something it does not: imagelist the .image records' filenames segnames_lv4 .anchor of each body heading at level 4 segnames_lv_0_to_4 .segment_epub of each heading at level 4 or less tag_associations .anchor, .anchor_tag and .segment_* per object section_keys_sequenced which sections are non-empty, plus the run's own output flags That distinction is the line the .ssp skill draws: assembling an index is fine and cannot drift, because a wrong index would not survive the round trip that produced its inputs; recomputing dom_status or ancestors would be re-derivation and is exactly what ocda exists to remove. If a value is not in the file and cannot be indexed out of what is, it must be added to the artefact instead of reconstructed here. The counts come from the @doc_has block rather than being recounted, for the same reason. +/ template spineDocHasFromAbstraction() { import std.algorithm : sort, uniq; import std.array; import std.conv : to; import std.json : JSONValue; import std.regex; import sisudoc.ocda.meta.metadoc_object_setter; import sisudoc.ocda.meta.rgx; mixin ObjectSetter; mixin spineRgxIn; /+ ↓ the sections, in the order a document holds them +/ enum string[] doc_sections = [ "head", "toc", "body", "endnotes", "glossary", "bibliography", "bookindex", "blurb", "tail", ]; /+ ↓ the tail sections that become segments of their own, in the order after_doc_determine_segnames appends them +/ enum string[] tail_sections = [ "endnotes", "glossary", "bibliography", "bookindex", "blurb", ]; private uint _count(string[string] _doc_has, string _key) { if (auto _v = _key in _doc_has) { try { return (*_v).to!uint; } catch (Exception ex) { return 0; } } return 0; } private bool _has_section(A)(A _abst, string _section) { if (auto _s = _section in _abst) { return (*_s).length > 1; } return false; } /+ ↓ which document sections each output format walks, and in what order. the same rule the parser applies: the three that are always there, then each tail section that the document actually has, then "tail" for the formats that close with one. It depends on the run as well as on the document, since only html and epub take the tail. +/ string[][string] sectionKeysSequenced(A,O)(A _abst, O _opt_action) { string[][string] _keys = [ "scroll": ["head", "toc", "body",], "seg": ["head", "toc", "body",], "sql": ["head", "body",], "latex": ["head", "toc", "body",], "text": ["head", "toc", "body",], ]; if (_has_section(_abst, "endnotes")) { _keys["scroll"] ~= "endnotes"; _keys["seg"] ~= "endnotes"; _keys["latex"] ~= "endnotes"; _keys["text"] ~= "endnotes"; } if (_has_section(_abst, "glossary")) { foreach (_k; ["scroll", "seg", "sql", "latex", "text"]) { _keys[_k] ~= "glossary"; } } if (_has_section(_abst, "bibliography")) { foreach (_k; ["scroll", "seg", "sql", "latex", "text"]) { _keys[_k] ~= "bibliography"; } } if (_has_section(_abst, "bookindex")) { foreach (_k; ["scroll", "seg", "sql", "latex", "text"]) { _keys[_k] ~= "bookindex"; } } if (_has_section(_abst, "blurb")) { foreach (_k; ["scroll", "seg", "sql", "latex", "text"]) { _keys[_k] ~= "blurb"; } } if (_opt_action.html_scroll || _opt_action.html_seg || _opt_action.epub) { _keys["scroll"] ~= "tail"; _keys["seg"] ~= "tail"; } return _keys; } /+ ↓ the whole of it +/ ST_DocHas docHasFromAbstraction(A,O)( A _abst, string[string] _doc_has_block, O _opt_action, ) { static auto rgx = RgxI(); ST_DocHas _out; _out.inline_links = _count(_doc_has_block, "inline_links"); _out.inline_notes_reg = _count(_doc_has_block, "inline_notes_reg"); _out.inline_notes_star = _count(_doc_has_block, "inline_notes_star"); _out.codeblocks = _count(_doc_has_block, "codeblocks"); _out.tables = _count(_doc_has_block, "tables"); _out.blocks = _count(_doc_has_block, "blocks"); _out.groups = _count(_doc_has_block, "groups"); _out.poems = _count(_doc_has_block, "poems"); _out.quotes = _count(_doc_has_block, "quotes"); string[] _images; string[string][string] _tag_assoc; /+ ↓ segnames["html"] is seeded with the toc, which is a segment of its own before any level 4 heading opens one +/ _out.segnames_lv4 ~= "toc"; foreach (_section; doc_sections) { if (_section !in _abst) { continue; } /+ ↓ head is always walked; every other section only when it holds more than the placeholder object, which is the guard the parser puts on each of its section loops +/ if (_section != "head" && !_has_section(_abst, _section)) { continue; } foreach (obj; _abst[_section]) { /+ ↓ the images this object names, as the abstraction recorded them +/ foreach (_img; obj.metainfo.sha256.images) { _images ~= _img.fileName; } /+ ↓ every object: its html anchor is in a segment, and its epub segment anchor stands for itself +/ if (obj.tags.anchor_tag_html.length > 0) { _tag_assoc[obj.tags.anchor_tag_html]["seg_lv4"] = obj.tags.in_segment_html; } if (obj.tags.segment_anchor_tag_epub.length > 0) { _tag_assoc[obj.tags.segment_anchor_tag_epub]["seg_lv1to4"] = obj.tags.segment_anchor_tag_epub; } /+ ↓ a heading's own anchor tags, and the anchors declared inline in an object's text, both point at the segment the object is in +/ /+ ↓ a heading above level 4 opens no html segment of its own, so a cross reference to it has to land on the level 4 segment that follows it. .segment is exactly that (the segment a heading opens or falls into), where .segment_html_is is the segment an object sits in; epub segments go down to level 1, so there .segment_epub stands for the heading itself +/ if (obj.tags.heading_lev_anchor_tag.length > 0 && obj.tags.in_segment_html.length > 0 && obj.metainfo.heading_lev_markup >= 4 ) { _tag_assoc[obj.tags.heading_lev_anchor_tag]["seg_lv4"] = obj.tags.in_segment_html; if (obj.tags.segment_anchor_tag_epub.length > 0) { _tag_assoc[obj.tags.heading_lev_anchor_tag]["seg_lv1to4"] = obj.tags.segment_anchor_tag_epub; } } foreach (m; obj.text.matchAll(rgx.inline_link_anchor)) { string _a = m["anchor"].to!string; if (_a.length == 0 || _a in _tag_assoc) { continue; } _tag_assoc[_a]["seg_lv4"] = obj.tags.html_segment_anchor_tag_is; _tag_assoc[_a]["seg_lv1to4"] = obj.tags.epub_segment_anchor_tag_is; } /+ ↓ the segment name lists, in document order +/ if (obj.metainfo.is_a == "heading") { if (obj.metainfo.heading_lev_markup <= 4) { _out.segnames_lv_0_to_4 ~= obj.tags.segment_anchor_tag_epub; } if (_section == "body" && obj.metainfo.heading_lev_markup == 4) { _out.segnames_lv4 ~= obj.tags.anchor_tag_html; } /+ ↓ a heading above level 4 opens no html segment of its own, so a cross reference to it lands on the level 4 segment that follows. That is .segment_lv4_is, resolved in ocda where the finished sections are to hand and carried by the artefact, so there is nothing to work out here +/ if (obj.tags.segment_lv4_is.length > 0) { /+ ↓ seg_lv4 only, and not through heading_lev_anchor_tag: that names the level 4 heading itself, whose own entry says it opens that segment rather than merely sits in it. epub segments go down to level 1, so seg_lv1to4 is untouched +/ foreach (_k; [obj.metainfo.identifier, obj.tags.anchor_tag_html]) { if (_k.length > 0) { _tag_assoc[_k]["seg_lv4"] = obj.tags.segment_lv4_is; } } } } } } foreach (_section; tail_sections) { if (_has_section(_abst, _section)) { _out.segnames_lv4 ~= _section; } } /+ ↓ the body's objects, in a pass of their own after every anchor above has been recorded, because the parser does it in that order and the order is what decides two of these entries. . each body object's identifier maps to the segment it is in. seg_lv4 is set only where it is not already, an anchor of the same name having said it *names* a segment rather than sits in one; seg_lv1to4 is set always. A heading's own anchor can be a bare number taken from its text ("2. Shorter Terms" gives the anchor "2"), which then collides with the ocn of an unrelated object, and this is the order that resolves the collision the way the parser resolves it: in free_culture, ocn 2 keeps seg_lv4 "them" from the anchor and takes seg_lv1to4 "_part_1" from itself. +/ if (_has_section(_abst, "body")) { foreach (obj; _abst["body"]) { if (obj.metainfo.identifier.length == 0) { continue; } if (!((obj.metainfo.identifier in _tag_assoc) && ("seg_lv4" in _tag_assoc[obj.metainfo.identifier])) ) { _tag_assoc[obj.metainfo.identifier]["seg_lv4"] = obj.tags.html_segment_anchor_tag_is; } /+ ↓ An object that is not written into any segment still carries an identifier, and can share it: a poem block and its verses have one ocn between them, the verses are written and the block itself is not. The block comes second, so an unguarded assignment replaced a good segment name with an empty one, and the link came out as ".xhtml#297". seg_lv4 above is guarded already, which is why this showed in the epub and not in the html. +/ if (obj.tags.epub_segment_anchor_tag_is.length > 0) { _tag_assoc[obj.metainfo.identifier]["seg_lv1to4"] = obj.tags.epub_segment_anchor_tag_is; } } } _out.imagelist = _images.sort.uniq.array; _out.tag_associations = _tag_assoc; _out.section_keys_sequenced = sectionKeysSequenced(_abst, _opt_action); return _out; } } #+END_SRC * org includes ** ocda ssp format version #+NAME: ocda_ssp_format_version #+HEADER: :noweb yes #+BEGIN_SRC emacs-lisp <<./sisudoc_spine_version_info_and_doc_header_including_copyright_and_license.org:ocda_ssp_format_version()>> #+END_SRC ** project version #+NAME: spine_version #+HEADER: :noweb yes #+BEGIN_SRC emacs-lisp <<./sisudoc_spine_version_info_and_doc_header_including_copyright_and_license.org:spine_project_version()>> #+END_SRC ** year #+NAME: year #+HEADER: :noweb yes #+BEGIN_SRC emacs-lisp <<./sisudoc_spine_version_info_and_doc_header_including_copyright_and_license.org:year()>> #+END_SRC ** document header including copyright & license #+NAME: doc_header_including_copyright_and_license #+HEADER: :noweb yes #+BEGIN_SRC emacs-lisp <<./sisudoc_spine_version_info_and_doc_header_including_copyright_and_license.org:spine_doc_header_including_copyright_and_license()>> #+END_SRC * __END__