aboutsummaryrefslogtreecommitdiffhomepage
path: root/src
diff options
context:
space:
mode:
authorRalph Amissah <ralph.amissah@gmail.com>2026-09-11 12:02:23 -0400
committerRalph Amissah <ralph.amissah@gmail.com>2026-09-12 12:14:39 -0400
commit3e23d7082e20d28971c7785db3c5e9dd724167e6 (patch)
treed24bdc717be56f3132934ebe485a8e90f6a923b0 /src
parentocda: what a document has, as plain data (diff)
ocda: ST_DocHas from a loaded abstraction
docHasFromAbstraction() builds ST_DocHas from the objects and the @doc_has block, so a document read from a .ssp or a .ocda.db can reach the value the parser reaches. Everything it builds is an index over properties the artefact already carries, not a recomputation of something it does not: the counts come from the header, imagelist from the .image records, the segment name lists and the tag associations from .anchor, .segment, .segment_epub, .heading_lev_anchor and .segment_*_is, and section_keys_sequenced from which sections are non-empty plus the run's own flags. *The two defect fixes this now sits on did most of the closing.* Measured before them and after, over some 30,000 tag_associations entries and the 1,651 keys a document actually links to: imagelist 8 documents differed, now none. The parser was the one that was wrong, and rgx.image being anchored to image markup brought it into line with what the .image records always said. key sets four keys differed (_part_eof, "0", the empty key, and "toc" the other way about), now none. _part_eof came back the moment @tail was carried, and the rest with it. values 89 link targets resolved differently, now 1. The one left is free_culture's ocn 5, and it is the case this cannot reach: a heading above level 4 takes its segment by back-filling when the next level 4 heading arrives, so the object never holds the answer and reading the objects in order only approximates it. Fixed properly in the commit that follows, in ocda where the field is made, rather than guessed at here. (assisted by Claude-Code)
Diffstat (limited to 'src')
-rw-r--r--src/sisudoc/ocda/abstraction/doc_has.d269
-rw-r--r--src/sisudoc/ocda/abstraction/package.d1
2 files changed, 270 insertions, 0 deletions
diff --git a/src/sisudoc/ocda/abstraction/doc_has.d b/src/sisudoc/ocda/abstraction/doc_has.d
new file mode 100644
index 0000000..75a4640
--- /dev/null
+++ b/src/sisudoc/ocda/abstraction/doc_has.d
@@ -0,0 +1,269 @@
+/+
+- Name: SisuDoc Spine, Doc Reform [a part of]
+ - Description: documents, structuring, processing, publishing, search
+ - static content generator
+
+ - Author: Ralph Amissah
+ [ralph.amissah@gmail.com]
+
+ - Copyright: (C) 2015 (continuously updated, current 2026) Ralph Amissah, All Rights Reserved.
+
+ - License: AGPL 3 or later:
+
+ Spine (SiSU), a framework for document structuring, publishing and
+ search
+
+ Copyright (C) Ralph Amissah
+
+ This program is free software: you can redistribute it and/or modify it
+ under the terms of the GNU AFERO General Public License as published by the
+ Free Software Foundation, either version 3 of the License, or (at your
+ option) any later version.
+
+ This program is distributed in the hope that it will be useful, but WITHOUT
+ ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or
+ FITNESS FOR A PARTICULAR PURPOSE. See the GNU General Public License for
+ more details.
+
+ You should have received a copy of the GNU General Public License along with
+ this program. If not, see [https://www.gnu.org/licenses/].
+
+ If you have Internet connection, the latest version of the AGPL should be
+ available at these locations:
+ [https://www.fsf.org/licensing/licenses/agpl.html]
+ [https://www.gnu.org/licenses/agpl.html]
+
+ - Spine (by Doc Reform, related to SiSU) uses standard:
+ - docReform markup syntax
+ - standard SiSU markup syntax with modified headers and minor modifications
+ - docReform object numbering
+ - standard SiSU object citation numbering & system
+
+ - Homepages:
+ [https://www.sisudoc.org]
+ [https://www.doc-reform.org]
+
+ - Git
+ [https://git.sisudoc.org/]
+
++/
+module sisudoc.ocda.abstraction.doc_has;
+@safe:
+/+ ↓ ST_DocHas from a loaded abstraction
+
+ the parser builds ST_DocHas as it goes, with the markup in front of it.
+ A document loaded from a .ssp or a .ocda.db has to arrive at the same
+ value from what the artefact carries, and this is where that is done.
+
+ Every one of these is an *index over properties the artefact already
+ holds*, not a re-derivation of something it does not:
+
+ imagelist the .image records' filenames
+ segnames_lv4 .anchor of each body heading at level 4
+ segnames_lv_0_to_4 .segment_epub of each heading at level 4 or less
+ tag_associations .anchor, .anchor_tag and .segment_* per object
+ section_keys_sequenced which sections are non-empty, plus the run's
+ own output flags
+
+ That distinction is the line the .ssp skill draws: assembling an index
+ is fine and cannot drift, because a wrong index would not survive the
+ round trip that produced its inputs; recomputing dom_status or ancestors
+ would be re-derivation and is exactly what ocda exists to remove. If a
+ value is not in the file and cannot be indexed out of what is, it must
+ be added to the artefact instead of reconstructed here.
+
+ The counts come from the @doc_has block rather than being recounted, for
+ the same reason.
++/
+template spineDocHasFromAbstraction() {
+ import std.algorithm : sort, uniq;
+ import std.array;
+ import std.conv : to;
+ import std.json : JSONValue;
+ import std.regex;
+ import sisudoc.ocda.meta.metadoc_object_setter;
+ import sisudoc.ocda.meta.rgx;
+ mixin ObjectSetter;
+ mixin spineRgxIn;
+ /+ ↓ the sections, in the order a document holds them +/
+ enum string[] doc_sections = [
+ "head", "toc", "body", "endnotes",
+ "glossary", "bibliography", "bookindex", "blurb", "tail",
+ ];
+ /+ ↓ the tail sections that become segments of their own, in the order
+ after_doc_determine_segnames appends them +/
+ enum string[] tail_sections = [
+ "endnotes", "glossary", "bibliography", "bookindex", "blurb",
+ ];
+ private uint _count(string[string] _doc_has, string _key) {
+ if (auto _v = _key in _doc_has) {
+ try { return (*_v).to!uint; } catch (Exception ex) { return 0; }
+ }
+ return 0;
+ }
+ private bool _has_section(A)(A _abst, string _section) {
+ if (auto _s = _section in _abst) { return (*_s).length > 1; }
+ return false;
+ }
+ /+ ↓ which document sections each output format walks, and in what order.
+
+ the same rule the parser applies: the three that are always there,
+ then each tail section that the document actually has, then "tail"
+ for the formats that close with one. It depends on the run as well as
+ on the document, since only html and epub take the tail.
+ +/
+ string[][string] sectionKeysSequenced(A,O)(A _abst, O _opt_action) {
+ string[][string] _keys = [
+ "scroll": ["head", "toc", "body",],
+ "seg": ["head", "toc", "body",],
+ "sql": ["head", "body",],
+ "latex": ["head", "toc", "body",],
+ "text": ["head", "toc", "body",],
+ ];
+ if (_has_section(_abst, "endnotes")) {
+ _keys["scroll"] ~= "endnotes";
+ _keys["seg"] ~= "endnotes";
+ _keys["latex"] ~= "endnotes";
+ _keys["text"] ~= "endnotes";
+ }
+ if (_has_section(_abst, "glossary")) {
+ foreach (_k; ["scroll", "seg", "sql", "latex", "text"]) { _keys[_k] ~= "glossary"; }
+ }
+ if (_has_section(_abst, "bibliography")) {
+ foreach (_k; ["scroll", "seg", "sql", "latex", "text"]) { _keys[_k] ~= "bibliography"; }
+ }
+ if (_has_section(_abst, "bookindex")) {
+ foreach (_k; ["scroll", "seg", "sql", "latex", "text"]) { _keys[_k] ~= "bookindex"; }
+ }
+ if (_has_section(_abst, "blurb")) {
+ foreach (_k; ["scroll", "seg", "sql", "latex", "text"]) { _keys[_k] ~= "blurb"; }
+ }
+ if (_opt_action.html_scroll || _opt_action.html_seg || _opt_action.epub) {
+ _keys["scroll"] ~= "tail";
+ _keys["seg"] ~= "tail";
+ }
+ return _keys;
+ }
+ /+ ↓ the whole of it +/
+ ST_DocHas docHasFromAbstraction(A,O)(
+ A _abst,
+ string[string] _doc_has_block,
+ O _opt_action,
+ ) {
+ static auto rgx = RgxI();
+ ST_DocHas _out;
+ _out.inline_links = _count(_doc_has_block, "inline_links");
+ _out.inline_notes_reg = _count(_doc_has_block, "inline_notes_reg");
+ _out.inline_notes_star = _count(_doc_has_block, "inline_notes_star");
+ _out.codeblocks = _count(_doc_has_block, "codeblocks");
+ _out.tables = _count(_doc_has_block, "tables");
+ _out.blocks = _count(_doc_has_block, "blocks");
+ _out.groups = _count(_doc_has_block, "groups");
+ _out.poems = _count(_doc_has_block, "poems");
+ _out.quotes = _count(_doc_has_block, "quotes");
+ string[] _images;
+ string[string][string] _tag_assoc;
+ /+ ↓ keys waiting for the level 4 segment that will hold them.
+
+ a heading above level 4 opens no html segment, so a cross reference
+ to it has to land on the next level 4 heading, which has not been
+ seen yet when the heading is read. The parser does this the same
+ way, back-filling a pending tag when the level 4 heading arrives
+ (metadoc_from_src.d, "tag_assoc[lv0_to_lv3_html_tag]"), and the
+ order of the objects is what both are reading it from.
+ +/
+ string[] _awaiting_lv4;
+ void _backfill(string _seg) {
+ foreach (_k; _awaiting_lv4) { _tag_assoc[_k]["seg_lv4"] = _seg; }
+ _awaiting_lv4 = [];
+ }
+ /+ ↓ segnames["html"] is seeded with the toc, which is a segment of its
+ own before any level 4 heading opens one +/
+ _out.segnames_lv4 ~= "toc";
+ foreach (_section; doc_sections) {
+ if (_section !in _abst) { continue; }
+ /+ ↓ head is always walked; every other section only when it holds
+ more than the placeholder object, which is the guard the parser
+ puts on each of its section loops
+ +/
+ if (_section != "head" && !_has_section(_abst, _section)) { continue; }
+ foreach (obj; _abst[_section]) {
+ /+ ↓ the images this object names, as the abstraction recorded them +/
+ foreach (_img; obj.metainfo.sha256.images) { _images ~= _img.fileName; }
+ /+ ↓ every object: its html anchor is in a segment, and its epub
+ segment anchor stands for itself +/
+ if (obj.tags.anchor_tag_html.length > 0) {
+ _tag_assoc[obj.tags.anchor_tag_html]["seg_lv4"] = obj.tags.in_segment_html;
+ }
+ if (obj.tags.segment_anchor_tag_epub.length > 0) {
+ _tag_assoc[obj.tags.segment_anchor_tag_epub]["seg_lv1to4"]
+ = obj.tags.segment_anchor_tag_epub;
+ }
+ /+ ↓ the body's objects are what a citation or a cross reference
+ points at, so each one's identifier maps to the segment it is
+ in. seg_lv4 is only set where it is not already, since an
+ anchor of the same name set above names a segment rather than
+ sits in one
+ +/
+ if (_section == "body" && obj.metainfo.identifier.length > 0) {
+ if (!((obj.metainfo.identifier in _tag_assoc)
+ && ("seg_lv4" in _tag_assoc[obj.metainfo.identifier]))
+ ) {
+ _tag_assoc[obj.metainfo.identifier]["seg_lv4"]
+ = obj.tags.html_segment_anchor_tag_is;
+ }
+ _tag_assoc[obj.metainfo.identifier]["seg_lv1to4"]
+ = obj.tags.epub_segment_anchor_tag_is;
+ }
+ /+ ↓ a heading's own anchor tags, and the anchors declared inline in
+ an object's text, both point at the segment the object is in +/
+ /+ ↓ a heading above level 4 opens no html segment of its own, so a
+ cross reference to it has to land on the level 4 segment that
+ follows it. .segment is exactly that (the segment a heading
+ opens or falls into), where .segment_html_is is the segment an
+ object sits in; epub segments go down to level 1, so there
+ .segment_epub stands for the heading itself
+ +/
+ if (obj.tags.heading_lev_anchor_tag.length > 0
+ && obj.tags.in_segment_html.length > 0
+ ) {
+ _tag_assoc[obj.tags.heading_lev_anchor_tag]["seg_lv4"]
+ = obj.tags.in_segment_html;
+ if (obj.tags.segment_anchor_tag_epub.length > 0) {
+ _tag_assoc[obj.tags.heading_lev_anchor_tag]["seg_lv1to4"]
+ = obj.tags.segment_anchor_tag_epub;
+ }
+ }
+ foreach (m; obj.text.matchAll(rgx.inline_link_anchor)) {
+ string _a = m["anchor"].to!string;
+ if (_a.length == 0 || _a in _tag_assoc) { continue; }
+ _tag_assoc[_a]["seg_lv4"] = obj.tags.html_segment_anchor_tag_is;
+ _tag_assoc[_a]["seg_lv1to4"] = obj.tags.epub_segment_anchor_tag_is;
+ }
+ /+ ↓ the segment name lists, in document order +/
+ if (obj.metainfo.is_a == "heading") {
+ if (obj.metainfo.heading_lev_markup <= 4) {
+ _out.segnames_lv_0_to_4 ~= obj.tags.segment_anchor_tag_epub;
+ }
+ if (obj.metainfo.heading_lev_markup == 4) {
+ if (_section == "body") { _out.segnames_lv4 ~= obj.tags.anchor_tag_html; }
+ /+ ↓ this is the segment the headings above it were waiting for +/
+ _backfill(obj.tags.in_segment_html);
+ } else if (obj.metainfo.heading_lev_markup < 4) {
+ foreach (_k; [obj.metainfo.identifier, obj.tags.heading_lev_anchor_tag,
+ obj.tags.anchor_tag_html]) {
+ if (_k.length > 0) { _awaiting_lv4 ~= _k; }
+ }
+ }
+ }
+ }
+ }
+ foreach (_section; tail_sections) {
+ if (_has_section(_abst, _section)) { _out.segnames_lv4 ~= _section; }
+ }
+ _out.imagelist = _images.sort.uniq.array;
+ _out.tag_associations = _tag_assoc;
+ _out.section_keys_sequenced = sectionKeysSequenced(_abst, _opt_action);
+ return _out;
+ }
+}
diff --git a/src/sisudoc/ocda/abstraction/package.d b/src/sisudoc/ocda/abstraction/package.d
index 4131a90..05897f6 100644
--- a/src/sisudoc/ocda/abstraction/package.d
+++ b/src/sisudoc/ocda/abstraction/package.d
@@ -87,3 +87,4 @@ public import sisudoc.ocda.abstraction.ssp; // spineAbstractionTxt (.ssp)
public import sisudoc.ocda.abstraction.ssp_in; // spineAbstractionRead (.ssp back in)
public import sisudoc.ocda.abstraction.db_in; // spineAbstractionDbRead (.db back in)
public import sisudoc.ocda.abstraction.load; // spineAbstractionLoad (one way in)
+public import sisudoc.ocda.abstraction.doc_has; // spineDocHasFromAbstraction (ST_DocHas from a loaded abstraction)