aboutsummaryrefslogtreecommitdiffhomepage
diff options
context:
space:
mode:
-rw-r--r--org/ocda.org241
-rw-r--r--src/sisudoc/ocda/meta/ocn_align.d282
2 files changed, 523 insertions, 0 deletions
diff --git a/org/ocda.org b/org/ocda.org
index 72549af..a9f0368 100644
--- a/org/ocda.org
+++ b/org/ocda.org
@@ -1758,6 +1758,247 @@ ST_docAbstraction ret;
return ret;
#+END_SRC
+* ocda align
+
+#+HEADER: :tangle "../src/sisudoc/ocda/meta/ocn_align.d"
+#+HEADER: :noweb yes
+#+BEGIN_SRC d
+<<doc_header_including_copyright_and_license>>
+/++
+ ocn alignment across the languages of a document<br><br>
+
+ the check that the translations of a document still number their objects
+ the same way<br><br>
+
+ [sisudoc.ocda.meta.ocn_align]
++/
+module sisudoc.ocda.meta.ocn_align;
+@safe:
+/+ ↓ the languages of a document share their object numbering.
+ .
+ This is the basis of a citation in a multi-language document: an ocn names
+ the same object in every language, so a reference given in one language is
+ followable in any of them. Nothing enforces it. It holds because the
+ translations follow the source object for object, and it stops holding the
+ moment a translator drops a paragraph, merges two, or turns a heading into
+ a sentence. The document still builds, every output is produced, and the
+ numbering has quietly diverges.
+ .
+ This says so. Each language leaves a profile as it is abstracted: how
+ many numbered objects it has, and what kind of object each ocn is. After
+ the last language of a document, the profiles are compared against the
+ document's first language, the one the manifest names first.
+ .
+ Two kinds of divergence, in the order they are worth hearing:
+ .
+ count the languages do not have the same number of numbered
+ objects. Everything after the first difference is renumbered,
+ so a single dropped paragraph moves every ocn below it. This
+ is reported alone, because a kind comparison after it would
+ report hundreds of consequences of the one cause.
+ .
+ type the same ocn is a different type/kind of object. The numbering
+ still lines up, but the object at that number is a heading in
+ one language and a paragraph in another, which is the
+ signature of a heading translated as running text.
+ .
+ A warning by default; --strict makes it a failure.
+ .
+ It is deliberately not part of the abstraction. Nothing here changes what
+ a document is, and nothing is written into the .ssp: a language cannot
+ know whether it aligns with its siblings until they have all been read,
+ and an artefact that recorded it would be recording something about the
+ run rather than about the document.
++/
+/+ ↓ mixed in as a named mixin (mixin spineOcnAlign _name;) and its imports
+ are inside its functions rather than at template scope, both for the
+ same reason: what a mixin template declares at its own scope is visible
+ in the scope it is mixed into, and an import there can hijack a name
+ the host was already resolving by UFCS.
++/
+template spineOcnAlign() {
+ /+ ↓ what one language of a document leaves behind for the comparison.
+ kinds is the ocn to object kind map, and it is the whole of what a
+ language has to say here: the text is expected to differ, the
+ structure is not.
+ +/
+ struct ST_OcnProfile {
+ string doc_key; // the document, without its language
+ string lang;
+ bool is_reference; // the language the manifest names first
+ string db_file; // where to note the outcome, "" if no database
+ int ocn_max;
+ int ocn_count;
+ string[int] kinds;
+ }
+ /+ ↓ one language's answer, and why +/
+ struct ST_OcnAlignResult {
+ string doc_key;
+ string lang;
+ string reference_lang;
+ bool aligned;
+ string[] notes;
+ }
+ /+ ↓ the kind of an object, for comparison.
+ a heading carries its level, because a B heading becoming a C heading
+ is a structural change at the same ocn and is worth the same warning
+ as a heading becoming a paragraph.
+ +/
+ private string _kindOf(O)(O obj) {
+ return (obj.metainfo.is_a == "heading")
+ ? "heading:" ~ obj.metainfo.marked_up_level
+ : obj.metainfo.is_a;
+ }
+ /+ ↓ the profile of one language, taken from its abstraction.
+ only numbered objects: an object with no ocn is not citable and is not
+ part of what the languages promise each other.
+ +/
+ ST_OcnProfile ocnProfile(D)(D doc) {
+ import std.algorithm : canFind, sort;
+ import std.conv : to;
+ ST_OcnProfile _p;
+ _p.doc_key = doc.matters.src.doc_uid_out_no_lang;
+ _p.lang = doc.matters.src.language;
+ _p.is_reference = (doc.matters.pod.manifest_list_of_languages.length > 0)
+ ? (doc.matters.src.language == doc.matters.pod.manifest_list_of_languages[0])
+ : true;
+ /+ ↓ document order, the same section order the .ssp is written in, and
+ not the abstraction's key order: an associative array does not
+ promise one, and two runs over the same document would then be
+ entitled to profile it differently. Any section not on the list is
+ still swept, by name, so that a new one is not silently skipped.
+ +/
+ string[] _sections = ["head", "toc", "body", "endnotes",
+ "glossary", "bibliography", "bookindex", "blurb", "tail"];
+ string[] _extra;
+ foreach (_k; doc.abstraction.byKey) {
+ if (!_sections.canFind(_k)) { _extra ~= _k; }
+ }
+ foreach (_k; _extra.sort) { _sections ~= _k; }
+ foreach (section; _sections) {
+ if (section !in doc.abstraction) { continue; }
+ foreach (obj; doc.abstraction[section]) {
+ int _ocn = obj.metainfo.ocn.to!int;
+ if (_ocn <= 0) { continue; }
+ /+ ↓ first in document order wins. An ocn is not unique: a poem and
+ its first verse share one, by design, so the same number names
+ two objects and the first of them is the one both languages
+ will have reached first.
+ +/
+ if (_ocn !in _p.kinds) { _p.kinds[_ocn] = _kindOf(obj); }
+ _p.ocn_count += 1;
+ if (_ocn > _p.ocn_max) { _p.ocn_max = _ocn; }
+ }
+ }
+ return _p;
+ }
+ /+ ↓ how many differing ocns to name before saying "and N more".
+ enough to see the shape of the divergence, few enough that a document
+ whose translation has genuinely gone its own way does not bury the
+ rest of the run.
+ +/
+ enum size_t OCN_ALIGN_NOTES_MAX = 5;
+ /+ ↓ the comparison, for every document in the run at once.
+ the profiles arrive in whatever order the run produced them, which
+ under --parallel is not the manifest order, so the reference language
+ is taken from the flag each profile carries rather than from position.
+ +/
+ ST_OcnAlignResult[] ocnAlignCompare(ST_OcnProfile[] _profiles) {
+ import std.algorithm : sort;
+ import std.conv : to;
+ ST_OcnAlignResult[] _out;
+ ST_OcnProfile[][string] _by_doc;
+ foreach (_p; _profiles) { _by_doc[_p.doc_key] ~= _p; }
+ foreach (_doc_key; _by_doc.keys.sort) {
+ /+ ↓ by language name, not in the order the run produced them: under
+ --parallel that order is whatever the threads finished in, and a
+ check people read should say the same thing twice running
+ +/
+ auto _langs = _by_doc[_doc_key].sort!((a, b) => a.lang < b.lang).release;
+ if (_langs.length < 2) { continue; } // nothing to compare against
+ ST_OcnProfile _ref;
+ bool _have_ref = false;
+ foreach (_p; _langs) {
+ if (_p.is_reference) { _ref = _p; _have_ref = true; break; }
+ }
+ if (!_have_ref) { continue; }
+ foreach (_p; _langs) {
+ if (_p.lang == _ref.lang) { continue; }
+ ST_OcnAlignResult _r;
+ _r.doc_key = _doc_key;
+ _r.lang = _p.lang;
+ _r.reference_lang = _ref.lang;
+ _r.aligned = true;
+ if (_p.ocn_count != _ref.ocn_count) {
+ _r.aligned = false;
+ long _diff = _p.ocn_count.to!long - _ref.ocn_count.to!long;
+ _r.notes ~= "numbered objects: " ~ _ref.lang ~ " "
+ ~ _ref.ocn_count.to!string ~ ", " ~ _p.lang ~ " "
+ ~ _p.ocn_count.to!string
+ ~ " (" ~ ((_diff > 0) ? "+" : "") ~ _diff.to!string ~ ")";
+ if (_p.ocn_max != _ref.ocn_max) {
+ _r.notes ~= "highest ocn: " ~ _ref.lang ~ " "
+ ~ _ref.ocn_max.to!string ~ ", " ~ _p.lang ~ " "
+ ~ _p.ocn_max.to!string;
+ }
+ /+ ↓ no type/kind comparison after a count difference: past the first
+ dropped or added object every ocn names something else, and
+ the hundreds of lines that follow are all the one fault
+ +/
+ _out ~= _r;
+ continue;
+ }
+ size_t _kind_diffs = 0;
+ foreach (_ocn; _ref.kinds.keys.sort) {
+ if (auto _k = _ocn in _p.kinds) {
+ if (*_k == _ref.kinds[_ocn]) { continue; }
+ _kind_diffs += 1;
+ if (_kind_diffs <= OCN_ALIGN_NOTES_MAX) {
+ _r.notes ~= "ocn " ~ _ocn.to!string ~ ": "
+ ~ _ref.kinds[_ocn] ~ " in " ~ _ref.lang ~ ", "
+ ~ *_k ~ " in " ~ _p.lang;
+ }
+ } else {
+ _kind_diffs += 1;
+ if (_kind_diffs <= OCN_ALIGN_NOTES_MAX) {
+ _r.notes ~= "ocn " ~ _ocn.to!string ~ ": "
+ ~ _ref.kinds[_ocn] ~ " in " ~ _ref.lang ~ ", absent in "
+ ~ _p.lang;
+ }
+ }
+ }
+ if (_kind_diffs > OCN_ALIGN_NOTES_MAX) {
+ _r.notes ~= "and " ~ (_kind_diffs - OCN_ALIGN_NOTES_MAX).to!string
+ ~ " more";
+ }
+ if (_kind_diffs > 0) { _r.aligned = false; }
+ _out ~= _r;
+ }
+ }
+ return _out;
+ }
+ /+ ↓ the report, as the lines to print.
+ the document is named once and its languages indented under it, so a
+ run over a collection reads as a list of documents rather than a list
+ of languages.
+ +/
+ string[] ocnAlignReportLines(ST_OcnAlignResult[] _results) {
+ string[] _out;
+ string _last_doc;
+ foreach (_r; _results) {
+ if (_r.aligned) { continue; }
+ if (_r.doc_key != _last_doc) {
+ _out ~= "WARNING ocn alignment: " ~ _r.doc_key;
+ _last_doc = _r.doc_key;
+ }
+ _out ~= " [" ~ _r.lang ~ "] differs from [" ~ _r.reference_lang ~ "]";
+ foreach (_n; _r.notes) { _out ~= " " ~ _n; }
+ }
+ return _out;
+ }
+}
+#+END_SRC
+
* org includes
** project version
diff --git a/src/sisudoc/ocda/meta/ocn_align.d b/src/sisudoc/ocda/meta/ocn_align.d
new file mode 100644
index 0000000..9f9fdaf
--- /dev/null
+++ b/src/sisudoc/ocda/meta/ocn_align.d
@@ -0,0 +1,282 @@
+/+
+- Name: SisuDoc Spine, Doc Reform [a part of]
+ - Description: documents, structuring, processing, publishing, search
+ - static content generator
+
+ - Author: Ralph Amissah
+ [ralph.amissah@gmail.com]
+
+ - Copyright: (C) 2015 (continuously updated, current 2026) Ralph Amissah, All Rights Reserved.
+
+ - License: AGPL 3 or later:
+
+ Spine (SiSU), a framework for document structuring, publishing and
+ search
+
+ Copyright (C) Ralph Amissah
+
+ This program is free software: you can redistribute it and/or modify it
+ under the terms of the GNU AFERO General Public License as published by the
+ Free Software Foundation, either version 3 of the License, or (at your
+ option) any later version.
+
+ This program is distributed in the hope that it will be useful, but WITHOUT
+ ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or
+ FITNESS FOR A PARTICULAR PURPOSE. See the GNU General Public License for
+ more details.
+
+ You should have received a copy of the GNU General Public License along with
+ this program. If not, see [https://www.gnu.org/licenses/].
+
+ If you have Internet connection, the latest version of the AGPL should be
+ available at these locations:
+ [https://www.fsf.org/licensing/licenses/agpl.html]
+ [https://www.gnu.org/licenses/agpl.html]
+
+ - Spine (by Doc Reform, related to SiSU) uses standard:
+ - docReform markup syntax
+ - standard SiSU markup syntax with modified headers and minor modifications
+ - docReform object numbering
+ - standard SiSU object citation numbering & system
+
+ - Homepages:
+ [https://www.sisudoc.org]
+ [https://www.doc-reform.org]
+
+ - Git
+ [https://git.sisudoc.org/]
+
++/
+/++
+ ocn alignment across the languages of a document<br><br>
+
+ the check that the translations of a document still number their objects
+ the same way<br><br>
+
+ [sisudoc.ocda.meta.ocn_align]
++/
+module sisudoc.ocda.meta.ocn_align;
+@safe:
+/+ ↓ the languages of a document share their object numbering.
+ .
+ This is the basis of a citation in a multi-language document: an ocn names
+ the same object in every language, so a reference given in one language is
+ followable in any of them. Nothing enforces it. It holds because the
+ translations follow the source object for object, and it stops holding the
+ moment a translator drops a paragraph, merges two, or turns a heading into
+ a sentence. The document still builds, every output is produced, and the
+ numbering has quietly diverges.
+ .
+ This says so. Each language leaves a profile as it is abstracted: how
+ many numbered objects it has, and what kind of object each ocn is. After
+ the last language of a document, the profiles are compared against the
+ document's first language, the one the manifest names first.
+ .
+ Two kinds of divergence, in the order they are worth hearing:
+ .
+ count the languages do not have the same number of numbered
+ objects. Everything after the first difference is renumbered,
+ so a single dropped paragraph moves every ocn below it. This
+ is reported alone, because a kind comparison after it would
+ report hundreds of consequences of the one cause.
+ .
+ type the same ocn is a different type/kind of object. The numbering
+ still lines up, but the object at that number is a heading in
+ one language and a paragraph in another, which is the
+ signature of a heading translated as running text.
+ .
+ A warning by default; --strict makes it a failure.
+ .
+ It is deliberately not part of the abstraction. Nothing here changes what
+ a document is, and nothing is written into the .ssp: a language cannot
+ know whether it aligns with its siblings until they have all been read,
+ and an artefact that recorded it would be recording something about the
+ run rather than about the document.
++/
+/+ ↓ mixed in as a named mixin (mixin spineOcnAlign _name;) and its imports
+ are inside its functions rather than at template scope, both for the
+ same reason: what a mixin template declares at its own scope is visible
+ in the scope it is mixed into, and an import there can hijack a name
+ the host was already resolving by UFCS.
++/
+template spineOcnAlign() {
+ /+ ↓ what one language of a document leaves behind for the comparison.
+ kinds is the ocn to object kind map, and it is the whole of what a
+ language has to say here: the text is expected to differ, the
+ structure is not.
+ +/
+ struct ST_OcnProfile {
+ string doc_key; // the document, without its language
+ string lang;
+ bool is_reference; // the language the manifest names first
+ string db_file; // where to note the outcome, "" if no database
+ int ocn_max;
+ int ocn_count;
+ string[int] kinds;
+ }
+ /+ ↓ one language's answer, and why +/
+ struct ST_OcnAlignResult {
+ string doc_key;
+ string lang;
+ string reference_lang;
+ bool aligned;
+ string[] notes;
+ }
+ /+ ↓ the kind of an object, for comparison.
+ a heading carries its level, because a B heading becoming a C heading
+ is a structural change at the same ocn and is worth the same warning
+ as a heading becoming a paragraph.
+ +/
+ private string _kindOf(O)(O obj) {
+ return (obj.metainfo.is_a == "heading")
+ ? "heading:" ~ obj.metainfo.marked_up_level
+ : obj.metainfo.is_a;
+ }
+ /+ ↓ the profile of one language, taken from its abstraction.
+ only numbered objects: an object with no ocn is not citable and is not
+ part of what the languages promise each other.
+ +/
+ ST_OcnProfile ocnProfile(D)(D doc) {
+ import std.algorithm : canFind, sort;
+ import std.conv : to;
+ ST_OcnProfile _p;
+ _p.doc_key = doc.matters.src.doc_uid_out_no_lang;
+ _p.lang = doc.matters.src.language;
+ _p.is_reference = (doc.matters.pod.manifest_list_of_languages.length > 0)
+ ? (doc.matters.src.language == doc.matters.pod.manifest_list_of_languages[0])
+ : true;
+ /+ ↓ document order, the same section order the .ssp is written in, and
+ not the abstraction's key order: an associative array does not
+ promise one, and two runs over the same document would then be
+ entitled to profile it differently. Any section not on the list is
+ still swept, by name, so that a new one is not silently skipped.
+ +/
+ string[] _sections = ["head", "toc", "body", "endnotes",
+ "glossary", "bibliography", "bookindex", "blurb", "tail"];
+ string[] _extra;
+ foreach (_k; doc.abstraction.byKey) {
+ if (!_sections.canFind(_k)) { _extra ~= _k; }
+ }
+ foreach (_k; _extra.sort) { _sections ~= _k; }
+ foreach (section; _sections) {
+ if (section !in doc.abstraction) { continue; }
+ foreach (obj; doc.abstraction[section]) {
+ int _ocn = obj.metainfo.ocn.to!int;
+ if (_ocn <= 0) { continue; }
+ /+ ↓ first in document order wins. An ocn is not unique: a poem and
+ its first verse share one, by design, so the same number names
+ two objects and the first of them is the one both languages
+ will have reached first.
+ +/
+ if (_ocn !in _p.kinds) { _p.kinds[_ocn] = _kindOf(obj); }
+ _p.ocn_count += 1;
+ if (_ocn > _p.ocn_max) { _p.ocn_max = _ocn; }
+ }
+ }
+ return _p;
+ }
+ /+ ↓ how many differing ocns to name before saying "and N more".
+ enough to see the shape of the divergence, few enough that a document
+ whose translation has genuinely gone its own way does not bury the
+ rest of the run.
+ +/
+ enum size_t OCN_ALIGN_NOTES_MAX = 5;
+ /+ ↓ the comparison, for every document in the run at once.
+ the profiles arrive in whatever order the run produced them, which
+ under --parallel is not the manifest order, so the reference language
+ is taken from the flag each profile carries rather than from position.
+ +/
+ ST_OcnAlignResult[] ocnAlignCompare(ST_OcnProfile[] _profiles) {
+ import std.algorithm : sort;
+ import std.conv : to;
+ ST_OcnAlignResult[] _out;
+ ST_OcnProfile[][string] _by_doc;
+ foreach (_p; _profiles) { _by_doc[_p.doc_key] ~= _p; }
+ foreach (_doc_key; _by_doc.keys.sort) {
+ /+ ↓ by language name, not in the order the run produced them: under
+ --parallel that order is whatever the threads finished in, and a
+ check people read should say the same thing twice running
+ +/
+ auto _langs = _by_doc[_doc_key].sort!((a, b) => a.lang < b.lang).release;
+ if (_langs.length < 2) { continue; } // nothing to compare against
+ ST_OcnProfile _ref;
+ bool _have_ref = false;
+ foreach (_p; _langs) {
+ if (_p.is_reference) { _ref = _p; _have_ref = true; break; }
+ }
+ if (!_have_ref) { continue; }
+ foreach (_p; _langs) {
+ if (_p.lang == _ref.lang) { continue; }
+ ST_OcnAlignResult _r;
+ _r.doc_key = _doc_key;
+ _r.lang = _p.lang;
+ _r.reference_lang = _ref.lang;
+ _r.aligned = true;
+ if (_p.ocn_count != _ref.ocn_count) {
+ _r.aligned = false;
+ long _diff = _p.ocn_count.to!long - _ref.ocn_count.to!long;
+ _r.notes ~= "numbered objects: " ~ _ref.lang ~ " "
+ ~ _ref.ocn_count.to!string ~ ", " ~ _p.lang ~ " "
+ ~ _p.ocn_count.to!string
+ ~ " (" ~ ((_diff > 0) ? "+" : "") ~ _diff.to!string ~ ")";
+ if (_p.ocn_max != _ref.ocn_max) {
+ _r.notes ~= "highest ocn: " ~ _ref.lang ~ " "
+ ~ _ref.ocn_max.to!string ~ ", " ~ _p.lang ~ " "
+ ~ _p.ocn_max.to!string;
+ }
+ /+ ↓ no type/kind comparison after a count difference: past the first
+ dropped or added object every ocn names something else, and
+ the hundreds of lines that follow are all the one fault
+ +/
+ _out ~= _r;
+ continue;
+ }
+ size_t _kind_diffs = 0;
+ foreach (_ocn; _ref.kinds.keys.sort) {
+ if (auto _k = _ocn in _p.kinds) {
+ if (*_k == _ref.kinds[_ocn]) { continue; }
+ _kind_diffs += 1;
+ if (_kind_diffs <= OCN_ALIGN_NOTES_MAX) {
+ _r.notes ~= "ocn " ~ _ocn.to!string ~ ": "
+ ~ _ref.kinds[_ocn] ~ " in " ~ _ref.lang ~ ", "
+ ~ *_k ~ " in " ~ _p.lang;
+ }
+ } else {
+ _kind_diffs += 1;
+ if (_kind_diffs <= OCN_ALIGN_NOTES_MAX) {
+ _r.notes ~= "ocn " ~ _ocn.to!string ~ ": "
+ ~ _ref.kinds[_ocn] ~ " in " ~ _ref.lang ~ ", absent in "
+ ~ _p.lang;
+ }
+ }
+ }
+ if (_kind_diffs > OCN_ALIGN_NOTES_MAX) {
+ _r.notes ~= "and " ~ (_kind_diffs - OCN_ALIGN_NOTES_MAX).to!string
+ ~ " more";
+ }
+ if (_kind_diffs > 0) { _r.aligned = false; }
+ _out ~= _r;
+ }
+ }
+ return _out;
+ }
+ /+ ↓ the report, as the lines to print.
+ the document is named once and its languages indented under it, so a
+ run over a collection reads as a list of documents rather than a list
+ of languages.
+ +/
+ string[] ocnAlignReportLines(ST_OcnAlignResult[] _results) {
+ string[] _out;
+ string _last_doc;
+ foreach (_r; _results) {
+ if (_r.aligned) { continue; }
+ if (_r.doc_key != _last_doc) {
+ _out ~= "WARNING ocn alignment: " ~ _r.doc_key;
+ _last_doc = _r.doc_key;
+ }
+ _out ~= " [" ~ _r.lang ~ "] differs from [" ~ _r.reference_lang ~ "]";
+ foreach (_n; _r.notes) { _out ~= " " ~ _n; }
+ }
+ return _out;
+ }
+}