diff options
| -rw-r--r-- | org/ocda.org | 241 | ||||
| -rw-r--r-- | src/sisudoc/ocda/meta/ocn_align.d | 282 |
2 files changed, 523 insertions, 0 deletions
diff --git a/org/ocda.org b/org/ocda.org index 72549af..a9f0368 100644 --- a/org/ocda.org +++ b/org/ocda.org @@ -1758,6 +1758,247 @@ ST_docAbstraction ret; return ret; #+END_SRC +* ocda align + +#+HEADER: :tangle "../src/sisudoc/ocda/meta/ocn_align.d" +#+HEADER: :noweb yes +#+BEGIN_SRC d +<<doc_header_including_copyright_and_license>> +/++ + ocn alignment across the languages of a document<br><br> + + the check that the translations of a document still number their objects + the same way<br><br> + + [sisudoc.ocda.meta.ocn_align] ++/ +module sisudoc.ocda.meta.ocn_align; +@safe: +/+ ↓ the languages of a document share their object numbering. + . + This is the basis of a citation in a multi-language document: an ocn names + the same object in every language, so a reference given in one language is + followable in any of them. Nothing enforces it. It holds because the + translations follow the source object for object, and it stops holding the + moment a translator drops a paragraph, merges two, or turns a heading into + a sentence. The document still builds, every output is produced, and the + numbering has quietly diverges. + . + This says so. Each language leaves a profile as it is abstracted: how + many numbered objects it has, and what kind of object each ocn is. After + the last language of a document, the profiles are compared against the + document's first language, the one the manifest names first. + . + Two kinds of divergence, in the order they are worth hearing: + . + count the languages do not have the same number of numbered + objects. Everything after the first difference is renumbered, + so a single dropped paragraph moves every ocn below it. This + is reported alone, because a kind comparison after it would + report hundreds of consequences of the one cause. + . + type the same ocn is a different type/kind of object. The numbering + still lines up, but the object at that number is a heading in + one language and a paragraph in another, which is the + signature of a heading translated as running text. + . + A warning by default; --strict makes it a failure. + . + It is deliberately not part of the abstraction. Nothing here changes what + a document is, and nothing is written into the .ssp: a language cannot + know whether it aligns with its siblings until they have all been read, + and an artefact that recorded it would be recording something about the + run rather than about the document. ++/ +/+ ↓ mixed in as a named mixin (mixin spineOcnAlign _name;) and its imports + are inside its functions rather than at template scope, both for the + same reason: what a mixin template declares at its own scope is visible + in the scope it is mixed into, and an import there can hijack a name + the host was already resolving by UFCS. ++/ +template spineOcnAlign() { + /+ ↓ what one language of a document leaves behind for the comparison. + kinds is the ocn to object kind map, and it is the whole of what a + language has to say here: the text is expected to differ, the + structure is not. + +/ + struct ST_OcnProfile { + string doc_key; // the document, without its language + string lang; + bool is_reference; // the language the manifest names first + string db_file; // where to note the outcome, "" if no database + int ocn_max; + int ocn_count; + string[int] kinds; + } + /+ ↓ one language's answer, and why +/ + struct ST_OcnAlignResult { + string doc_key; + string lang; + string reference_lang; + bool aligned; + string[] notes; + } + /+ ↓ the kind of an object, for comparison. + a heading carries its level, because a B heading becoming a C heading + is a structural change at the same ocn and is worth the same warning + as a heading becoming a paragraph. + +/ + private string _kindOf(O)(O obj) { + return (obj.metainfo.is_a == "heading") + ? "heading:" ~ obj.metainfo.marked_up_level + : obj.metainfo.is_a; + } + /+ ↓ the profile of one language, taken from its abstraction. + only numbered objects: an object with no ocn is not citable and is not + part of what the languages promise each other. + +/ + ST_OcnProfile ocnProfile(D)(D doc) { + import std.algorithm : canFind, sort; + import std.conv : to; + ST_OcnProfile _p; + _p.doc_key = doc.matters.src.doc_uid_out_no_lang; + _p.lang = doc.matters.src.language; + _p.is_reference = (doc.matters.pod.manifest_list_of_languages.length > 0) + ? (doc.matters.src.language == doc.matters.pod.manifest_list_of_languages[0]) + : true; + /+ ↓ document order, the same section order the .ssp is written in, and + not the abstraction's key order: an associative array does not + promise one, and two runs over the same document would then be + entitled to profile it differently. Any section not on the list is + still swept, by name, so that a new one is not silently skipped. + +/ + string[] _sections = ["head", "toc", "body", "endnotes", + "glossary", "bibliography", "bookindex", "blurb", "tail"]; + string[] _extra; + foreach (_k; doc.abstraction.byKey) { + if (!_sections.canFind(_k)) { _extra ~= _k; } + } + foreach (_k; _extra.sort) { _sections ~= _k; } + foreach (section; _sections) { + if (section !in doc.abstraction) { continue; } + foreach (obj; doc.abstraction[section]) { + int _ocn = obj.metainfo.ocn.to!int; + if (_ocn <= 0) { continue; } + /+ ↓ first in document order wins. An ocn is not unique: a poem and + its first verse share one, by design, so the same number names + two objects and the first of them is the one both languages + will have reached first. + +/ + if (_ocn !in _p.kinds) { _p.kinds[_ocn] = _kindOf(obj); } + _p.ocn_count += 1; + if (_ocn > _p.ocn_max) { _p.ocn_max = _ocn; } + } + } + return _p; + } + /+ ↓ how many differing ocns to name before saying "and N more". + enough to see the shape of the divergence, few enough that a document + whose translation has genuinely gone its own way does not bury the + rest of the run. + +/ + enum size_t OCN_ALIGN_NOTES_MAX = 5; + /+ ↓ the comparison, for every document in the run at once. + the profiles arrive in whatever order the run produced them, which + under --parallel is not the manifest order, so the reference language + is taken from the flag each profile carries rather than from position. + +/ + ST_OcnAlignResult[] ocnAlignCompare(ST_OcnProfile[] _profiles) { + import std.algorithm : sort; + import std.conv : to; + ST_OcnAlignResult[] _out; + ST_OcnProfile[][string] _by_doc; + foreach (_p; _profiles) { _by_doc[_p.doc_key] ~= _p; } + foreach (_doc_key; _by_doc.keys.sort) { + /+ ↓ by language name, not in the order the run produced them: under + --parallel that order is whatever the threads finished in, and a + check people read should say the same thing twice running + +/ + auto _langs = _by_doc[_doc_key].sort!((a, b) => a.lang < b.lang).release; + if (_langs.length < 2) { continue; } // nothing to compare against + ST_OcnProfile _ref; + bool _have_ref = false; + foreach (_p; _langs) { + if (_p.is_reference) { _ref = _p; _have_ref = true; break; } + } + if (!_have_ref) { continue; } + foreach (_p; _langs) { + if (_p.lang == _ref.lang) { continue; } + ST_OcnAlignResult _r; + _r.doc_key = _doc_key; + _r.lang = _p.lang; + _r.reference_lang = _ref.lang; + _r.aligned = true; + if (_p.ocn_count != _ref.ocn_count) { + _r.aligned = false; + long _diff = _p.ocn_count.to!long - _ref.ocn_count.to!long; + _r.notes ~= "numbered objects: " ~ _ref.lang ~ " " + ~ _ref.ocn_count.to!string ~ ", " ~ _p.lang ~ " " + ~ _p.ocn_count.to!string + ~ " (" ~ ((_diff > 0) ? "+" : "") ~ _diff.to!string ~ ")"; + if (_p.ocn_max != _ref.ocn_max) { + _r.notes ~= "highest ocn: " ~ _ref.lang ~ " " + ~ _ref.ocn_max.to!string ~ ", " ~ _p.lang ~ " " + ~ _p.ocn_max.to!string; + } + /+ ↓ no type/kind comparison after a count difference: past the first + dropped or added object every ocn names something else, and + the hundreds of lines that follow are all the one fault + +/ + _out ~= _r; + continue; + } + size_t _kind_diffs = 0; + foreach (_ocn; _ref.kinds.keys.sort) { + if (auto _k = _ocn in _p.kinds) { + if (*_k == _ref.kinds[_ocn]) { continue; } + _kind_diffs += 1; + if (_kind_diffs <= OCN_ALIGN_NOTES_MAX) { + _r.notes ~= "ocn " ~ _ocn.to!string ~ ": " + ~ _ref.kinds[_ocn] ~ " in " ~ _ref.lang ~ ", " + ~ *_k ~ " in " ~ _p.lang; + } + } else { + _kind_diffs += 1; + if (_kind_diffs <= OCN_ALIGN_NOTES_MAX) { + _r.notes ~= "ocn " ~ _ocn.to!string ~ ": " + ~ _ref.kinds[_ocn] ~ " in " ~ _ref.lang ~ ", absent in " + ~ _p.lang; + } + } + } + if (_kind_diffs > OCN_ALIGN_NOTES_MAX) { + _r.notes ~= "and " ~ (_kind_diffs - OCN_ALIGN_NOTES_MAX).to!string + ~ " more"; + } + if (_kind_diffs > 0) { _r.aligned = false; } + _out ~= _r; + } + } + return _out; + } + /+ ↓ the report, as the lines to print. + the document is named once and its languages indented under it, so a + run over a collection reads as a list of documents rather than a list + of languages. + +/ + string[] ocnAlignReportLines(ST_OcnAlignResult[] _results) { + string[] _out; + string _last_doc; + foreach (_r; _results) { + if (_r.aligned) { continue; } + if (_r.doc_key != _last_doc) { + _out ~= "WARNING ocn alignment: " ~ _r.doc_key; + _last_doc = _r.doc_key; + } + _out ~= " [" ~ _r.lang ~ "] differs from [" ~ _r.reference_lang ~ "]"; + foreach (_n; _r.notes) { _out ~= " " ~ _n; } + } + return _out; + } +} +#+END_SRC + * org includes ** project version diff --git a/src/sisudoc/ocda/meta/ocn_align.d b/src/sisudoc/ocda/meta/ocn_align.d new file mode 100644 index 0000000..9f9fdaf --- /dev/null +++ b/src/sisudoc/ocda/meta/ocn_align.d @@ -0,0 +1,282 @@ +/+ +- Name: SisuDoc Spine, Doc Reform [a part of] + - Description: documents, structuring, processing, publishing, search + - static content generator + + - Author: Ralph Amissah + [ralph.amissah@gmail.com] + + - Copyright: (C) 2015 (continuously updated, current 2026) Ralph Amissah, All Rights Reserved. + + - License: AGPL 3 or later: + + Spine (SiSU), a framework for document structuring, publishing and + search + + Copyright (C) Ralph Amissah + + This program is free software: you can redistribute it and/or modify it + under the terms of the GNU AFERO General Public License as published by the + Free Software Foundation, either version 3 of the License, or (at your + option) any later version. + + This program is distributed in the hope that it will be useful, but WITHOUT + ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or + FITNESS FOR A PARTICULAR PURPOSE. See the GNU General Public License for + more details. + + You should have received a copy of the GNU General Public License along with + this program. If not, see [https://www.gnu.org/licenses/]. + + If you have Internet connection, the latest version of the AGPL should be + available at these locations: + [https://www.fsf.org/licensing/licenses/agpl.html] + [https://www.gnu.org/licenses/agpl.html] + + - Spine (by Doc Reform, related to SiSU) uses standard: + - docReform markup syntax + - standard SiSU markup syntax with modified headers and minor modifications + - docReform object numbering + - standard SiSU object citation numbering & system + + - Homepages: + [https://www.sisudoc.org] + [https://www.doc-reform.org] + + - Git + [https://git.sisudoc.org/] + ++/ +/++ + ocn alignment across the languages of a document<br><br> + + the check that the translations of a document still number their objects + the same way<br><br> + + [sisudoc.ocda.meta.ocn_align] ++/ +module sisudoc.ocda.meta.ocn_align; +@safe: +/+ ↓ the languages of a document share their object numbering. + . + This is the basis of a citation in a multi-language document: an ocn names + the same object in every language, so a reference given in one language is + followable in any of them. Nothing enforces it. It holds because the + translations follow the source object for object, and it stops holding the + moment a translator drops a paragraph, merges two, or turns a heading into + a sentence. The document still builds, every output is produced, and the + numbering has quietly diverges. + . + This says so. Each language leaves a profile as it is abstracted: how + many numbered objects it has, and what kind of object each ocn is. After + the last language of a document, the profiles are compared against the + document's first language, the one the manifest names first. + . + Two kinds of divergence, in the order they are worth hearing: + . + count the languages do not have the same number of numbered + objects. Everything after the first difference is renumbered, + so a single dropped paragraph moves every ocn below it. This + is reported alone, because a kind comparison after it would + report hundreds of consequences of the one cause. + . + type the same ocn is a different type/kind of object. The numbering + still lines up, but the object at that number is a heading in + one language and a paragraph in another, which is the + signature of a heading translated as running text. + . + A warning by default; --strict makes it a failure. + . + It is deliberately not part of the abstraction. Nothing here changes what + a document is, and nothing is written into the .ssp: a language cannot + know whether it aligns with its siblings until they have all been read, + and an artefact that recorded it would be recording something about the + run rather than about the document. ++/ +/+ ↓ mixed in as a named mixin (mixin spineOcnAlign _name;) and its imports + are inside its functions rather than at template scope, both for the + same reason: what a mixin template declares at its own scope is visible + in the scope it is mixed into, and an import there can hijack a name + the host was already resolving by UFCS. ++/ +template spineOcnAlign() { + /+ ↓ what one language of a document leaves behind for the comparison. + kinds is the ocn to object kind map, and it is the whole of what a + language has to say here: the text is expected to differ, the + structure is not. + +/ + struct ST_OcnProfile { + string doc_key; // the document, without its language + string lang; + bool is_reference; // the language the manifest names first + string db_file; // where to note the outcome, "" if no database + int ocn_max; + int ocn_count; + string[int] kinds; + } + /+ ↓ one language's answer, and why +/ + struct ST_OcnAlignResult { + string doc_key; + string lang; + string reference_lang; + bool aligned; + string[] notes; + } + /+ ↓ the kind of an object, for comparison. + a heading carries its level, because a B heading becoming a C heading + is a structural change at the same ocn and is worth the same warning + as a heading becoming a paragraph. + +/ + private string _kindOf(O)(O obj) { + return (obj.metainfo.is_a == "heading") + ? "heading:" ~ obj.metainfo.marked_up_level + : obj.metainfo.is_a; + } + /+ ↓ the profile of one language, taken from its abstraction. + only numbered objects: an object with no ocn is not citable and is not + part of what the languages promise each other. + +/ + ST_OcnProfile ocnProfile(D)(D doc) { + import std.algorithm : canFind, sort; + import std.conv : to; + ST_OcnProfile _p; + _p.doc_key = doc.matters.src.doc_uid_out_no_lang; + _p.lang = doc.matters.src.language; + _p.is_reference = (doc.matters.pod.manifest_list_of_languages.length > 0) + ? (doc.matters.src.language == doc.matters.pod.manifest_list_of_languages[0]) + : true; + /+ ↓ document order, the same section order the .ssp is written in, and + not the abstraction's key order: an associative array does not + promise one, and two runs over the same document would then be + entitled to profile it differently. Any section not on the list is + still swept, by name, so that a new one is not silently skipped. + +/ + string[] _sections = ["head", "toc", "body", "endnotes", + "glossary", "bibliography", "bookindex", "blurb", "tail"]; + string[] _extra; + foreach (_k; doc.abstraction.byKey) { + if (!_sections.canFind(_k)) { _extra ~= _k; } + } + foreach (_k; _extra.sort) { _sections ~= _k; } + foreach (section; _sections) { + if (section !in doc.abstraction) { continue; } + foreach (obj; doc.abstraction[section]) { + int _ocn = obj.metainfo.ocn.to!int; + if (_ocn <= 0) { continue; } + /+ ↓ first in document order wins. An ocn is not unique: a poem and + its first verse share one, by design, so the same number names + two objects and the first of them is the one both languages + will have reached first. + +/ + if (_ocn !in _p.kinds) { _p.kinds[_ocn] = _kindOf(obj); } + _p.ocn_count += 1; + if (_ocn > _p.ocn_max) { _p.ocn_max = _ocn; } + } + } + return _p; + } + /+ ↓ how many differing ocns to name before saying "and N more". + enough to see the shape of the divergence, few enough that a document + whose translation has genuinely gone its own way does not bury the + rest of the run. + +/ + enum size_t OCN_ALIGN_NOTES_MAX = 5; + /+ ↓ the comparison, for every document in the run at once. + the profiles arrive in whatever order the run produced them, which + under --parallel is not the manifest order, so the reference language + is taken from the flag each profile carries rather than from position. + +/ + ST_OcnAlignResult[] ocnAlignCompare(ST_OcnProfile[] _profiles) { + import std.algorithm : sort; + import std.conv : to; + ST_OcnAlignResult[] _out; + ST_OcnProfile[][string] _by_doc; + foreach (_p; _profiles) { _by_doc[_p.doc_key] ~= _p; } + foreach (_doc_key; _by_doc.keys.sort) { + /+ ↓ by language name, not in the order the run produced them: under + --parallel that order is whatever the threads finished in, and a + check people read should say the same thing twice running + +/ + auto _langs = _by_doc[_doc_key].sort!((a, b) => a.lang < b.lang).release; + if (_langs.length < 2) { continue; } // nothing to compare against + ST_OcnProfile _ref; + bool _have_ref = false; + foreach (_p; _langs) { + if (_p.is_reference) { _ref = _p; _have_ref = true; break; } + } + if (!_have_ref) { continue; } + foreach (_p; _langs) { + if (_p.lang == _ref.lang) { continue; } + ST_OcnAlignResult _r; + _r.doc_key = _doc_key; + _r.lang = _p.lang; + _r.reference_lang = _ref.lang; + _r.aligned = true; + if (_p.ocn_count != _ref.ocn_count) { + _r.aligned = false; + long _diff = _p.ocn_count.to!long - _ref.ocn_count.to!long; + _r.notes ~= "numbered objects: " ~ _ref.lang ~ " " + ~ _ref.ocn_count.to!string ~ ", " ~ _p.lang ~ " " + ~ _p.ocn_count.to!string + ~ " (" ~ ((_diff > 0) ? "+" : "") ~ _diff.to!string ~ ")"; + if (_p.ocn_max != _ref.ocn_max) { + _r.notes ~= "highest ocn: " ~ _ref.lang ~ " " + ~ _ref.ocn_max.to!string ~ ", " ~ _p.lang ~ " " + ~ _p.ocn_max.to!string; + } + /+ ↓ no type/kind comparison after a count difference: past the first + dropped or added object every ocn names something else, and + the hundreds of lines that follow are all the one fault + +/ + _out ~= _r; + continue; + } + size_t _kind_diffs = 0; + foreach (_ocn; _ref.kinds.keys.sort) { + if (auto _k = _ocn in _p.kinds) { + if (*_k == _ref.kinds[_ocn]) { continue; } + _kind_diffs += 1; + if (_kind_diffs <= OCN_ALIGN_NOTES_MAX) { + _r.notes ~= "ocn " ~ _ocn.to!string ~ ": " + ~ _ref.kinds[_ocn] ~ " in " ~ _ref.lang ~ ", " + ~ *_k ~ " in " ~ _p.lang; + } + } else { + _kind_diffs += 1; + if (_kind_diffs <= OCN_ALIGN_NOTES_MAX) { + _r.notes ~= "ocn " ~ _ocn.to!string ~ ": " + ~ _ref.kinds[_ocn] ~ " in " ~ _ref.lang ~ ", absent in " + ~ _p.lang; + } + } + } + if (_kind_diffs > OCN_ALIGN_NOTES_MAX) { + _r.notes ~= "and " ~ (_kind_diffs - OCN_ALIGN_NOTES_MAX).to!string + ~ " more"; + } + if (_kind_diffs > 0) { _r.aligned = false; } + _out ~= _r; + } + } + return _out; + } + /+ ↓ the report, as the lines to print. + the document is named once and its languages indented under it, so a + run over a collection reads as a list of documents rather than a list + of languages. + +/ + string[] ocnAlignReportLines(ST_OcnAlignResult[] _results) { + string[] _out; + string _last_doc; + foreach (_r; _results) { + if (_r.aligned) { continue; } + if (_r.doc_key != _last_doc) { + _out ~= "WARNING ocn alignment: " ~ _r.doc_key; + _last_doc = _r.doc_key; + } + _out ~= " [" ~ _r.lang ~ "] differs from [" ~ _r.reference_lang ~ "]"; + foreach (_n; _r.notes) { _out ~= " " ~ _n; } + } + return _out; + } +} |
