diff options
| author | Ralph Amissah <ralph.amissah@gmail.com> | 2026-09-11 14:13:25 -0400 |
|---|---|---|
| committer | Ralph Amissah <ralph.amissah@gmail.com> | 2026-09-12 12:14:39 -0400 |
| commit | 5ba5620bf9b3c0f42a8ac6aeccf7c93ee6329fc4 (patch) | |
| tree | 1d7f7f8c9d2144f4107d888d7e1f2d3515c40a50 /src/sisudoc/ocda | |
| parent | spine: process a document from .ocda.db or .ssp (diff) | |
ocda: the images an artefact carries, and the ones it only describes
The document markup sample collection builds byte identically from
either artefact: 1457 files, no differences, from markup, from the 35
.ssp files, and from the 35 .ocda.db files.
A .ocda.db carries its images as blobs. dbReadFiles() reads them and
they are written where the output writers look for images: a
directory of this run's making, removed when the run ends, not the
pod beside the artefact, (which could be a working published tree). The
document's source path moves with it, image_dir_path being reached from
the document's own file rather than from the pod.
Each blob is checked against the sha256 the database recorded beside it.
The two were written together, so a mismatch means the file is damaged.
(the point of having recorded the digest).
A .ssp only describes its images, so those are the ones in the pod it
sits in, and they are checked against the digests it recorded.
That asymmetry arises from of what the two artefacts are: the ocda.db is
self-sufficient and can prove it is looking at the right image, the .ssp
describes a pod it sits alongside.
Found while wiring it: setting the pod directory alone was not enough,
since image_dir_path is derived from the source file's path
(../../image from media/text/<lang>/), so the source path has to be
moved with it.
(assisted by Claude-Code)
Diffstat (limited to 'src/sisudoc/ocda')
| -rw-r--r-- | src/sisudoc/ocda/abstraction/db_in.d | 32 | ||||
| -rw-r--r-- | src/sisudoc/ocda/abstraction/doc_from_artefact.d | 126 |
2 files changed, 149 insertions, 9 deletions
diff --git a/src/sisudoc/ocda/abstraction/db_in.d b/src/sisudoc/ocda/abstraction/db_in.d index 7f920bd..6e90820 100644 --- a/src/sisudoc/ocda/abstraction/db_in.d +++ b/src/sisudoc/ocda/abstraction/db_in.d @@ -305,4 +305,36 @@ template spineAbstractionDbRead() { } return doc; } + /+ ↓ the files a database carries. + + a .ocda.db holds its images as blobs, which is what lets it travel on + its own where a .ssp cannot: a .ssp describes its images by name, + size, pixels and sha256, but does not carry them. The role column + says what a file is; today only 'image' is written. + +/ + struct ST_ArtefactFile { + string role; + string name; + ulong bytes; + string sha256; + ubyte[] data; + } + @trusted ST_ArtefactFile[] dbReadFiles(string db_file, string role = "image") { + ST_ArtefactFile[] _out; + if (!db_file.exists) { return _out; } + auto db = Database(db_file, SQLITE_OPEN_READONLY); + foreach (row; db.execute( + "SELECT role, name, bytes, sha256, data FROM files WHERE role = '" ~ role + ~ "' ORDER BY name") + ) { + ST_ArtefactFile _f; + _f.role = row["role"].as!string; + _f.name = row["name"].as!string; + _f.bytes = row["bytes"].as!ulong; + _f.sha256 = row["sha256"].as!string; + _f.data = row["data"].as!(ubyte[]); + _out ~= _f; + } + return _out; + } } diff --git a/src/sisudoc/ocda/abstraction/doc_from_artefact.d b/src/sisudoc/ocda/abstraction/doc_from_artefact.d index e45cbf0..3a32d25 100644 --- a/src/sisudoc/ocda/abstraction/doc_from_artefact.d +++ b/src/sisudoc/ocda/abstraction/doc_from_artefact.d @@ -50,26 +50,25 @@ module sisudoc.ocda.abstraction.doc_from_artefact; @safe: /+ ↓ a document, from an artefact rather than from markup - spineAbstraction() parses markup and hands the output writers a doc: an abstraction and a doc_matters. This does the same from a .ssp or a .ocda.db, filling the same docMattersMake() arguments, so that what comes out is the value the writers already take and they never learn which of the two made it. - + . Three of the seven arguments come from the run and are the same either way: program_info, opt_action, and the conf half of conf_make_meta, which is site and run scoped (urls, papersize, the search database) and has no business in a document's own artefact. The other four come from the artefact: - + . conf_make_meta meta and make from the @meta and @make blocks, over the site config the run assembled doc_has docHasFromAbstraction, over the objects doc_digest the markup digest from @source manifest synthesised from the artefact's path and the document's own name and language - + . What is not here, deliberately: insert_file_list, which is read only by source_pod.d for --source and --pod2. No artefact carries the markup, so a pod cannot be rebuilt from one, and the empty list is the truthful @@ -80,6 +79,7 @@ template spineDocFromArtefact() { import std.array; import std.conv : to; import std.path; + import std.process : thisProcessID; import std.regex; import std.stdio; import std.string; @@ -106,7 +106,6 @@ template spineDocFromArtefact() { return _member; } /+ ↓ the document's own header, over the site config the run assembled. - every string field of MetaComposite is looked for under its own key, so a field added there and emitted by the writer arrives here without anything being added. The arr fields are split from the string they @@ -168,14 +167,13 @@ template spineDocFromArtefact() { return _out; } /+ ↓ where a document loaded from an artefact says it is. - an artefact names its document (% Source:) and its language (@source language) but not where the pod holding its images sits, so that is taken from where the artefact itself was found: - + . pod/<doc>/media/abstraction/<uid>.ssp -> pod/<doc>/ pod/<uid>.ocda.db -> pod/<doc>/ - + . which is where --pod2 puts them both. The images beside a .ssp are then found where html.d already looks; a .ocda.db carries its own and does not need the directory to exist. @@ -206,8 +204,90 @@ template spineDocFromArtefact() { .chainPath(_ssp.source).array).to!string; return _out; } + /+ ↓ a database's images, written where the writers look for them. + returns the pod directory to use in place of the one beside the + artefact, or "" when the database carries no images and the pod + beside it should stand. Each blob is checked against the sha256 the + database recorded with it: the two were written together, so a + mismatch means the file is damaged, and saying so is the point of + having recorded the digest at all. + +/ + @trusted private string _imagesExtract(O)(string _artefact, O _opt_action) { + import std.digest : toHexString; + import std.digest.sha : sha256Of; + import std.file : mkdirRecurse, write, tempDir; + import sisudoc.ocda.abstraction.db_in : spineAbstractionDbRead; + mixin spineAbstractionDbRead _dbr; + auto _files = _dbr.dbReadFiles(_artefact, "image"); + if (_files.length == 0) { return ""; } + string _root = (tempDir.chainPath("spine-ocda-" + ~ _artefact.baseName ~ "-" ~ thisProcessID.to!string).array).to!string; + string _img_dir = (_root.chainPath("media").chainPath("image").array).to!string; + try { + _img_dir.mkdirRecurse; + } catch (Exception ex) { + stderr.writeln("WARNING: could not make an image directory for ", _artefact, + ": ", ex.msg); + return ""; + } + foreach (_f; _files) { + string _got = _f.data.sha256Of.toHexString.to!string; + if (_f.sha256.length > 0 && _got != _f.sha256) { + stderr.writeln("WARNING: image ", _f.name, " in ", _artefact.baseName, + " does not match the digest recorded with it (", _f.sha256, " expected, ", + _got, " found); it is written out as it stands"); + } + try { + (_img_dir.chainPath(_f.name).array).to!string.write(_f.data); + } catch (Exception ex) { + stderr.writeln("WARNING: could not write ", _f.name, ": ", ex.msg); + } + } + if (_opt_action.vox_gt_1) { + writeln(" ", _files.length, " image(s) from the database, in ", _img_dir); + } + return _root; + } + /+ ↓ the images a .ssp describes, against the ones beside it. + a .ssp carries no image bytes, so what it can do is say whether the + files in the pod it sits in are the ones the abstraction was built + from. Nothing checked that before: a changed or wrong image in the + directory paired silently. + +/ + @trusted private void _imagesVerify(S,O)(string _pod_dir, S _ssp, O _opt_action) { + import std.digest : toHexString; + import std.digest.sha : sha256Of; + import std.file : exists, read; + string _img_dir = (_pod_dir.chainPath("media").chainPath("image").array).to!string; + int _checked, _missing, _wrong; + foreach (_section; _ssp.section_order) { + foreach (obj; _ssp.abstraction[_section]) { + foreach (_img; obj.metainfo.sha256.images) { + if (_img.fileName.length == 0) { continue; } + string _pth = (_img_dir.chainPath(_img.fileName).array).to!string; + if (!_pth.exists) { ++_missing; continue; } + ++_checked; + string _want = _img.fileHash_sha256.toHexString.to!string; + string _got = (cast(ubyte[]) _pth.read).sha256Of.toHexString.to!string; + if (_want != _got && _want != "00000000000000000000000000000000" + ~ "00000000000000000000000000000000") { + ++_wrong; + stderr.writeln("WARNING: image ", _img.fileName, " beside ", + _ssp.source, " is not the one the abstraction was built from"); + } + } + } + } + if (_missing > 0) { + stderr.writeln("WARNING: ", _missing, " image(s) the abstraction names are", + " not in ", _img_dir); + } + if (_opt_action.vox_gt_1 && _checked > 0) { + writeln(" ", _checked, " image(s) checked against the abstraction's digests", + (_wrong == 0) ? ", all matching" : ""); + } + } /+ ↓ the doc the output writers take, from an artefact. - the reading is done here rather than by the caller, and that is not only tidiness: main() is @system, so a template mixed in there infers @system for everything it declares, and the output writers are @safe @@ -226,6 +306,33 @@ template spineDocFromArtefact() { auto _loaded = abstractionLoad(_artefact); auto _ssp = _loaded.doc; auto _pths = artefactPaths(_artefact, _ssp); + /+ ↓ the images. + a .ocda.db carries its own, which is what lets it travel on its + own, so they are taken from it and written where the output writers + look for images: a directory of this run's making, not the pod + beside the artefact, which may be somebody's published tree. A .ssp + only describes its images, so those are the ones in the pod it sits + in, and are checked against the digests it recorded. + +/ + string _images_tmp; + if (_loaded.loaded) { + if (_artefact.endsWith(".ssp")) { + _imagesVerify(_pths.pod_dir, _ssp, _opt_action); + } else { + string _extracted = _imagesExtract(_artefact, _opt_action); + if (_extracted.length > 0) { + /+ ↓ the source path has to move with it: image_dir_path is + reached from the document's own file, not from the pod +/ + _images_tmp = _extracted; + _pths.pod_dir = _extracted; + _pths.src_file_with_path = (_extracted + .chainPath("media").chainPath("text") + .chainPath(("language" in _ssp.source_info) + ? _ssp.source_info["language"] : "en") + .chainPath(_ssp.source).array).to!string; + } + } + } auto _manifest = PathMatters!()(_opt_action, _env, _pths.pod_dir, _pths.src_file_with_path); auto _cmm = confMakeMetaFromHeader(_ssp, _site_conf); auto _has = docHasFromAbstraction(_ssp.abstraction, _ssp.doc_has, _opt_action); @@ -250,6 +357,7 @@ template spineDocFromArtefact() { const auto abstraction() { return _abstraction; } auto matters() { return _matters; } bool loaded() { return _loaded.loaded; } + string images_tmp() { return _images_tmp; } string note() { return _loaded.note; } string source_name() { return abstractionSourceName(_loaded.source); } } |
