aboutsummaryrefslogtreecommitdiffhomepage
path: root/src/sisudoc/outputs
diff options
context:
space:
mode:
authorRalph Amissah <ralph.amissah@gmail.com>2026-09-21 13:33:44 -0400
committerRalph Amissah <ralph.amissah@gmail.com>2026-09-22 16:17:59 -0400
commit4943c93919f349e2d47e6a74f498779fb13f7862 (patch)
treee54696dbe4b8299b76f383641a75f95dd9379a36 /src/sisudoc/outputs
parentocda db: materialise a pod from carried source (diff)
ocda db: carry catalogues & text blobs compressed
The pod carries tools/po4a and the database (until now) did not, so a pod written back out of a database came back a lossy copy, without its translation catalogues. the database now carries them, walked and name-checked exactly as the pod writer walks and checks them, once on the last language. They are compressed to save space, the test sample carrying 6.25 MB of catalogue as read would have made that one document's database two thirds larger. Files gains a compression column: NULL or 'none' for the bytes as read, 'zstd' for a frame. What to compress is decided by role, not by size (source, conf, manifest and tools are text and are compressed; images being compressed already are not). The same kind of file is always stored the same way. Document objects are untouched and stay raw: objects_fts is external content over objects.text and reads that column directly. bytes and sha256 remain those of the original file, so every digest check, digests.txt line and source.digest rebuilt from these rows works unchanged, and a reader that does not decompress can still say what it is looking at. dbReadFiles decompresses, so no caller learns how a blob is stored. It asks pragma_table_info whether the column is there at all: a database written before it reads as raw. A row that claims zstd and will not decompress is named and left out, rather than handed back as a frame where markup should be. live-manual: 9,490,432 bytes without the catalogues, 9,908,224 with them. Output from the materialised pod is identical to output from the original pod, 679 files each side, no differences. (assisted by Claude-Code)
Diffstat (limited to 'src/sisudoc/outputs')
-rw-r--r--src/sisudoc/outputs/io_out/sqlite_ocda_db.d105
1 files changed, 88 insertions, 17 deletions
diff --git a/src/sisudoc/outputs/io_out/sqlite_ocda_db.d b/src/sisudoc/outputs/io_out/sqlite_ocda_db.d
index d30295a..f312d91 100644
--- a/src/sisudoc/outputs/io_out/sqlite_ocda_db.d
+++ b/src/sisudoc/outputs/io_out/sqlite_ocda_db.d
@@ -62,6 +62,7 @@ template spineAbstractionDb() {
import sisudoc.outputs.io_out.paths_output;
import sisudoc.ocda.io_in.paths_source;
import sisudoc.ocda.abstraction.ssp : ssp_format_version;
+ import sisudoc.ocda.zstd : zstdCompress;
/+ ↓ the document is passed in rather than taken from doc, because the
caller hands over one that has been through the .ssp: written as text
and read back. so the database is a function of the .ssp and cannot
@@ -251,20 +252,34 @@ template spineAbstractionDb() {
-- materialised from this database has to put them.
-- conf conf/document_make, the document's own configuration.
-- manifest pod.manifest, as the author wrote it.
+ -- tools tools/po4a, the translation catalogues, the same tree
+ -- the pod carries: a pod written back out of this file is
+ -- then the pod that went in, and not most of it.
--
-- source, conf and manifest are what make this a document source and
-- not only a serialised abstraction: with them a pod can be written
-- back out of the file and rebuilt from the markup, rather than only
-- rendered from the objects.
+ -- bytes and sha256 are always those of the ORIGINAL file, whatever
+ -- compression says, so that a digest check, a digests.txt line and a
+ -- source.digest rebuilt from these rows all keep working, and a reader
+ -- that does not decompress can still say what it is looking at.
+ --
+ -- compression says how data is stored: NULL or 'none' for the bytes as
+ -- read, 'zstd' for a zstd frame. It is decided by role and not by size,
+ -- so the same kind of file always stores the same way and no reader has
+ -- to measure anything to predict it: the text roles are compressed and
+ -- images are not, png and jpeg being compressed already.
CREATE TABLE IF NOT EXISTS files (
- id INTEGER PRIMARY KEY,
- role TEXT NOT NULL,
- name TEXT NOT NULL,
- bytes INTEGER NOT NULL,
- sha256 TEXT,
- width INTEGER,
- height INTEGER,
- data BLOB,
+ id INTEGER PRIMARY KEY,
+ role TEXT NOT NULL,
+ name TEXT NOT NULL,
+ bytes INTEGER NOT NULL,
+ sha256 TEXT,
+ width INTEGER,
+ height INTEGER,
+ compression TEXT,
+ data BLOB,
UNIQUE(role, name)
);
@@ -704,8 +719,9 @@ template spineAbstractionDb() {
import std.digest.sha : sha256Of;
auto file_stmt = db.prepare(
"INSERT OR IGNORE INTO files"
- ~ " (role, name, bytes, sha256, width, height, data)"
- ~ " VALUES (:role, :name, :bytes, :sha256, :width, :height, :data)"
+ ~ " (role, name, bytes, sha256, width, height, compression, data)"
+ ~ " VALUES (:role, :name, :bytes, :sha256, :width, :height,"
+ ~ " :compression, :data)"
);
int[string] _w, _h;
foreach (section; section_order) {
@@ -745,6 +761,7 @@ template spineAbstractionDb() {
import std.typecons : Nullable;
file_stmt.bind(":height", Nullable!int());
}
+ file_stmt.bind(":compression", "none");
file_stmt.bind(":data", _bytes);
file_stmt.execute();
file_stmt.reset();
@@ -776,13 +793,35 @@ template spineAbstractionDb() {
}
return;
}
- file_stmt.bind(":role", _role);
- file_stmt.bind(":name", _name);
- file_stmt.bind(":bytes", cast(long) _bytes.length);
- file_stmt.bind(":sha256", sha256Of(_bytes).toHexString.to!string);
- file_stmt.bind(":width", Nullable!int());
- file_stmt.bind(":height", Nullable!int());
- file_stmt.bind(":data", _bytes);
+ /+ ↓ the text roles are stored compressed, the images are not.
+ By role and not by size: the same kind of file then always
+ stores the same way, and nobody has to measure a document to
+ know how its database will look. The bytes and the digest
+ recorded stay those of the original file either way.
+ .
+ A tiny file can come out a few bytes larger, a zstd frame
+ having some overhead. Left alone: a rule to avoid that would
+ be the size threshold this deliberately does not have.
+ +/
+ ubyte[] _stored = _bytes;
+ string _compression = "none";
+ try {
+ _stored = zstdCompress(_bytes, doc_matters.opt.action.pod_compression);
+ _compression = "zstd";
+ } catch (Exception ex) {
+ stderr.writeln("WARNING: could not compress ", _name, ": ", ex.msg,
+ "; stored as it stands");
+ _stored = _bytes;
+ _compression = "none";
+ }
+ file_stmt.bind(":role", _role);
+ file_stmt.bind(":name", _name);
+ file_stmt.bind(":bytes", cast(long) _bytes.length);
+ file_stmt.bind(":sha256", sha256Of(_bytes).toHexString.to!string);
+ file_stmt.bind(":width", Nullable!int());
+ file_stmt.bind(":height", Nullable!int());
+ file_stmt.bind(":compression", _compression);
+ file_stmt.bind(":data", _stored);
file_stmt.execute();
file_stmt.reset();
}
@@ -808,6 +847,38 @@ template spineAbstractionDb() {
_carry("manifest", "pod.manifest",
doc_matters.pod.manifest_file_with_path);
}
+ /+ ↓ the translation catalogues, the same tools/po4a tree the pod
+ writer carries and by the same rule: a directory walk, because a
+ catalogue set is whatever the translator has, with every name
+ checked before it is taken. Carrying less than the pod does
+ would make a materialised pod a lossy copy of the original,
+ which is the one thing this is for.
+ .
+ Done once, on the last language, as the pod writer does it (the
+ tree is not language specific, and a pass per language would
+ re-read all of it (5.5 MB ten times over, for live-manual) to
+ insert rows the first pass already wrote).
+ +/
+ if (doc_matters.src.is_pod
+ && doc_matters.src.language == doc_matters.pod.manifest_list_of_languages[$-1]
+ ) {
+ import sisudoc.ocda.io_in.carried_names;
+ mixin spineCarriedNames;
+ enum string _po4a_root = "tools/po4a";
+ string _tools_in = doc_matters.pod.manifest_path ~ "/" ~ _po4a_root;
+ if (exists(_tools_in) && _tools_in.isDir) {
+ foreach (string _f; dirEntries(_tools_in, SpanMode.depth)) {
+ if (!(_f.isFile)) { continue; }
+ string _rel = _po4a_root ~ "/" ~ _f[(_tools_in.length + 1) .. $];
+ string _bad = validateCarriedPath(_rel);
+ if (_bad.length > 0) {
+ writeln("WARNING not carried into the database: ", _bad);
+ continue;
+ }
+ _carry("tools", _rel, _f);
+ }
+ }
+ }
}
file_stmt.finalize();
}