diff options
| -rw-r--r-- | org/document_source_conversions.org | 7 | ||||
| -rw-r--r-- | org/in_abstraction_artefacts.org | 43 | ||||
| -rw-r--r-- | org/out_ocda_sqlite_db.org | 105 | ||||
| -rw-r--r-- | src/sisudoc/ocda/abstraction/db_in.d | 43 | ||||
| -rw-r--r-- | src/sisudoc/ocda/abstraction/pod_from_db.d | 6 | ||||
| -rw-r--r-- | src/sisudoc/outputs/io_out/sqlite_ocda_db.d | 105 |
6 files changed, 264 insertions, 45 deletions
diff --git a/org/document_source_conversions.org b/org/document_source_conversions.org index cde085e..386d1ec 100644 --- a/org/document_source_conversions.org +++ b/org/document_source_conversions.org @@ -52,6 +52,7 @@ module sisudoc.ocda.abstraction.pod_from_db; <dest>/<pod>/conf/document_make role='conf' <dest>/<pod>/media/text/<lang>/<file> role='source' <dest>/<pod>/media/image/<file> role='image' + <dest>/<pod>/tools/po4a/... role='tools' . The pod's name comes from the database's own filename: <doc>.ocda.db is named by doc_uid_out_no_lang, which is the pod name and the document's @@ -114,7 +115,8 @@ template spinePodFromDb() { case "image": return "media/image/" ~ _name; case "source": case "conf": - case "manifest": return _name; + case "manifest": + case "tools": return _name; default: return ""; } } @@ -131,7 +133,7 @@ template spinePodFromDb() { return _out; } _dbr.ST_ArtefactFile[] _files; - foreach (_role; ["manifest", "conf", "source", "image"]) { + foreach (_role; ["manifest", "conf", "source", "image", "tools"]) { _files ~= _dbr.dbReadFiles(_db_file, _role); } if (_files.length == 0) { @@ -235,4 +237,3 @@ template spinePodFromDb() { #+END_SRC * __END__ - diff --git a/org/in_abstraction_artefacts.org b/org/in_abstraction_artefacts.org index f0ee34c..71d8d65 100644 --- a/org/in_abstraction_artefacts.org +++ b/org/in_abstraction_artefacts.org @@ -914,7 +914,13 @@ template spineAbstractionDbRead() { a .ocda.db holds its images as blobs, which is what lets it travel on its own where a .ssp cannot: a .ssp describes its images by name, size, pixels and sha256, but does not carry them. The role column - says what a file is; today only 'image' is written. + says what a file is: 'image', and the markup, configuration, manifest + and tools carried with it. + . + data comes back as the original file, decompressed where the row says + it was compressed, so that no caller has to know how a blob is stored. + bytes and sha256 are the original's either way, and are what a caller + checks the data it is given against. +/ struct ST_ArtefactFile { string role; @@ -924,21 +930,52 @@ template spineAbstractionDbRead() { ubyte[] data; } @trusted ST_ArtefactFile[] dbReadFiles(string db_file, string role = "image") { + import sisudoc.ocda.zstd : zstdDecompress; ST_ArtefactFile[] _out; if (!db_file.exists) { return _out; } auto db = Database(db_file, SQLITE_OPEN_READONLY); + /+ ↓ whether this file has a compression column at all. A database + written before the column has none, and selecting it would throw + rather than answer; everything in such a file is stored as read. + +/ + bool _has_compression = false; foreach (row; db.execute( - "SELECT role, name, bytes, sha256, data FROM files WHERE role = '" ~ role - ~ "' ORDER BY name") + "SELECT count(*) AS n FROM pragma_table_info('files')" + ~ " WHERE name = 'compression'") ) { + _has_compression = (row["n"].as!long > 0); + } + auto _stmt = db.prepare( + "SELECT role, name, bytes, sha256, " + ~ (_has_compression ? "compression" : "NULL AS compression") + ~ ", data FROM files WHERE role = :role ORDER BY name" + ); + _stmt.bind(":role", role); + foreach (row; _stmt.execute()) { ST_ArtefactFile _f; _f.role = row["role"].as!string; _f.name = row["name"].as!string; _f.bytes = row["bytes"].as!ulong; _f.sha256 = row["sha256"].as!string; _f.data = row["data"].as!(ubyte[]); + string _compression = row["compression"].as!string; + if (_compression == "zstd") { + try { + _f.data = zstdDecompress(_f.data); + } catch (Exception ex) { + /+ ↓ a row that says it is compressed and will not decompress is + not usable, and handing back the frame as if it were the file + would put a zstd frame where markup should be. Named, and + left out. + +/ + stderr.writeln("WARNING: ", db_file.baseName, " carries ", _f.name, + " as zstd and it will not decompress: ", ex.msg, "; left out"); + continue; + } + } _out ~= _f; } + _stmt.finalize(); return _out; } } diff --git a/org/out_ocda_sqlite_db.org b/org/out_ocda_sqlite_db.org index 08a4305..e8c8490 100644 --- a/org/out_ocda_sqlite_db.org +++ b/org/out_ocda_sqlite_db.org @@ -44,6 +44,7 @@ template spineAbstractionDb() { import sisudoc.outputs.io_out.paths_output; import sisudoc.ocda.io_in.paths_source; import sisudoc.ocda.abstraction.ssp : ssp_format_version; + import sisudoc.ocda.zstd : zstdCompress; /+ ↓ the document is passed in rather than taken from doc, because the caller hands over one that has been through the .ssp: written as text and read back. so the database is a function of the .ssp and cannot @@ -233,20 +234,34 @@ template spineAbstractionDb() { -- materialised from this database has to put them. -- conf conf/document_make, the document's own configuration. -- manifest pod.manifest, as the author wrote it. + -- tools tools/po4a, the translation catalogues, the same tree + -- the pod carries: a pod written back out of this file is + -- then the pod that went in, and not most of it. -- -- source, conf and manifest are what make this a document source and -- not only a serialised abstraction: with them a pod can be written -- back out of the file and rebuilt from the markup, rather than only -- rendered from the objects. + -- bytes and sha256 are always those of the ORIGINAL file, whatever + -- compression says, so that a digest check, a digests.txt line and a + -- source.digest rebuilt from these rows all keep working, and a reader + -- that does not decompress can still say what it is looking at. + -- + -- compression says how data is stored: NULL or 'none' for the bytes as + -- read, 'zstd' for a zstd frame. It is decided by role and not by size, + -- so the same kind of file always stores the same way and no reader has + -- to measure anything to predict it: the text roles are compressed and + -- images are not, png and jpeg being compressed already. CREATE TABLE IF NOT EXISTS files ( - id INTEGER PRIMARY KEY, - role TEXT NOT NULL, - name TEXT NOT NULL, - bytes INTEGER NOT NULL, - sha256 TEXT, - width INTEGER, - height INTEGER, - data BLOB, + id INTEGER PRIMARY KEY, + role TEXT NOT NULL, + name TEXT NOT NULL, + bytes INTEGER NOT NULL, + sha256 TEXT, + width INTEGER, + height INTEGER, + compression TEXT, + data BLOB, UNIQUE(role, name) ); @@ -686,8 +701,9 @@ template spineAbstractionDb() { import std.digest.sha : sha256Of; auto file_stmt = db.prepare( "INSERT OR IGNORE INTO files" - ~ " (role, name, bytes, sha256, width, height, data)" - ~ " VALUES (:role, :name, :bytes, :sha256, :width, :height, :data)" + ~ " (role, name, bytes, sha256, width, height, compression, data)" + ~ " VALUES (:role, :name, :bytes, :sha256, :width, :height," + ~ " :compression, :data)" ); int[string] _w, _h; foreach (section; section_order) { @@ -727,6 +743,7 @@ template spineAbstractionDb() { import std.typecons : Nullable; file_stmt.bind(":height", Nullable!int()); } + file_stmt.bind(":compression", "none"); file_stmt.bind(":data", _bytes); file_stmt.execute(); file_stmt.reset(); @@ -758,13 +775,35 @@ template spineAbstractionDb() { } return; } - file_stmt.bind(":role", _role); - file_stmt.bind(":name", _name); - file_stmt.bind(":bytes", cast(long) _bytes.length); - file_stmt.bind(":sha256", sha256Of(_bytes).toHexString.to!string); - file_stmt.bind(":width", Nullable!int()); - file_stmt.bind(":height", Nullable!int()); - file_stmt.bind(":data", _bytes); + /+ ↓ the text roles are stored compressed, the images are not. + By role and not by size: the same kind of file then always + stores the same way, and nobody has to measure a document to + know how its database will look. The bytes and the digest + recorded stay those of the original file either way. + . + A tiny file can come out a few bytes larger, a zstd frame + having some overhead. Left alone: a rule to avoid that would + be the size threshold this deliberately does not have. + +/ + ubyte[] _stored = _bytes; + string _compression = "none"; + try { + _stored = zstdCompress(_bytes, doc_matters.opt.action.pod_compression); + _compression = "zstd"; + } catch (Exception ex) { + stderr.writeln("WARNING: could not compress ", _name, ": ", ex.msg, + "; stored as it stands"); + _stored = _bytes; + _compression = "none"; + } + file_stmt.bind(":role", _role); + file_stmt.bind(":name", _name); + file_stmt.bind(":bytes", cast(long) _bytes.length); + file_stmt.bind(":sha256", sha256Of(_bytes).toHexString.to!string); + file_stmt.bind(":width", Nullable!int()); + file_stmt.bind(":height", Nullable!int()); + file_stmt.bind(":compression", _compression); + file_stmt.bind(":data", _stored); file_stmt.execute(); file_stmt.reset(); } @@ -790,6 +829,38 @@ template spineAbstractionDb() { _carry("manifest", "pod.manifest", doc_matters.pod.manifest_file_with_path); } + /+ ↓ the translation catalogues, the same tools/po4a tree the pod + writer carries and by the same rule: a directory walk, because a + catalogue set is whatever the translator has, with every name + checked before it is taken. Carrying less than the pod does + would make a materialised pod a lossy copy of the original, + which is the one thing this is for. + . + Done once, on the last language, as the pod writer does it (the + tree is not language specific, and a pass per language would + re-read all of it (5.5 MB ten times over, for live-manual) to + insert rows the first pass already wrote). + +/ + if (doc_matters.src.is_pod + && doc_matters.src.language == doc_matters.pod.manifest_list_of_languages[$-1] + ) { + import sisudoc.ocda.io_in.carried_names; + mixin spineCarriedNames; + enum string _po4a_root = "tools/po4a"; + string _tools_in = doc_matters.pod.manifest_path ~ "/" ~ _po4a_root; + if (exists(_tools_in) && _tools_in.isDir) { + foreach (string _f; dirEntries(_tools_in, SpanMode.depth)) { + if (!(_f.isFile)) { continue; } + string _rel = _po4a_root ~ "/" ~ _f[(_tools_in.length + 1) .. $]; + string _bad = validateCarriedPath(_rel); + if (_bad.length > 0) { + writeln("WARNING not carried into the database: ", _bad); + continue; + } + _carry("tools", _rel, _f); + } + } + } } file_stmt.finalize(); } diff --git a/src/sisudoc/ocda/abstraction/db_in.d b/src/sisudoc/ocda/abstraction/db_in.d index 53ac95d..0a01364 100644 --- a/src/sisudoc/ocda/abstraction/db_in.d +++ b/src/sisudoc/ocda/abstraction/db_in.d @@ -380,7 +380,13 @@ template spineAbstractionDbRead() { a .ocda.db holds its images as blobs, which is what lets it travel on its own where a .ssp cannot: a .ssp describes its images by name, size, pixels and sha256, but does not carry them. The role column - says what a file is; today only 'image' is written. + says what a file is: 'image', and the markup, configuration, manifest + and tools carried with it. + . + data comes back as the original file, decompressed where the row says + it was compressed, so that no caller has to know how a blob is stored. + bytes and sha256 are the original's either way, and are what a caller + checks the data it is given against. +/ struct ST_ArtefactFile { string role; @@ -390,21 +396,52 @@ template spineAbstractionDbRead() { ubyte[] data; } @trusted ST_ArtefactFile[] dbReadFiles(string db_file, string role = "image") { + import sisudoc.ocda.zstd : zstdDecompress; ST_ArtefactFile[] _out; if (!db_file.exists) { return _out; } auto db = Database(db_file, SQLITE_OPEN_READONLY); + /+ ↓ whether this file has a compression column at all. A database + written before the column has none, and selecting it would throw + rather than answer; everything in such a file is stored as read. + +/ + bool _has_compression = false; foreach (row; db.execute( - "SELECT role, name, bytes, sha256, data FROM files WHERE role = '" ~ role - ~ "' ORDER BY name") + "SELECT count(*) AS n FROM pragma_table_info('files')" + ~ " WHERE name = 'compression'") ) { + _has_compression = (row["n"].as!long > 0); + } + auto _stmt = db.prepare( + "SELECT role, name, bytes, sha256, " + ~ (_has_compression ? "compression" : "NULL AS compression") + ~ ", data FROM files WHERE role = :role ORDER BY name" + ); + _stmt.bind(":role", role); + foreach (row; _stmt.execute()) { ST_ArtefactFile _f; _f.role = row["role"].as!string; _f.name = row["name"].as!string; _f.bytes = row["bytes"].as!ulong; _f.sha256 = row["sha256"].as!string; _f.data = row["data"].as!(ubyte[]); + string _compression = row["compression"].as!string; + if (_compression == "zstd") { + try { + _f.data = zstdDecompress(_f.data); + } catch (Exception ex) { + /+ ↓ a row that says it is compressed and will not decompress is + not usable, and handing back the frame as if it were the file + would put a zstd frame where markup should be. Named, and + left out. + +/ + stderr.writeln("WARNING: ", db_file.baseName, " carries ", _f.name, + " as zstd and it will not decompress: ", ex.msg, "; left out"); + continue; + } + } _out ~= _f; } + _stmt.finalize(); return _out; } } diff --git a/src/sisudoc/ocda/abstraction/pod_from_db.d b/src/sisudoc/ocda/abstraction/pod_from_db.d index 0bdb3bc..2e231d3 100644 --- a/src/sisudoc/ocda/abstraction/pod_from_db.d +++ b/src/sisudoc/ocda/abstraction/pod_from_db.d @@ -74,6 +74,7 @@ module sisudoc.ocda.abstraction.pod_from_db; <dest>/<pod>/conf/document_make role='conf' <dest>/<pod>/media/text/<lang>/<file> role='source' <dest>/<pod>/media/image/<file> role='image' + <dest>/<pod>/tools/po4a/... role='tools' . The pod's name comes from the database's own filename: <doc>.ocda.db is named by doc_uid_out_no_lang, which is the pod name and the document's @@ -136,7 +137,8 @@ template spinePodFromDb() { case "image": return "media/image/" ~ _name; case "source": case "conf": - case "manifest": return _name; + case "manifest": + case "tools": return _name; default: return ""; } } @@ -153,7 +155,7 @@ template spinePodFromDb() { return _out; } _dbr.ST_ArtefactFile[] _files; - foreach (_role; ["manifest", "conf", "source", "image"]) { + foreach (_role; ["manifest", "conf", "source", "image", "tools"]) { _files ~= _dbr.dbReadFiles(_db_file, _role); } if (_files.length == 0) { diff --git a/src/sisudoc/outputs/io_out/sqlite_ocda_db.d b/src/sisudoc/outputs/io_out/sqlite_ocda_db.d index d30295a..f312d91 100644 --- a/src/sisudoc/outputs/io_out/sqlite_ocda_db.d +++ b/src/sisudoc/outputs/io_out/sqlite_ocda_db.d @@ -62,6 +62,7 @@ template spineAbstractionDb() { import sisudoc.outputs.io_out.paths_output; import sisudoc.ocda.io_in.paths_source; import sisudoc.ocda.abstraction.ssp : ssp_format_version; + import sisudoc.ocda.zstd : zstdCompress; /+ ↓ the document is passed in rather than taken from doc, because the caller hands over one that has been through the .ssp: written as text and read back. so the database is a function of the .ssp and cannot @@ -251,20 +252,34 @@ template spineAbstractionDb() { -- materialised from this database has to put them. -- conf conf/document_make, the document's own configuration. -- manifest pod.manifest, as the author wrote it. + -- tools tools/po4a, the translation catalogues, the same tree + -- the pod carries: a pod written back out of this file is + -- then the pod that went in, and not most of it. -- -- source, conf and manifest are what make this a document source and -- not only a serialised abstraction: with them a pod can be written -- back out of the file and rebuilt from the markup, rather than only -- rendered from the objects. + -- bytes and sha256 are always those of the ORIGINAL file, whatever + -- compression says, so that a digest check, a digests.txt line and a + -- source.digest rebuilt from these rows all keep working, and a reader + -- that does not decompress can still say what it is looking at. + -- + -- compression says how data is stored: NULL or 'none' for the bytes as + -- read, 'zstd' for a zstd frame. It is decided by role and not by size, + -- so the same kind of file always stores the same way and no reader has + -- to measure anything to predict it: the text roles are compressed and + -- images are not, png and jpeg being compressed already. CREATE TABLE IF NOT EXISTS files ( - id INTEGER PRIMARY KEY, - role TEXT NOT NULL, - name TEXT NOT NULL, - bytes INTEGER NOT NULL, - sha256 TEXT, - width INTEGER, - height INTEGER, - data BLOB, + id INTEGER PRIMARY KEY, + role TEXT NOT NULL, + name TEXT NOT NULL, + bytes INTEGER NOT NULL, + sha256 TEXT, + width INTEGER, + height INTEGER, + compression TEXT, + data BLOB, UNIQUE(role, name) ); @@ -704,8 +719,9 @@ template spineAbstractionDb() { import std.digest.sha : sha256Of; auto file_stmt = db.prepare( "INSERT OR IGNORE INTO files" - ~ " (role, name, bytes, sha256, width, height, data)" - ~ " VALUES (:role, :name, :bytes, :sha256, :width, :height, :data)" + ~ " (role, name, bytes, sha256, width, height, compression, data)" + ~ " VALUES (:role, :name, :bytes, :sha256, :width, :height," + ~ " :compression, :data)" ); int[string] _w, _h; foreach (section; section_order) { @@ -745,6 +761,7 @@ template spineAbstractionDb() { import std.typecons : Nullable; file_stmt.bind(":height", Nullable!int()); } + file_stmt.bind(":compression", "none"); file_stmt.bind(":data", _bytes); file_stmt.execute(); file_stmt.reset(); @@ -776,13 +793,35 @@ template spineAbstractionDb() { } return; } - file_stmt.bind(":role", _role); - file_stmt.bind(":name", _name); - file_stmt.bind(":bytes", cast(long) _bytes.length); - file_stmt.bind(":sha256", sha256Of(_bytes).toHexString.to!string); - file_stmt.bind(":width", Nullable!int()); - file_stmt.bind(":height", Nullable!int()); - file_stmt.bind(":data", _bytes); + /+ ↓ the text roles are stored compressed, the images are not. + By role and not by size: the same kind of file then always + stores the same way, and nobody has to measure a document to + know how its database will look. The bytes and the digest + recorded stay those of the original file either way. + . + A tiny file can come out a few bytes larger, a zstd frame + having some overhead. Left alone: a rule to avoid that would + be the size threshold this deliberately does not have. + +/ + ubyte[] _stored = _bytes; + string _compression = "none"; + try { + _stored = zstdCompress(_bytes, doc_matters.opt.action.pod_compression); + _compression = "zstd"; + } catch (Exception ex) { + stderr.writeln("WARNING: could not compress ", _name, ": ", ex.msg, + "; stored as it stands"); + _stored = _bytes; + _compression = "none"; + } + file_stmt.bind(":role", _role); + file_stmt.bind(":name", _name); + file_stmt.bind(":bytes", cast(long) _bytes.length); + file_stmt.bind(":sha256", sha256Of(_bytes).toHexString.to!string); + file_stmt.bind(":width", Nullable!int()); + file_stmt.bind(":height", Nullable!int()); + file_stmt.bind(":compression", _compression); + file_stmt.bind(":data", _stored); file_stmt.execute(); file_stmt.reset(); } @@ -808,6 +847,38 @@ template spineAbstractionDb() { _carry("manifest", "pod.manifest", doc_matters.pod.manifest_file_with_path); } + /+ ↓ the translation catalogues, the same tools/po4a tree the pod + writer carries and by the same rule: a directory walk, because a + catalogue set is whatever the translator has, with every name + checked before it is taken. Carrying less than the pod does + would make a materialised pod a lossy copy of the original, + which is the one thing this is for. + . + Done once, on the last language, as the pod writer does it (the + tree is not language specific, and a pass per language would + re-read all of it (5.5 MB ten times over, for live-manual) to + insert rows the first pass already wrote). + +/ + if (doc_matters.src.is_pod + && doc_matters.src.language == doc_matters.pod.manifest_list_of_languages[$-1] + ) { + import sisudoc.ocda.io_in.carried_names; + mixin spineCarriedNames; + enum string _po4a_root = "tools/po4a"; + string _tools_in = doc_matters.pod.manifest_path ~ "/" ~ _po4a_root; + if (exists(_tools_in) && _tools_in.isDir) { + foreach (string _f; dirEntries(_tools_in, SpanMode.depth)) { + if (!(_f.isFile)) { continue; } + string _rel = _po4a_root ~ "/" ~ _f[(_tools_in.length + 1) .. $]; + string _bad = validateCarriedPath(_rel); + if (_bad.length > 0) { + writeln("WARNING not carried into the database: ", _bad); + continue; + } + _carry("tools", _rel, _f); + } + } + } } file_stmt.finalize(); } |
