aboutsummaryrefslogtreecommitdiffhomepage
diff options
context:
space:
mode:
-rw-r--r--org/document_source_conversions.org7
-rw-r--r--org/in_abstraction_artefacts.org43
-rw-r--r--org/out_ocda_sqlite_db.org105
-rw-r--r--src/sisudoc/ocda/abstraction/db_in.d43
-rw-r--r--src/sisudoc/ocda/abstraction/pod_from_db.d6
-rw-r--r--src/sisudoc/outputs/io_out/sqlite_ocda_db.d105
6 files changed, 264 insertions, 45 deletions
diff --git a/org/document_source_conversions.org b/org/document_source_conversions.org
index cde085e..386d1ec 100644
--- a/org/document_source_conversions.org
+++ b/org/document_source_conversions.org
@@ -52,6 +52,7 @@ module sisudoc.ocda.abstraction.pod_from_db;
<dest>/<pod>/conf/document_make role='conf'
<dest>/<pod>/media/text/<lang>/<file> role='source'
<dest>/<pod>/media/image/<file> role='image'
+ <dest>/<pod>/tools/po4a/... role='tools'
.
The pod's name comes from the database's own filename: <doc>.ocda.db is
named by doc_uid_out_no_lang, which is the pod name and the document's
@@ -114,7 +115,8 @@ template spinePodFromDb() {
case "image": return "media/image/" ~ _name;
case "source":
case "conf":
- case "manifest": return _name;
+ case "manifest":
+ case "tools": return _name;
default: return "";
}
}
@@ -131,7 +133,7 @@ template spinePodFromDb() {
return _out;
}
_dbr.ST_ArtefactFile[] _files;
- foreach (_role; ["manifest", "conf", "source", "image"]) {
+ foreach (_role; ["manifest", "conf", "source", "image", "tools"]) {
_files ~= _dbr.dbReadFiles(_db_file, _role);
}
if (_files.length == 0) {
@@ -235,4 +237,3 @@ template spinePodFromDb() {
#+END_SRC
* __END__
-
diff --git a/org/in_abstraction_artefacts.org b/org/in_abstraction_artefacts.org
index f0ee34c..71d8d65 100644
--- a/org/in_abstraction_artefacts.org
+++ b/org/in_abstraction_artefacts.org
@@ -914,7 +914,13 @@ template spineAbstractionDbRead() {
a .ocda.db holds its images as blobs, which is what lets it travel on
its own where a .ssp cannot: a .ssp describes its images by name,
size, pixels and sha256, but does not carry them. The role column
- says what a file is; today only 'image' is written.
+ says what a file is: 'image', and the markup, configuration, manifest
+ and tools carried with it.
+ .
+ data comes back as the original file, decompressed where the row says
+ it was compressed, so that no caller has to know how a blob is stored.
+ bytes and sha256 are the original's either way, and are what a caller
+ checks the data it is given against.
+/
struct ST_ArtefactFile {
string role;
@@ -924,21 +930,52 @@ template spineAbstractionDbRead() {
ubyte[] data;
}
@trusted ST_ArtefactFile[] dbReadFiles(string db_file, string role = "image") {
+ import sisudoc.ocda.zstd : zstdDecompress;
ST_ArtefactFile[] _out;
if (!db_file.exists) { return _out; }
auto db = Database(db_file, SQLITE_OPEN_READONLY);
+ /+ ↓ whether this file has a compression column at all. A database
+ written before the column has none, and selecting it would throw
+ rather than answer; everything in such a file is stored as read.
+ +/
+ bool _has_compression = false;
foreach (row; db.execute(
- "SELECT role, name, bytes, sha256, data FROM files WHERE role = '" ~ role
- ~ "' ORDER BY name")
+ "SELECT count(*) AS n FROM pragma_table_info('files')"
+ ~ " WHERE name = 'compression'")
) {
+ _has_compression = (row["n"].as!long > 0);
+ }
+ auto _stmt = db.prepare(
+ "SELECT role, name, bytes, sha256, "
+ ~ (_has_compression ? "compression" : "NULL AS compression")
+ ~ ", data FROM files WHERE role = :role ORDER BY name"
+ );
+ _stmt.bind(":role", role);
+ foreach (row; _stmt.execute()) {
ST_ArtefactFile _f;
_f.role = row["role"].as!string;
_f.name = row["name"].as!string;
_f.bytes = row["bytes"].as!ulong;
_f.sha256 = row["sha256"].as!string;
_f.data = row["data"].as!(ubyte[]);
+ string _compression = row["compression"].as!string;
+ if (_compression == "zstd") {
+ try {
+ _f.data = zstdDecompress(_f.data);
+ } catch (Exception ex) {
+ /+ ↓ a row that says it is compressed and will not decompress is
+ not usable, and handing back the frame as if it were the file
+ would put a zstd frame where markup should be. Named, and
+ left out.
+ +/
+ stderr.writeln("WARNING: ", db_file.baseName, " carries ", _f.name,
+ " as zstd and it will not decompress: ", ex.msg, "; left out");
+ continue;
+ }
+ }
_out ~= _f;
}
+ _stmt.finalize();
return _out;
}
}
diff --git a/org/out_ocda_sqlite_db.org b/org/out_ocda_sqlite_db.org
index 08a4305..e8c8490 100644
--- a/org/out_ocda_sqlite_db.org
+++ b/org/out_ocda_sqlite_db.org
@@ -44,6 +44,7 @@ template spineAbstractionDb() {
import sisudoc.outputs.io_out.paths_output;
import sisudoc.ocda.io_in.paths_source;
import sisudoc.ocda.abstraction.ssp : ssp_format_version;
+ import sisudoc.ocda.zstd : zstdCompress;
/+ ↓ the document is passed in rather than taken from doc, because the
caller hands over one that has been through the .ssp: written as text
and read back. so the database is a function of the .ssp and cannot
@@ -233,20 +234,34 @@ template spineAbstractionDb() {
-- materialised from this database has to put them.
-- conf conf/document_make, the document's own configuration.
-- manifest pod.manifest, as the author wrote it.
+ -- tools tools/po4a, the translation catalogues, the same tree
+ -- the pod carries: a pod written back out of this file is
+ -- then the pod that went in, and not most of it.
--
-- source, conf and manifest are what make this a document source and
-- not only a serialised abstraction: with them a pod can be written
-- back out of the file and rebuilt from the markup, rather than only
-- rendered from the objects.
+ -- bytes and sha256 are always those of the ORIGINAL file, whatever
+ -- compression says, so that a digest check, a digests.txt line and a
+ -- source.digest rebuilt from these rows all keep working, and a reader
+ -- that does not decompress can still say what it is looking at.
+ --
+ -- compression says how data is stored: NULL or 'none' for the bytes as
+ -- read, 'zstd' for a zstd frame. It is decided by role and not by size,
+ -- so the same kind of file always stores the same way and no reader has
+ -- to measure anything to predict it: the text roles are compressed and
+ -- images are not, png and jpeg being compressed already.
CREATE TABLE IF NOT EXISTS files (
- id INTEGER PRIMARY KEY,
- role TEXT NOT NULL,
- name TEXT NOT NULL,
- bytes INTEGER NOT NULL,
- sha256 TEXT,
- width INTEGER,
- height INTEGER,
- data BLOB,
+ id INTEGER PRIMARY KEY,
+ role TEXT NOT NULL,
+ name TEXT NOT NULL,
+ bytes INTEGER NOT NULL,
+ sha256 TEXT,
+ width INTEGER,
+ height INTEGER,
+ compression TEXT,
+ data BLOB,
UNIQUE(role, name)
);
@@ -686,8 +701,9 @@ template spineAbstractionDb() {
import std.digest.sha : sha256Of;
auto file_stmt = db.prepare(
"INSERT OR IGNORE INTO files"
- ~ " (role, name, bytes, sha256, width, height, data)"
- ~ " VALUES (:role, :name, :bytes, :sha256, :width, :height, :data)"
+ ~ " (role, name, bytes, sha256, width, height, compression, data)"
+ ~ " VALUES (:role, :name, :bytes, :sha256, :width, :height,"
+ ~ " :compression, :data)"
);
int[string] _w, _h;
foreach (section; section_order) {
@@ -727,6 +743,7 @@ template spineAbstractionDb() {
import std.typecons : Nullable;
file_stmt.bind(":height", Nullable!int());
}
+ file_stmt.bind(":compression", "none");
file_stmt.bind(":data", _bytes);
file_stmt.execute();
file_stmt.reset();
@@ -758,13 +775,35 @@ template spineAbstractionDb() {
}
return;
}
- file_stmt.bind(":role", _role);
- file_stmt.bind(":name", _name);
- file_stmt.bind(":bytes", cast(long) _bytes.length);
- file_stmt.bind(":sha256", sha256Of(_bytes).toHexString.to!string);
- file_stmt.bind(":width", Nullable!int());
- file_stmt.bind(":height", Nullable!int());
- file_stmt.bind(":data", _bytes);
+ /+ ↓ the text roles are stored compressed, the images are not.
+ By role and not by size: the same kind of file then always
+ stores the same way, and nobody has to measure a document to
+ know how its database will look. The bytes and the digest
+ recorded stay those of the original file either way.
+ .
+ A tiny file can come out a few bytes larger, a zstd frame
+ having some overhead. Left alone: a rule to avoid that would
+ be the size threshold this deliberately does not have.
+ +/
+ ubyte[] _stored = _bytes;
+ string _compression = "none";
+ try {
+ _stored = zstdCompress(_bytes, doc_matters.opt.action.pod_compression);
+ _compression = "zstd";
+ } catch (Exception ex) {
+ stderr.writeln("WARNING: could not compress ", _name, ": ", ex.msg,
+ "; stored as it stands");
+ _stored = _bytes;
+ _compression = "none";
+ }
+ file_stmt.bind(":role", _role);
+ file_stmt.bind(":name", _name);
+ file_stmt.bind(":bytes", cast(long) _bytes.length);
+ file_stmt.bind(":sha256", sha256Of(_bytes).toHexString.to!string);
+ file_stmt.bind(":width", Nullable!int());
+ file_stmt.bind(":height", Nullable!int());
+ file_stmt.bind(":compression", _compression);
+ file_stmt.bind(":data", _stored);
file_stmt.execute();
file_stmt.reset();
}
@@ -790,6 +829,38 @@ template spineAbstractionDb() {
_carry("manifest", "pod.manifest",
doc_matters.pod.manifest_file_with_path);
}
+ /+ ↓ the translation catalogues, the same tools/po4a tree the pod
+ writer carries and by the same rule: a directory walk, because a
+ catalogue set is whatever the translator has, with every name
+ checked before it is taken. Carrying less than the pod does
+ would make a materialised pod a lossy copy of the original,
+ which is the one thing this is for.
+ .
+ Done once, on the last language, as the pod writer does it (the
+ tree is not language specific, and a pass per language would
+ re-read all of it (5.5 MB ten times over, for live-manual) to
+ insert rows the first pass already wrote).
+ +/
+ if (doc_matters.src.is_pod
+ && doc_matters.src.language == doc_matters.pod.manifest_list_of_languages[$-1]
+ ) {
+ import sisudoc.ocda.io_in.carried_names;
+ mixin spineCarriedNames;
+ enum string _po4a_root = "tools/po4a";
+ string _tools_in = doc_matters.pod.manifest_path ~ "/" ~ _po4a_root;
+ if (exists(_tools_in) && _tools_in.isDir) {
+ foreach (string _f; dirEntries(_tools_in, SpanMode.depth)) {
+ if (!(_f.isFile)) { continue; }
+ string _rel = _po4a_root ~ "/" ~ _f[(_tools_in.length + 1) .. $];
+ string _bad = validateCarriedPath(_rel);
+ if (_bad.length > 0) {
+ writeln("WARNING not carried into the database: ", _bad);
+ continue;
+ }
+ _carry("tools", _rel, _f);
+ }
+ }
+ }
}
file_stmt.finalize();
}
diff --git a/src/sisudoc/ocda/abstraction/db_in.d b/src/sisudoc/ocda/abstraction/db_in.d
index 53ac95d..0a01364 100644
--- a/src/sisudoc/ocda/abstraction/db_in.d
+++ b/src/sisudoc/ocda/abstraction/db_in.d
@@ -380,7 +380,13 @@ template spineAbstractionDbRead() {
a .ocda.db holds its images as blobs, which is what lets it travel on
its own where a .ssp cannot: a .ssp describes its images by name,
size, pixels and sha256, but does not carry them. The role column
- says what a file is; today only 'image' is written.
+ says what a file is: 'image', and the markup, configuration, manifest
+ and tools carried with it.
+ .
+ data comes back as the original file, decompressed where the row says
+ it was compressed, so that no caller has to know how a blob is stored.
+ bytes and sha256 are the original's either way, and are what a caller
+ checks the data it is given against.
+/
struct ST_ArtefactFile {
string role;
@@ -390,21 +396,52 @@ template spineAbstractionDbRead() {
ubyte[] data;
}
@trusted ST_ArtefactFile[] dbReadFiles(string db_file, string role = "image") {
+ import sisudoc.ocda.zstd : zstdDecompress;
ST_ArtefactFile[] _out;
if (!db_file.exists) { return _out; }
auto db = Database(db_file, SQLITE_OPEN_READONLY);
+ /+ ↓ whether this file has a compression column at all. A database
+ written before the column has none, and selecting it would throw
+ rather than answer; everything in such a file is stored as read.
+ +/
+ bool _has_compression = false;
foreach (row; db.execute(
- "SELECT role, name, bytes, sha256, data FROM files WHERE role = '" ~ role
- ~ "' ORDER BY name")
+ "SELECT count(*) AS n FROM pragma_table_info('files')"
+ ~ " WHERE name = 'compression'")
) {
+ _has_compression = (row["n"].as!long > 0);
+ }
+ auto _stmt = db.prepare(
+ "SELECT role, name, bytes, sha256, "
+ ~ (_has_compression ? "compression" : "NULL AS compression")
+ ~ ", data FROM files WHERE role = :role ORDER BY name"
+ );
+ _stmt.bind(":role", role);
+ foreach (row; _stmt.execute()) {
ST_ArtefactFile _f;
_f.role = row["role"].as!string;
_f.name = row["name"].as!string;
_f.bytes = row["bytes"].as!ulong;
_f.sha256 = row["sha256"].as!string;
_f.data = row["data"].as!(ubyte[]);
+ string _compression = row["compression"].as!string;
+ if (_compression == "zstd") {
+ try {
+ _f.data = zstdDecompress(_f.data);
+ } catch (Exception ex) {
+ /+ ↓ a row that says it is compressed and will not decompress is
+ not usable, and handing back the frame as if it were the file
+ would put a zstd frame where markup should be. Named, and
+ left out.
+ +/
+ stderr.writeln("WARNING: ", db_file.baseName, " carries ", _f.name,
+ " as zstd and it will not decompress: ", ex.msg, "; left out");
+ continue;
+ }
+ }
_out ~= _f;
}
+ _stmt.finalize();
return _out;
}
}
diff --git a/src/sisudoc/ocda/abstraction/pod_from_db.d b/src/sisudoc/ocda/abstraction/pod_from_db.d
index 0bdb3bc..2e231d3 100644
--- a/src/sisudoc/ocda/abstraction/pod_from_db.d
+++ b/src/sisudoc/ocda/abstraction/pod_from_db.d
@@ -74,6 +74,7 @@ module sisudoc.ocda.abstraction.pod_from_db;
<dest>/<pod>/conf/document_make role='conf'
<dest>/<pod>/media/text/<lang>/<file> role='source'
<dest>/<pod>/media/image/<file> role='image'
+ <dest>/<pod>/tools/po4a/... role='tools'
.
The pod's name comes from the database's own filename: <doc>.ocda.db is
named by doc_uid_out_no_lang, which is the pod name and the document's
@@ -136,7 +137,8 @@ template spinePodFromDb() {
case "image": return "media/image/" ~ _name;
case "source":
case "conf":
- case "manifest": return _name;
+ case "manifest":
+ case "tools": return _name;
default: return "";
}
}
@@ -153,7 +155,7 @@ template spinePodFromDb() {
return _out;
}
_dbr.ST_ArtefactFile[] _files;
- foreach (_role; ["manifest", "conf", "source", "image"]) {
+ foreach (_role; ["manifest", "conf", "source", "image", "tools"]) {
_files ~= _dbr.dbReadFiles(_db_file, _role);
}
if (_files.length == 0) {
diff --git a/src/sisudoc/outputs/io_out/sqlite_ocda_db.d b/src/sisudoc/outputs/io_out/sqlite_ocda_db.d
index d30295a..f312d91 100644
--- a/src/sisudoc/outputs/io_out/sqlite_ocda_db.d
+++ b/src/sisudoc/outputs/io_out/sqlite_ocda_db.d
@@ -62,6 +62,7 @@ template spineAbstractionDb() {
import sisudoc.outputs.io_out.paths_output;
import sisudoc.ocda.io_in.paths_source;
import sisudoc.ocda.abstraction.ssp : ssp_format_version;
+ import sisudoc.ocda.zstd : zstdCompress;
/+ ↓ the document is passed in rather than taken from doc, because the
caller hands over one that has been through the .ssp: written as text
and read back. so the database is a function of the .ssp and cannot
@@ -251,20 +252,34 @@ template spineAbstractionDb() {
-- materialised from this database has to put them.
-- conf conf/document_make, the document's own configuration.
-- manifest pod.manifest, as the author wrote it.
+ -- tools tools/po4a, the translation catalogues, the same tree
+ -- the pod carries: a pod written back out of this file is
+ -- then the pod that went in, and not most of it.
--
-- source, conf and manifest are what make this a document source and
-- not only a serialised abstraction: with them a pod can be written
-- back out of the file and rebuilt from the markup, rather than only
-- rendered from the objects.
+ -- bytes and sha256 are always those of the ORIGINAL file, whatever
+ -- compression says, so that a digest check, a digests.txt line and a
+ -- source.digest rebuilt from these rows all keep working, and a reader
+ -- that does not decompress can still say what it is looking at.
+ --
+ -- compression says how data is stored: NULL or 'none' for the bytes as
+ -- read, 'zstd' for a zstd frame. It is decided by role and not by size,
+ -- so the same kind of file always stores the same way and no reader has
+ -- to measure anything to predict it: the text roles are compressed and
+ -- images are not, png and jpeg being compressed already.
CREATE TABLE IF NOT EXISTS files (
- id INTEGER PRIMARY KEY,
- role TEXT NOT NULL,
- name TEXT NOT NULL,
- bytes INTEGER NOT NULL,
- sha256 TEXT,
- width INTEGER,
- height INTEGER,
- data BLOB,
+ id INTEGER PRIMARY KEY,
+ role TEXT NOT NULL,
+ name TEXT NOT NULL,
+ bytes INTEGER NOT NULL,
+ sha256 TEXT,
+ width INTEGER,
+ height INTEGER,
+ compression TEXT,
+ data BLOB,
UNIQUE(role, name)
);
@@ -704,8 +719,9 @@ template spineAbstractionDb() {
import std.digest.sha : sha256Of;
auto file_stmt = db.prepare(
"INSERT OR IGNORE INTO files"
- ~ " (role, name, bytes, sha256, width, height, data)"
- ~ " VALUES (:role, :name, :bytes, :sha256, :width, :height, :data)"
+ ~ " (role, name, bytes, sha256, width, height, compression, data)"
+ ~ " VALUES (:role, :name, :bytes, :sha256, :width, :height,"
+ ~ " :compression, :data)"
);
int[string] _w, _h;
foreach (section; section_order) {
@@ -745,6 +761,7 @@ template spineAbstractionDb() {
import std.typecons : Nullable;
file_stmt.bind(":height", Nullable!int());
}
+ file_stmt.bind(":compression", "none");
file_stmt.bind(":data", _bytes);
file_stmt.execute();
file_stmt.reset();
@@ -776,13 +793,35 @@ template spineAbstractionDb() {
}
return;
}
- file_stmt.bind(":role", _role);
- file_stmt.bind(":name", _name);
- file_stmt.bind(":bytes", cast(long) _bytes.length);
- file_stmt.bind(":sha256", sha256Of(_bytes).toHexString.to!string);
- file_stmt.bind(":width", Nullable!int());
- file_stmt.bind(":height", Nullable!int());
- file_stmt.bind(":data", _bytes);
+ /+ ↓ the text roles are stored compressed, the images are not.
+ By role and not by size: the same kind of file then always
+ stores the same way, and nobody has to measure a document to
+ know how its database will look. The bytes and the digest
+ recorded stay those of the original file either way.
+ .
+ A tiny file can come out a few bytes larger, a zstd frame
+ having some overhead. Left alone: a rule to avoid that would
+ be the size threshold this deliberately does not have.
+ +/
+ ubyte[] _stored = _bytes;
+ string _compression = "none";
+ try {
+ _stored = zstdCompress(_bytes, doc_matters.opt.action.pod_compression);
+ _compression = "zstd";
+ } catch (Exception ex) {
+ stderr.writeln("WARNING: could not compress ", _name, ": ", ex.msg,
+ "; stored as it stands");
+ _stored = _bytes;
+ _compression = "none";
+ }
+ file_stmt.bind(":role", _role);
+ file_stmt.bind(":name", _name);
+ file_stmt.bind(":bytes", cast(long) _bytes.length);
+ file_stmt.bind(":sha256", sha256Of(_bytes).toHexString.to!string);
+ file_stmt.bind(":width", Nullable!int());
+ file_stmt.bind(":height", Nullable!int());
+ file_stmt.bind(":compression", _compression);
+ file_stmt.bind(":data", _stored);
file_stmt.execute();
file_stmt.reset();
}
@@ -808,6 +847,38 @@ template spineAbstractionDb() {
_carry("manifest", "pod.manifest",
doc_matters.pod.manifest_file_with_path);
}
+ /+ ↓ the translation catalogues, the same tools/po4a tree the pod
+ writer carries and by the same rule: a directory walk, because a
+ catalogue set is whatever the translator has, with every name
+ checked before it is taken. Carrying less than the pod does
+ would make a materialised pod a lossy copy of the original,
+ which is the one thing this is for.
+ .
+ Done once, on the last language, as the pod writer does it (the
+ tree is not language specific, and a pass per language would
+ re-read all of it (5.5 MB ten times over, for live-manual) to
+ insert rows the first pass already wrote).
+ +/
+ if (doc_matters.src.is_pod
+ && doc_matters.src.language == doc_matters.pod.manifest_list_of_languages[$-1]
+ ) {
+ import sisudoc.ocda.io_in.carried_names;
+ mixin spineCarriedNames;
+ enum string _po4a_root = "tools/po4a";
+ string _tools_in = doc_matters.pod.manifest_path ~ "/" ~ _po4a_root;
+ if (exists(_tools_in) && _tools_in.isDir) {
+ foreach (string _f; dirEntries(_tools_in, SpanMode.depth)) {
+ if (!(_f.isFile)) { continue; }
+ string _rel = _po4a_root ~ "/" ~ _f[(_tools_in.length + 1) .. $];
+ string _bad = validateCarriedPath(_rel);
+ if (_bad.length > 0) {
+ writeln("WARNING not carried into the database: ", _bad);
+ continue;
+ }
+ _carry("tools", _rel, _f);
+ }
+ }
+ }
}
file_stmt.finalize();
}