aboutsummaryrefslogtreecommitdiffhomepage
path: root/org/out_ocda_sqlite_db.org
diff options
context:
space:
mode:
authorRalph Amissah <ralph.amissah@gmail.com>2026-09-21 11:14:58 -0400
committerRalph Amissah <ralph.amissah@gmail.com>2026-09-22 14:43:16 -0400
commitdc50cca5e2ae3f13f1a585108b116245fc68a9ca (patch)
tree25b08fd30adff25cbf88bfe7d8653768c2921bb4 /org/out_ocda_sqlite_db.org
parentpaths: doc_uid_out_no_lang, doc name sans language (diff)
ocda db: one <doc>.ocda.db holding every language
The file is named for the document, not for one of its languages (and is written a language at a time: ten calls fill one file for a ten language document). The removal on entry was right when a call wrote the whole file. It now happens on the first language of the document, taken from the manifest rather than from whatever order the loop ran in, so nine languages are no longer written and thrown away. Also a partial unique index for the file level metadata rows. In SQL no null equals any other null, so UNIQUE(doc_id, key) does not constrain the rows written with a null doc_id, and schema.name and schema.version were inserted once per language: ten rows each, with nothing for INSERT OR REPLACE to replace. (assisted by Claude-Code)
Diffstat (limited to 'org/out_ocda_sqlite_db.org')
-rw-r--r--org/out_ocda_sqlite_db.org39
1 files changed, 29 insertions, 10 deletions
diff --git a/org/out_ocda_sqlite_db.org b/org/out_ocda_sqlite_db.org
index 1c8a3c7..8a42965 100644
--- a/org/out_ocda_sqlite_db.org
+++ b/org/out_ocda_sqlite_db.org
@@ -70,15 +70,28 @@ template spineAbstractionDb() {
}
} catch (Exception ex) {
}
+ /+ ↓ named for the document, not for one of its languages: the file holds
+ every language, so the name is doc_uid_out without its ".<lang>".
+ +/
string db_file = ((base_pth.chainPath(
- doc_matters.src.doc_uid_out ~ ".ocda.db")).asNormalizedPath).array;
-
- /+ ↓ remove existing file to start fresh +/
- try {
- if (exists(db_file)) {
- remove(db_file);
+ doc_matters.src.doc_uid_out_no_lang ~ ".ocda.db")).asNormalizedPath).array;
+
+ /+ ↓ the file is started fresh on the first language of the document and
+ added to by the rest.
+ .
+ It used to be removed on every call, which was right when a call wrote
+ the whole file. Now a call writes one language of it, so removing would
+ leave only the last language of a ten language document (discarding the
+ others). Which language is first comes from the manifest, not from the
+ order the loop happens to run in, so this does not depend on the run.
+ +/
+ if (doc_matters.src.language == doc_matters.pod.manifest_list_of_languages[0]) {
+ try {
+ if (exists(db_file)) {
+ remove(db_file);
+ }
+ } catch (Exception ex) {
}
- } catch (Exception ex) {
}
if (doc_matters.opt.action.vox_gt_1) {
@@ -98,9 +111,9 @@ template spineAbstractionDb() {
db.run("PRAGMA synchronous=OFF");
db.run("
- -- documents: a row per language of the document this database holds.
- -- One today, since a database is written per language; the table is
- -- what lets one hold a document's whole set, which is what it is for.
+ -- documents: a row per language of the document this database holds,
+ -- and a database holds the document's whole set: ten rows for a ten
+ -- language document, written one language at a time into the one file.
-- Everything a reader needs to tell one language's rows from another's
-- keys on documents.id, called doc_id wherever it is referred to.
CREATE TABLE IF NOT EXISTS documents (
@@ -122,6 +135,12 @@ template spineAbstractionDb() {
value TEXT NOT NULL,
UNIQUE(doc_id, key)
);
+ -- the file level rows need a uniqueness rule of their own: in SQL no
+ -- null equals any other null, so UNIQUE(doc_id, key) does not constrain
+ -- them and schema.name would be written once per language. A partial
+ -- index over key alone, for the null rows, is what makes them singular
+ CREATE UNIQUE INDEX IF NOT EXISTS idx_metadata_file_key
+ ON metadata(key) WHERE doc_id IS NULL;
-- objects: one row per abstraction object, in document order.
-- the fixed width arrays (ancestors, dom status, children, heading
-- ancestors text, table widths and aligns) are JSON arrays so that