diff options
| -rw-r--r-- | org/out_ocda_sqlite_db.org | 39 | ||||
| -rw-r--r-- | src/sisudoc/outputs/io_out/sqlite_ocda_db.d | 39 |
2 files changed, 58 insertions, 20 deletions
diff --git a/org/out_ocda_sqlite_db.org b/org/out_ocda_sqlite_db.org index 1c8a3c7..8a42965 100644 --- a/org/out_ocda_sqlite_db.org +++ b/org/out_ocda_sqlite_db.org @@ -70,15 +70,28 @@ template spineAbstractionDb() { } } catch (Exception ex) { } + /+ ↓ named for the document, not for one of its languages: the file holds + every language, so the name is doc_uid_out without its ".<lang>". + +/ string db_file = ((base_pth.chainPath( - doc_matters.src.doc_uid_out ~ ".ocda.db")).asNormalizedPath).array; - - /+ ↓ remove existing file to start fresh +/ - try { - if (exists(db_file)) { - remove(db_file); + doc_matters.src.doc_uid_out_no_lang ~ ".ocda.db")).asNormalizedPath).array; + + /+ ↓ the file is started fresh on the first language of the document and + added to by the rest. + . + It used to be removed on every call, which was right when a call wrote + the whole file. Now a call writes one language of it, so removing would + leave only the last language of a ten language document (discarding the + others). Which language is first comes from the manifest, not from the + order the loop happens to run in, so this does not depend on the run. + +/ + if (doc_matters.src.language == doc_matters.pod.manifest_list_of_languages[0]) { + try { + if (exists(db_file)) { + remove(db_file); + } + } catch (Exception ex) { } - } catch (Exception ex) { } if (doc_matters.opt.action.vox_gt_1) { @@ -98,9 +111,9 @@ template spineAbstractionDb() { db.run("PRAGMA synchronous=OFF"); db.run(" - -- documents: a row per language of the document this database holds. - -- One today, since a database is written per language; the table is - -- what lets one hold a document's whole set, which is what it is for. + -- documents: a row per language of the document this database holds, + -- and a database holds the document's whole set: ten rows for a ten + -- language document, written one language at a time into the one file. -- Everything a reader needs to tell one language's rows from another's -- keys on documents.id, called doc_id wherever it is referred to. CREATE TABLE IF NOT EXISTS documents ( @@ -122,6 +135,12 @@ template spineAbstractionDb() { value TEXT NOT NULL, UNIQUE(doc_id, key) ); + -- the file level rows need a uniqueness rule of their own: in SQL no + -- null equals any other null, so UNIQUE(doc_id, key) does not constrain + -- them and schema.name would be written once per language. A partial + -- index over key alone, for the null rows, is what makes them singular + CREATE UNIQUE INDEX IF NOT EXISTS idx_metadata_file_key + ON metadata(key) WHERE doc_id IS NULL; -- objects: one row per abstraction object, in document order. -- the fixed width arrays (ancestors, dom status, children, heading -- ancestors text, table widths and aligns) are JSON arrays so that diff --git a/src/sisudoc/outputs/io_out/sqlite_ocda_db.d b/src/sisudoc/outputs/io_out/sqlite_ocda_db.d index d94fa26..998c317 100644 --- a/src/sisudoc/outputs/io_out/sqlite_ocda_db.d +++ b/src/sisudoc/outputs/io_out/sqlite_ocda_db.d @@ -88,15 +88,28 @@ template spineAbstractionDb() { } } catch (Exception ex) { } + /+ ↓ named for the document, not for one of its languages: the file holds + every language, so the name is doc_uid_out without its ".<lang>". + +/ string db_file = ((base_pth.chainPath( - doc_matters.src.doc_uid_out ~ ".ocda.db")).asNormalizedPath).array; - - /+ ↓ remove existing file to start fresh +/ - try { - if (exists(db_file)) { - remove(db_file); + doc_matters.src.doc_uid_out_no_lang ~ ".ocda.db")).asNormalizedPath).array; + + /+ ↓ the file is started fresh on the first language of the document and + added to by the rest. + . + It used to be removed on every call, which was right when a call wrote + the whole file. Now a call writes one language of it, so removing would + leave only the last language of a ten language document (discarding the + others). Which language is first comes from the manifest, not from the + order the loop happens to run in, so this does not depend on the run. + +/ + if (doc_matters.src.language == doc_matters.pod.manifest_list_of_languages[0]) { + try { + if (exists(db_file)) { + remove(db_file); + } + } catch (Exception ex) { } - } catch (Exception ex) { } if (doc_matters.opt.action.vox_gt_1) { @@ -116,9 +129,9 @@ template spineAbstractionDb() { db.run("PRAGMA synchronous=OFF"); db.run(" - -- documents: a row per language of the document this database holds. - -- One today, since a database is written per language; the table is - -- what lets one hold a document's whole set, which is what it is for. + -- documents: a row per language of the document this database holds, + -- and a database holds the document's whole set: ten rows for a ten + -- language document, written one language at a time into the one file. -- Everything a reader needs to tell one language's rows from another's -- keys on documents.id, called doc_id wherever it is referred to. CREATE TABLE IF NOT EXISTS documents ( @@ -140,6 +153,12 @@ template spineAbstractionDb() { value TEXT NOT NULL, UNIQUE(doc_id, key) ); + -- the file level rows need a uniqueness rule of their own: in SQL no + -- null equals any other null, so UNIQUE(doc_id, key) does not constrain + -- them and schema.name would be written once per language. A partial + -- index over key alone, for the null rows, is what makes them singular + CREATE UNIQUE INDEX IF NOT EXISTS idx_metadata_file_key + ON metadata(key) WHERE doc_id IS NULL; -- objects: one row per abstraction object, in document order. -- the fixed width arrays (ancestors, dom status, children, heading -- ancestors text, table widths and aligns) are JSON arrays so that |
