aboutsummaryrefslogtreecommitdiffhomepage
diff options
context:
space:
mode:
-rw-r--r--org/in_markup_source_files.org34
-rw-r--r--org/in_zip_pod.org6
-rw-r--r--org/out_src_pod.org38
-rw-r--r--src/sisudoc/ocda/io_in/carried_names.d34
-rw-r--r--src/sisudoc/ocda/io_in/read_zip_pod.d6
-rw-r--r--src/sisudoc/outputs/io_out/source_pod.d38
6 files changed, 154 insertions, 2 deletions
diff --git a/org/in_markup_source_files.org b/org/in_markup_source_files.org
index ef1c913..7aa49dc 100644
--- a/org/in_markup_source_files.org
+++ b/org/in_markup_source_files.org
@@ -801,6 +801,40 @@ return ret;
+/
module sisudoc.ocda.io_in.carried_names;
@safe:
+/+ ↓ how many members one pod archive may hold, for both writer and the reader
+ .
+ It was 500 set in the reader alone, determined when a pod held markup and
+ images. A pod that carries translation catalogues has a different
+ calculation, the count is driven by how many *files* a document is made of
+ times the number of languages it has (rather than document size). Here are
+ figures based on the document sample set:
+ .
+ entries ~= languages x (markup_files + po_files + 1)
+ + pot_files + images + 2
+ .
+ the_wealth_of_networks 213,405 words, 1 file per language
+ 12 languages -> 49 entries
+ live-manual 24,724 words, 20 files per language
+ 12 languages -> 506 entries
+ .
+ (the big book is not what runs into this; the modular manual is. At 500
+ live-manual was two languages from producing an archive spine writes and
+ then refuses to read).
+ .
+ 10,000 covers a hundred-insert manual in thirty languages, about 6,200
+ entries, with room. The number is deliberately well clear of any real
+ document rather than snug above the largest one known, because the two
+ errors are not comparable: too low refuses a legitimate document with a
+ message that reads as corruption, while too high defers to a size cap a
+ moment later.
+ .
+ This is the weakest of the three guards and is not what bounds resource
+ use. MAX_ENTRY_SIZE bounds any one member and MAX_TOTAL_SIZE bounds the
+ whole extraction, both checked before a byte is written. A count bounds
+ metadata and per-file syscalls, which at this order of magnitude is
+ nothing beside what the size caps already allow.
++/
+enum size_t MAX_POD_ENTRY_COUNT = 10_000;
/+ ↓ names that arrive from outside, and decide where a file gets written
.
A pod zip and a .ocda.db both carry files by name, and both can arrive
diff --git a/org/in_zip_pod.org b/org/in_zip_pod.org
index c542916..2672155 100644
--- a/org/in_zip_pod.org
+++ b/org/in_zip_pod.org
@@ -40,11 +40,15 @@ template spineExtractZipPod() {
import std.stdio;
import std.string : indexOf;
import sisudoc.ocda.zstd;
+ import sisudoc.ocda.io_in.carried_names : MAX_POD_ENTRY_COUNT;
/+ security limits for zip extraction +/
enum size_t MAX_ENTRY_SIZE = 50 * 1024 * 1024; /+ 50 MB per entry +/
enum size_t MAX_TOTAL_SIZE = 500 * 1024 * 1024; /+ 500 MB total +/
- enum size_t MAX_ENTRY_COUNT = 500; /+ max entries in archive +/
+ /+ ↓ the entry count now comes from carried_names, so that the writer is
+ held to the same number the reader enforces.
+ +/
+ alias MAX_ENTRY_COUNT = MAX_POD_ENTRY_COUNT;
enum size_t MAX_PATH_DEPTH = 10; /+ max path components +/
/+ allowed entry name pattern: alphanumeric, dots, dashes, underscores, forward slashes +/
diff --git a/org/out_src_pod.org b/org/out_src_pod.org
index 96f784f..7efc52a 100644
--- a/org/out_src_pod.org
+++ b/org/out_src_pod.org
@@ -580,8 +580,46 @@ void podArchive_directory_tree(M,P)(M doc_matters, P pths_pod) { // create direc
#+BEGIN_SRC d
void zipArchive(M,F,Z)(M doc_matters, F fn_pod, Z zip) {
import sisudoc.ocda.zstd : zstdCompress, ZstdException;
+ import sisudoc.ocda.io_in.carried_names : MAX_POD_ENTRY_COUNT;
auto fn_src_in = doc_matters.src.filename;
if (doc_matters.opt.action.pod) {
+ /+ ↓ the writer is held to the number the reader enforces.
+ Previously it was not: a document with enough parts produced an
+ archive spine would write and then refused to read, reporting too many
+ entries, so a good file looked corrupt. The count is a judgement call;
+ a writer able to emit what the reader rejects is a fault whatever the
+ number is.
+ .
+ An error rather than a warning, because the artefact would be
+ unreadable: writing it and reporting success is the one outcome worth
+ ruling out. The message names the count, the limit and the document, so
+ that "this document has too many parts for one archive" is legible
+ without a debugger.
+ +/
+ if (zip.directory.length > MAX_POD_ENTRY_COUNT) {
+ writeln("ERROR >> pod archive not written: ", fn_pod);
+ writeln(" ", doc_matters.src.filename_base, " would hold ",
+ zip.directory.length, " members, over the ", MAX_POD_ENTRY_COUNT,
+ " a pod archive may carry.");
+ writeln(" a document's members are its languages times its markup",
+ " and catalogue files; spine would not be able to read back what",
+ " it wrote.");
+ /+ ↓ an archive left by an earlier pass is removed.
+ A missing artefact is an error anyone will notice; a quietly
+ truncated one is not.
+ +/
+ if (exists(fn_pod)) {
+ try {
+ fn_pod.remove;
+ writeln(" the incomplete archive left by an earlier language",
+ " has been removed.");
+ } catch (Exception ex) {
+ writeln("WARNING could not remove the incomplete archive: ",
+ fn_pod, " - ", ex.msg);
+ }
+ }
+ return;
+ }
if (exists(doc_matters.src.file_with_absolute_path)) {
try {
/+ ↓ the archive is built exactly as before and then wrapped in one
diff --git a/src/sisudoc/ocda/io_in/carried_names.d b/src/sisudoc/ocda/io_in/carried_names.d
index 09e8912..84a4f3c 100644
--- a/src/sisudoc/ocda/io_in/carried_names.d
+++ b/src/sisudoc/ocda/io_in/carried_names.d
@@ -56,6 +56,40 @@
+/
module sisudoc.ocda.io_in.carried_names;
@safe:
+/+ ↓ how many members one pod archive may hold, for both writer and the reader
+ .
+ It was 500 set in the reader alone, determined when a pod held markup and
+ images. A pod that carries translation catalogues has a different
+ calculation, the count is driven by how many *files* a document is made of
+ times the number of languages it has (rather than document size). Here are
+ figures based on the document sample set:
+ .
+ entries ~= languages x (markup_files + po_files + 1)
+ + pot_files + images + 2
+ .
+ the_wealth_of_networks 213,405 words, 1 file per language
+ 12 languages -> 49 entries
+ live-manual 24,724 words, 20 files per language
+ 12 languages -> 506 entries
+ .
+ (the big book is not what runs into this; the modular manual is. At 500
+ live-manual was two languages from producing an archive spine writes and
+ then refuses to read).
+ .
+ 10,000 covers a hundred-insert manual in thirty languages, about 6,200
+ entries, with room. The number is deliberately well clear of any real
+ document rather than snug above the largest one known, because the two
+ errors are not comparable: too low refuses a legitimate document with a
+ message that reads as corruption, while too high defers to a size cap a
+ moment later.
+ .
+ This is the weakest of the three guards and is not what bounds resource
+ use. MAX_ENTRY_SIZE bounds any one member and MAX_TOTAL_SIZE bounds the
+ whole extraction, both checked before a byte is written. A count bounds
+ metadata and per-file syscalls, which at this order of magnitude is
+ nothing beside what the size caps already allow.
++/
+enum size_t MAX_POD_ENTRY_COUNT = 10_000;
/+ ↓ names that arrive from outside, and decide where a file gets written
.
A pod zip and a .ocda.db both carry files by name, and both can arrive
diff --git a/src/sisudoc/ocda/io_in/read_zip_pod.d b/src/sisudoc/ocda/io_in/read_zip_pod.d
index d0eb65a..18a957c 100644
--- a/src/sisudoc/ocda/io_in/read_zip_pod.d
+++ b/src/sisudoc/ocda/io_in/read_zip_pod.d
@@ -64,11 +64,15 @@ template spineExtractZipPod() {
import std.stdio;
import std.string : indexOf;
import sisudoc.ocda.zstd;
+ import sisudoc.ocda.io_in.carried_names : MAX_POD_ENTRY_COUNT;
/+ security limits for zip extraction +/
enum size_t MAX_ENTRY_SIZE = 50 * 1024 * 1024; /+ 50 MB per entry +/
enum size_t MAX_TOTAL_SIZE = 500 * 1024 * 1024; /+ 500 MB total +/
- enum size_t MAX_ENTRY_COUNT = 500; /+ max entries in archive +/
+ /+ ↓ the entry count now comes from carried_names, so that the writer is
+ held to the same number the reader enforces.
+ +/
+ alias MAX_ENTRY_COUNT = MAX_POD_ENTRY_COUNT;
enum size_t MAX_PATH_DEPTH = 10; /+ max path components +/
/+ allowed entry name pattern: alphanumeric, dots, dashes, underscores, forward slashes +/
diff --git a/src/sisudoc/outputs/io_out/source_pod.d b/src/sisudoc/outputs/io_out/source_pod.d
index b929e06..bfbad70 100644
--- a/src/sisudoc/outputs/io_out/source_pod.d
+++ b/src/sisudoc/outputs/io_out/source_pod.d
@@ -547,8 +547,46 @@ template spinePod() {
}
void zipArchive(M,F,Z)(M doc_matters, F fn_pod, Z zip) {
import sisudoc.ocda.zstd : zstdCompress, ZstdException;
+ import sisudoc.ocda.io_in.carried_names : MAX_POD_ENTRY_COUNT;
auto fn_src_in = doc_matters.src.filename;
if (doc_matters.opt.action.pod) {
+ /+ ↓ the writer is held to the number the reader enforces.
+ Previously it was not: a document with enough parts produced an
+ archive spine would write and then refused to read, reporting too many
+ entries, so a good file looked corrupt. The count is a judgement call;
+ a writer able to emit what the reader rejects is a fault whatever the
+ number is.
+ .
+ An error rather than a warning, because the artefact would be
+ unreadable: writing it and reporting success is the one outcome worth
+ ruling out. The message names the count, the limit and the document, so
+ that "this document has too many parts for one archive" is legible
+ without a debugger.
+ +/
+ if (zip.directory.length > MAX_POD_ENTRY_COUNT) {
+ writeln("ERROR >> pod archive not written: ", fn_pod);
+ writeln(" ", doc_matters.src.filename_base, " would hold ",
+ zip.directory.length, " members, over the ", MAX_POD_ENTRY_COUNT,
+ " a pod archive may carry.");
+ writeln(" a document's members are its languages times its markup",
+ " and catalogue files; spine would not be able to read back what",
+ " it wrote.");
+ /+ ↓ an archive left by an earlier pass is removed.
+ A missing artefact is an error anyone will notice; a quietly
+ truncated one is not.
+ +/
+ if (exists(fn_pod)) {
+ try {
+ fn_pod.remove;
+ writeln(" the incomplete archive left by an earlier language",
+ " has been removed.");
+ } catch (Exception ex) {
+ writeln("WARNING could not remove the incomplete archive: ",
+ fn_pod, " - ", ex.msg);
+ }
+ }
+ return;
+ }
if (exists(doc_matters.src.file_with_absolute_path)) {
try {
/+ ↓ the archive is built exactly as before and then wrapped in one