diff options
| -rw-r--r-- | org/in_markup_source_files.org | 34 | ||||
| -rw-r--r-- | org/in_zip_pod.org | 6 | ||||
| -rw-r--r-- | org/out_src_pod.org | 38 | ||||
| -rw-r--r-- | src/sisudoc/ocda/io_in/carried_names.d | 34 | ||||
| -rw-r--r-- | src/sisudoc/ocda/io_in/read_zip_pod.d | 6 | ||||
| -rw-r--r-- | src/sisudoc/outputs/io_out/source_pod.d | 38 |
6 files changed, 154 insertions, 2 deletions
diff --git a/org/in_markup_source_files.org b/org/in_markup_source_files.org index ef1c913..7aa49dc 100644 --- a/org/in_markup_source_files.org +++ b/org/in_markup_source_files.org @@ -801,6 +801,40 @@ return ret; +/ module sisudoc.ocda.io_in.carried_names; @safe: +/+ ↓ how many members one pod archive may hold, for both writer and the reader + . + It was 500 set in the reader alone, determined when a pod held markup and + images. A pod that carries translation catalogues has a different + calculation, the count is driven by how many *files* a document is made of + times the number of languages it has (rather than document size). Here are + figures based on the document sample set: + . + entries ~= languages x (markup_files + po_files + 1) + + pot_files + images + 2 + . + the_wealth_of_networks 213,405 words, 1 file per language + 12 languages -> 49 entries + live-manual 24,724 words, 20 files per language + 12 languages -> 506 entries + . + (the big book is not what runs into this; the modular manual is. At 500 + live-manual was two languages from producing an archive spine writes and + then refuses to read). + . + 10,000 covers a hundred-insert manual in thirty languages, about 6,200 + entries, with room. The number is deliberately well clear of any real + document rather than snug above the largest one known, because the two + errors are not comparable: too low refuses a legitimate document with a + message that reads as corruption, while too high defers to a size cap a + moment later. + . + This is the weakest of the three guards and is not what bounds resource + use. MAX_ENTRY_SIZE bounds any one member and MAX_TOTAL_SIZE bounds the + whole extraction, both checked before a byte is written. A count bounds + metadata and per-file syscalls, which at this order of magnitude is + nothing beside what the size caps already allow. ++/ +enum size_t MAX_POD_ENTRY_COUNT = 10_000; /+ ↓ names that arrive from outside, and decide where a file gets written . A pod zip and a .ocda.db both carry files by name, and both can arrive diff --git a/org/in_zip_pod.org b/org/in_zip_pod.org index c542916..2672155 100644 --- a/org/in_zip_pod.org +++ b/org/in_zip_pod.org @@ -40,11 +40,15 @@ template spineExtractZipPod() { import std.stdio; import std.string : indexOf; import sisudoc.ocda.zstd; + import sisudoc.ocda.io_in.carried_names : MAX_POD_ENTRY_COUNT; /+ security limits for zip extraction +/ enum size_t MAX_ENTRY_SIZE = 50 * 1024 * 1024; /+ 50 MB per entry +/ enum size_t MAX_TOTAL_SIZE = 500 * 1024 * 1024; /+ 500 MB total +/ - enum size_t MAX_ENTRY_COUNT = 500; /+ max entries in archive +/ + /+ ↓ the entry count now comes from carried_names, so that the writer is + held to the same number the reader enforces. + +/ + alias MAX_ENTRY_COUNT = MAX_POD_ENTRY_COUNT; enum size_t MAX_PATH_DEPTH = 10; /+ max path components +/ /+ allowed entry name pattern: alphanumeric, dots, dashes, underscores, forward slashes +/ diff --git a/org/out_src_pod.org b/org/out_src_pod.org index 96f784f..7efc52a 100644 --- a/org/out_src_pod.org +++ b/org/out_src_pod.org @@ -580,8 +580,46 @@ void podArchive_directory_tree(M,P)(M doc_matters, P pths_pod) { // create direc #+BEGIN_SRC d void zipArchive(M,F,Z)(M doc_matters, F fn_pod, Z zip) { import sisudoc.ocda.zstd : zstdCompress, ZstdException; + import sisudoc.ocda.io_in.carried_names : MAX_POD_ENTRY_COUNT; auto fn_src_in = doc_matters.src.filename; if (doc_matters.opt.action.pod) { + /+ ↓ the writer is held to the number the reader enforces. + Previously it was not: a document with enough parts produced an + archive spine would write and then refused to read, reporting too many + entries, so a good file looked corrupt. The count is a judgement call; + a writer able to emit what the reader rejects is a fault whatever the + number is. + . + An error rather than a warning, because the artefact would be + unreadable: writing it and reporting success is the one outcome worth + ruling out. The message names the count, the limit and the document, so + that "this document has too many parts for one archive" is legible + without a debugger. + +/ + if (zip.directory.length > MAX_POD_ENTRY_COUNT) { + writeln("ERROR >> pod archive not written: ", fn_pod); + writeln(" ", doc_matters.src.filename_base, " would hold ", + zip.directory.length, " members, over the ", MAX_POD_ENTRY_COUNT, + " a pod archive may carry."); + writeln(" a document's members are its languages times its markup", + " and catalogue files; spine would not be able to read back what", + " it wrote."); + /+ ↓ an archive left by an earlier pass is removed. + A missing artefact is an error anyone will notice; a quietly + truncated one is not. + +/ + if (exists(fn_pod)) { + try { + fn_pod.remove; + writeln(" the incomplete archive left by an earlier language", + " has been removed."); + } catch (Exception ex) { + writeln("WARNING could not remove the incomplete archive: ", + fn_pod, " - ", ex.msg); + } + } + return; + } if (exists(doc_matters.src.file_with_absolute_path)) { try { /+ ↓ the archive is built exactly as before and then wrapped in one diff --git a/src/sisudoc/ocda/io_in/carried_names.d b/src/sisudoc/ocda/io_in/carried_names.d index 09e8912..84a4f3c 100644 --- a/src/sisudoc/ocda/io_in/carried_names.d +++ b/src/sisudoc/ocda/io_in/carried_names.d @@ -56,6 +56,40 @@ +/ module sisudoc.ocda.io_in.carried_names; @safe: +/+ ↓ how many members one pod archive may hold, for both writer and the reader + . + It was 500 set in the reader alone, determined when a pod held markup and + images. A pod that carries translation catalogues has a different + calculation, the count is driven by how many *files* a document is made of + times the number of languages it has (rather than document size). Here are + figures based on the document sample set: + . + entries ~= languages x (markup_files + po_files + 1) + + pot_files + images + 2 + . + the_wealth_of_networks 213,405 words, 1 file per language + 12 languages -> 49 entries + live-manual 24,724 words, 20 files per language + 12 languages -> 506 entries + . + (the big book is not what runs into this; the modular manual is. At 500 + live-manual was two languages from producing an archive spine writes and + then refuses to read). + . + 10,000 covers a hundred-insert manual in thirty languages, about 6,200 + entries, with room. The number is deliberately well clear of any real + document rather than snug above the largest one known, because the two + errors are not comparable: too low refuses a legitimate document with a + message that reads as corruption, while too high defers to a size cap a + moment later. + . + This is the weakest of the three guards and is not what bounds resource + use. MAX_ENTRY_SIZE bounds any one member and MAX_TOTAL_SIZE bounds the + whole extraction, both checked before a byte is written. A count bounds + metadata and per-file syscalls, which at this order of magnitude is + nothing beside what the size caps already allow. ++/ +enum size_t MAX_POD_ENTRY_COUNT = 10_000; /+ ↓ names that arrive from outside, and decide where a file gets written . A pod zip and a .ocda.db both carry files by name, and both can arrive diff --git a/src/sisudoc/ocda/io_in/read_zip_pod.d b/src/sisudoc/ocda/io_in/read_zip_pod.d index d0eb65a..18a957c 100644 --- a/src/sisudoc/ocda/io_in/read_zip_pod.d +++ b/src/sisudoc/ocda/io_in/read_zip_pod.d @@ -64,11 +64,15 @@ template spineExtractZipPod() { import std.stdio; import std.string : indexOf; import sisudoc.ocda.zstd; + import sisudoc.ocda.io_in.carried_names : MAX_POD_ENTRY_COUNT; /+ security limits for zip extraction +/ enum size_t MAX_ENTRY_SIZE = 50 * 1024 * 1024; /+ 50 MB per entry +/ enum size_t MAX_TOTAL_SIZE = 500 * 1024 * 1024; /+ 500 MB total +/ - enum size_t MAX_ENTRY_COUNT = 500; /+ max entries in archive +/ + /+ ↓ the entry count now comes from carried_names, so that the writer is + held to the same number the reader enforces. + +/ + alias MAX_ENTRY_COUNT = MAX_POD_ENTRY_COUNT; enum size_t MAX_PATH_DEPTH = 10; /+ max path components +/ /+ allowed entry name pattern: alphanumeric, dots, dashes, underscores, forward slashes +/ diff --git a/src/sisudoc/outputs/io_out/source_pod.d b/src/sisudoc/outputs/io_out/source_pod.d index b929e06..bfbad70 100644 --- a/src/sisudoc/outputs/io_out/source_pod.d +++ b/src/sisudoc/outputs/io_out/source_pod.d @@ -547,8 +547,46 @@ template spinePod() { } void zipArchive(M,F,Z)(M doc_matters, F fn_pod, Z zip) { import sisudoc.ocda.zstd : zstdCompress, ZstdException; + import sisudoc.ocda.io_in.carried_names : MAX_POD_ENTRY_COUNT; auto fn_src_in = doc_matters.src.filename; if (doc_matters.opt.action.pod) { + /+ ↓ the writer is held to the number the reader enforces. + Previously it was not: a document with enough parts produced an + archive spine would write and then refused to read, reporting too many + entries, so a good file looked corrupt. The count is a judgement call; + a writer able to emit what the reader rejects is a fault whatever the + number is. + . + An error rather than a warning, because the artefact would be + unreadable: writing it and reporting success is the one outcome worth + ruling out. The message names the count, the limit and the document, so + that "this document has too many parts for one archive" is legible + without a debugger. + +/ + if (zip.directory.length > MAX_POD_ENTRY_COUNT) { + writeln("ERROR >> pod archive not written: ", fn_pod); + writeln(" ", doc_matters.src.filename_base, " would hold ", + zip.directory.length, " members, over the ", MAX_POD_ENTRY_COUNT, + " a pod archive may carry."); + writeln(" a document's members are its languages times its markup", + " and catalogue files; spine would not be able to read back what", + " it wrote."); + /+ ↓ an archive left by an earlier pass is removed. + A missing artefact is an error anyone will notice; a quietly + truncated one is not. + +/ + if (exists(fn_pod)) { + try { + fn_pod.remove; + writeln(" the incomplete archive left by an earlier language", + " has been removed."); + } catch (Exception ex) { + writeln("WARNING could not remove the incomplete archive: ", + fn_pod, " - ", ex.msg); + } + } + return; + } if (exists(doc_matters.src.file_with_absolute_path)) { try { /+ ↓ the archive is built exactly as before and then wrapped in one |
