diff options
| author | Ralph Amissah <ralph.amissah@gmail.com> | 2026-09-21 13:03:03 -0400 |
|---|---|---|
| committer | Ralph Amissah <ralph.amissah@gmail.com> | 2026-09-22 15:11:02 -0400 |
| commit | ea1162e5cf6182ad296b4775c7acfb51ac435842 (patch) | |
| tree | f72e52aac4f2950e36d1c17d7cd974b4942de727 | |
| parent | ocda db: now carry markup source, conf & manifest (diff) | |
test: the carried markup, by rebuilding its digest
Check that they are the right files: source.digest is the sha256 over
one line per markup file, "<SHA256> <basename>", sorted and newline
joined, so the same value can be computed from the stored rows alone and
held against what the .ssp recorded. A file stored under the right name
with the wrong contents passes a count and fails this.
Two details the shell gets wrong by default: the sort has to be
LC_ALL=C, because the writer sorts by byte and a locale does not, and
the lines are joined with no trailing newline, so printf and not echo.
(assisted by Claude-Code)
| -rw-r--r-- | org/tests_for_document_abstraction_shell_scripts.org | 33 | ||||
| -rwxr-xr-x | test/test-abstraction-db.sh | 33 |
2 files changed, 66 insertions, 0 deletions
diff --git a/org/tests_for_document_abstraction_shell_scripts.org b/org/tests_for_document_abstraction_shell_scripts.org index b4d71f8..ab2dc35 100644 --- a/org/tests_for_document_abstraction_shell_scripts.org +++ b/org/tests_for_document_abstraction_shell_scripts.org @@ -682,6 +682,39 @@ for ssp in $(find "$OUT_DIR/pod" -name "*.ssp" 2>/dev/null | sort); do fi rm -f "$OUT_DIR/.sspmeta.$$" "$OUT_DIR/.dbmeta.$$" + # the markup, carried. every markup file of this language is stored in the + # database, and source.digest is rebuilt from the stored bytes and held + # against what the .ssp recorded. + # + # source.digest is the sha256 over one line per markup file, + # "<SHA256> <basename>", sorted by that whole line and joined with newlines + # and no trailing newline. So the same value can be computed here from the + # files rows alone, and if it matches, the bytes in the database are the + # bytes the abstraction was built from: not merely present, but the right + # ones. That is the check worth having, since a file stored under the right + # name with the wrong contents would pass a mere count. + # LC_ALL=C: the lines are sorted by byte, as the writer sorts them, not by + # whatever collation the locale would prefer + src_lines=$(sqlite3 "$db" "select sha256 || ' ' || + substr(name, length('media/text/$lang/') + 1) + from files where role='source' and name like 'media/text/$lang/%';" \ + 2>/dev/null | LC_ALL=C sort) + # printf '%s' rather than echo: the joined lines carry no trailing newline + d_dig=$(printf '%s' "$src_lines" | sha256sum | cut -d' ' -f1 | tr 'a-z' 'A-Z') + s_dig=$(awk '/^@source \{$/ {b=1; next} /^\}$/ {b=0} b && /^ digest: / {print $2}' "$ssp") + n_src=$(sqlite3 "$db" "select count(*) from files + where role='source' and name like 'media/text/$lang/%';" 2>/dev/null || echo 0) + if [ "$n_src" = "0" ]; then + [ $doc_fail -eq 0 ] && echo "$base" + echo " MISSING markup carried for [$lang]"; doc_fail=1 + elif [ -n "$s_dig" ] && [ "$d_dig" != "$s_dig" ]; then + [ $doc_fail -eq 0 ] && echo "$base" + echo " MISMATCH source.digest rebuilt from the $n_src carried file(s):" + echo " .ssp $s_dig" + echo " db $d_dig" + doc_fail=1 + fi + # the source filename, and the version the two share sf=$(sqlite3 "$db" "select value from metadata where doc_id=$did and key='source.filename';" 2>/dev/null || echo '') diff --git a/test/test-abstraction-db.sh b/test/test-abstraction-db.sh index 78c27b9..fc26d51 100755 --- a/test/test-abstraction-db.sh +++ b/test/test-abstraction-db.sh @@ -263,6 +263,39 @@ for ssp in $(find "$OUT_DIR/pod" -name "*.ssp" 2>/dev/null | sort); do fi rm -f "$OUT_DIR/.sspmeta.$$" "$OUT_DIR/.dbmeta.$$" + # the markup, carried. every markup file of this language is stored in the + # database, and source.digest is rebuilt from the stored bytes and held + # against what the .ssp recorded. + # + # source.digest is the sha256 over one line per markup file, + # "<SHA256> <basename>", sorted by that whole line and joined with newlines + # and no trailing newline. So the same value can be computed here from the + # files rows alone, and if it matches, the bytes in the database are the + # bytes the abstraction was built from: not merely present, but the right + # ones. That is the check worth having, since a file stored under the right + # name with the wrong contents would pass a mere count. + # LC_ALL=C: the lines are sorted by byte, as the writer sorts them, not by + # whatever collation the locale would prefer + src_lines=$(sqlite3 "$db" "select sha256 || ' ' || + substr(name, length('media/text/$lang/') + 1) + from files where role='source' and name like 'media/text/$lang/%';" \ + 2>/dev/null | LC_ALL=C sort) + # printf '%s' rather than echo: the joined lines carry no trailing newline + d_dig=$(printf '%s' "$src_lines" | sha256sum | cut -d' ' -f1 | tr 'a-z' 'A-Z') + s_dig=$(awk '/^@source \{$/ {b=1; next} /^\}$/ {b=0} b && /^ digest: / {print $2}' "$ssp") + n_src=$(sqlite3 "$db" "select count(*) from files + where role='source' and name like 'media/text/$lang/%';" 2>/dev/null || echo 0) + if [ "$n_src" = "0" ]; then + [ $doc_fail -eq 0 ] && echo "$base" + echo " MISSING markup carried for [$lang]"; doc_fail=1 + elif [ -n "$s_dig" ] && [ "$d_dig" != "$s_dig" ]; then + [ $doc_fail -eq 0 ] && echo "$base" + echo " MISMATCH source.digest rebuilt from the $n_src carried file(s):" + echo " .ssp $s_dig" + echo " db $d_dig" + doc_fail=1 + fi + # the source filename, and the version the two share sf=$(sqlite3 "$db" "select value from metadata where doc_id=$did and key='source.filename';" 2>/dev/null || echo '') |
