aboutsummaryrefslogtreecommitdiffhomepage
diff options
context:
space:
mode:
authorRalph Amissah <ralph.amissah@gmail.com>2026-09-21 13:03:03 -0400
committerRalph Amissah <ralph.amissah@gmail.com>2026-09-22 15:11:02 -0400
commitea1162e5cf6182ad296b4775c7acfb51ac435842 (patch)
treef72e52aac4f2950e36d1c17d7cd974b4942de727
parentocda db: now carry markup source, conf & manifest (diff)
test: the carried markup, by rebuilding its digest
Check that they are the right files: source.digest is the sha256 over one line per markup file, "<SHA256> <basename>", sorted and newline joined, so the same value can be computed from the stored rows alone and held against what the .ssp recorded. A file stored under the right name with the wrong contents passes a count and fails this. Two details the shell gets wrong by default: the sort has to be LC_ALL=C, because the writer sorts by byte and a locale does not, and the lines are joined with no trailing newline, so printf and not echo. (assisted by Claude-Code)
-rw-r--r--org/tests_for_document_abstraction_shell_scripts.org33
-rwxr-xr-xtest/test-abstraction-db.sh33
2 files changed, 66 insertions, 0 deletions
diff --git a/org/tests_for_document_abstraction_shell_scripts.org b/org/tests_for_document_abstraction_shell_scripts.org
index b4d71f8..ab2dc35 100644
--- a/org/tests_for_document_abstraction_shell_scripts.org
+++ b/org/tests_for_document_abstraction_shell_scripts.org
@@ -682,6 +682,39 @@ for ssp in $(find "$OUT_DIR/pod" -name "*.ssp" 2>/dev/null | sort); do
fi
rm -f "$OUT_DIR/.sspmeta.$$" "$OUT_DIR/.dbmeta.$$"
+ # the markup, carried. every markup file of this language is stored in the
+ # database, and source.digest is rebuilt from the stored bytes and held
+ # against what the .ssp recorded.
+ #
+ # source.digest is the sha256 over one line per markup file,
+ # "<SHA256> <basename>", sorted by that whole line and joined with newlines
+ # and no trailing newline. So the same value can be computed here from the
+ # files rows alone, and if it matches, the bytes in the database are the
+ # bytes the abstraction was built from: not merely present, but the right
+ # ones. That is the check worth having, since a file stored under the right
+ # name with the wrong contents would pass a mere count.
+ # LC_ALL=C: the lines are sorted by byte, as the writer sorts them, not by
+ # whatever collation the locale would prefer
+ src_lines=$(sqlite3 "$db" "select sha256 || ' ' ||
+ substr(name, length('media/text/$lang/') + 1)
+ from files where role='source' and name like 'media/text/$lang/%';" \
+ 2>/dev/null | LC_ALL=C sort)
+ # printf '%s' rather than echo: the joined lines carry no trailing newline
+ d_dig=$(printf '%s' "$src_lines" | sha256sum | cut -d' ' -f1 | tr 'a-z' 'A-Z')
+ s_dig=$(awk '/^@source \{$/ {b=1; next} /^\}$/ {b=0} b && /^ digest: / {print $2}' "$ssp")
+ n_src=$(sqlite3 "$db" "select count(*) from files
+ where role='source' and name like 'media/text/$lang/%';" 2>/dev/null || echo 0)
+ if [ "$n_src" = "0" ]; then
+ [ $doc_fail -eq 0 ] && echo "$base"
+ echo " MISSING markup carried for [$lang]"; doc_fail=1
+ elif [ -n "$s_dig" ] && [ "$d_dig" != "$s_dig" ]; then
+ [ $doc_fail -eq 0 ] && echo "$base"
+ echo " MISMATCH source.digest rebuilt from the $n_src carried file(s):"
+ echo " .ssp $s_dig"
+ echo " db $d_dig"
+ doc_fail=1
+ fi
+
# the source filename, and the version the two share
sf=$(sqlite3 "$db" "select value from metadata
where doc_id=$did and key='source.filename';" 2>/dev/null || echo '')
diff --git a/test/test-abstraction-db.sh b/test/test-abstraction-db.sh
index 78c27b9..fc26d51 100755
--- a/test/test-abstraction-db.sh
+++ b/test/test-abstraction-db.sh
@@ -263,6 +263,39 @@ for ssp in $(find "$OUT_DIR/pod" -name "*.ssp" 2>/dev/null | sort); do
fi
rm -f "$OUT_DIR/.sspmeta.$$" "$OUT_DIR/.dbmeta.$$"
+ # the markup, carried. every markup file of this language is stored in the
+ # database, and source.digest is rebuilt from the stored bytes and held
+ # against what the .ssp recorded.
+ #
+ # source.digest is the sha256 over one line per markup file,
+ # "<SHA256> <basename>", sorted by that whole line and joined with newlines
+ # and no trailing newline. So the same value can be computed here from the
+ # files rows alone, and if it matches, the bytes in the database are the
+ # bytes the abstraction was built from: not merely present, but the right
+ # ones. That is the check worth having, since a file stored under the right
+ # name with the wrong contents would pass a mere count.
+ # LC_ALL=C: the lines are sorted by byte, as the writer sorts them, not by
+ # whatever collation the locale would prefer
+ src_lines=$(sqlite3 "$db" "select sha256 || ' ' ||
+ substr(name, length('media/text/$lang/') + 1)
+ from files where role='source' and name like 'media/text/$lang/%';" \
+ 2>/dev/null | LC_ALL=C sort)
+ # printf '%s' rather than echo: the joined lines carry no trailing newline
+ d_dig=$(printf '%s' "$src_lines" | sha256sum | cut -d' ' -f1 | tr 'a-z' 'A-Z')
+ s_dig=$(awk '/^@source \{$/ {b=1; next} /^\}$/ {b=0} b && /^ digest: / {print $2}' "$ssp")
+ n_src=$(sqlite3 "$db" "select count(*) from files
+ where role='source' and name like 'media/text/$lang/%';" 2>/dev/null || echo 0)
+ if [ "$n_src" = "0" ]; then
+ [ $doc_fail -eq 0 ] && echo "$base"
+ echo " MISSING markup carried for [$lang]"; doc_fail=1
+ elif [ -n "$s_dig" ] && [ "$d_dig" != "$s_dig" ]; then
+ [ $doc_fail -eq 0 ] && echo "$base"
+ echo " MISMATCH source.digest rebuilt from the $n_src carried file(s):"
+ echo " .ssp $s_dig"
+ echo " db $d_dig"
+ doc_fail=1
+ fi
+
# the source filename, and the version the two share
sf=$(sqlite3 "$db" "select value from metadata
where doc_id=$did and key='source.filename';" 2>/dev/null || echo '')