aboutsummaryrefslogtreecommitdiffhomepage
path: root/src/sisudoc/outputs
diff options
context:
space:
mode:
Diffstat (limited to 'src/sisudoc/outputs')
-rw-r--r--src/sisudoc/outputs/io_out/epub3.d145
-rw-r--r--src/sisudoc/outputs/io_out/rgx.d4
-rw-r--r--src/sisudoc/outputs/io_out/sqlite_collection_db.d36
-rw-r--r--src/sisudoc/outputs/io_out/xmls.d130
4 files changed, 286 insertions, 29 deletions
diff --git a/src/sisudoc/outputs/io_out/epub3.d b/src/sisudoc/outputs/io_out/epub3.d
index 420c536..8a3baaf 100644
--- a/src/sisudoc/outputs/io_out/epub3.d
+++ b/src/sisudoc/outputs/io_out/epub3.d
@@ -106,6 +106,140 @@ template outputEPub3() {
default: return "image/" ~ _ext.toLower;
}
}
+ /+ ↓ the landmarks navigation.
+ This is what EPUB3 offers in place of the EPUB2 <guide> that was
+ dropped: not a list of headings, which is what the toc nav is, but a
+ handful of named places a reader offers as buttons - start reading
+ here, the contents, the index. It is a nav in the navigation document
+ rather than a section of the package, and it is hidden, being
+ machinery rather than content.
+ .
+ Every entry names a file that is in the manifest, found the same way
+ the manifest found it: the first heading of the section that opens a
+ segment of its own.
+ +/
+ string _epub_landmark_file(D)(D doc, string _section) {
+ if (!(_section in doc.abstraction)) { return ""; }
+ string _first;
+ foreach (obj; doc.abstraction[_section]) {
+ if (obj.metainfo.is_a != "heading"
+ || obj.metainfo.heading_lev_markup > 4
+ || obj.tags.segment_anchor_tag_epub.length == 0
+ ) { continue; }
+ /+ ↓ a level 4 heading is where the text of the section actually
+ starts. Above it are the part wrappers, which hold no text, so
+ a landmark that lands on one has sent the reader to a title and
+ not to the thing named. Take the level 4 where there is one. +/
+ if (obj.metainfo.heading_lev_markup == 4) {
+ return obj.tags.segment_anchor_tag_epub ~ ".xhtml";
+ }
+ if (_first.empty) { _first = obj.tags.segment_anchor_tag_epub ~ ".xhtml"; }
+ }
+ return _first;
+ }
+ string _epub_landmarks(D)(D doc) {
+ string[3][] _wanted = [
+ ["toc", "toc", "Table of Contents"],
+ ["body", "bodymatter", "Start of Content"],
+ ["bookindex", "index", "Index"],
+ ["glossary", "glossary", "Glossary"],
+ ["bibliography", "bibliography", "Bibliography"],
+ ];
+ string _items;
+ foreach (_w; _wanted) {
+ string _file = _epub_landmark_file(doc, _w[0]);
+ if (_file.empty) { continue; }
+ _items ~= format(
+ " <li><a epub:type=\"%s\" href=\"%s\">%s</a></li>\n",
+ _w[1], _file, _w[2]);
+ }
+ if (_items.empty) { return ""; }
+ return " <nav epub:type=\"landmarks\" id=\"landmarks\" hidden=\"\">\n"
+ ~ " <h2>Guide</h2>\n"
+ ~ " <ol>\n"
+ ~ _items
+ ~ " </ol>\n"
+ ~ " </nav>\n";
+ }
+ /+ ↓ the accessibility metadata EPUB Accessibility 1.1 asks for.
+ accessMode, accessibilityFeature and accessibilityHazard are required
+ of every EPUB by that specification (W3C Recommendation, 2024-10-17),
+ and accessModeSufficient and accessibilitySummary are recommended.
+ epubcheck does not test for them, which is why their absence never
+ showed in the error count; a reader looking for a book it can use
+ does test for them, and so does anyone distributing into the EU under
+ Directive 2019/882.
+ .
+ Every value here is derived from the document rather than asserted.
+ Nothing claims conformance to WCAG: that is a claim about an
+ evaluation that has not been made, and dcterms:conformsTo is
+ deliberately absent.
+ .
+ "schema" is a reserved prefix in EPUB 3, so it needs no declaring.
+ +/
+ struct ST_epubA11y {
+ bool images;
+ bool some_image_lacks_alt;
+ bool bookindex;
+ }
+ ST_epubA11y _epub_a11y(D)(D doc) {
+ auto xhtml_format = outputXHTMLs();
+ ST_epubA11y _a;
+ foreach (section; doc.matters.has.keys_seq.seg) {
+ if (section == "tail") { continue; }
+ if (!(section in doc.abstraction)) { continue; }
+ if (section == "bookindex") { _a.bookindex = true; }
+ foreach (obj; doc.abstraction[section]) {
+ foreach (m; obj.text.matchAll(rgx.inline_image)) {
+ _a.images = true;
+ if (xhtml_format._image_alt(m["post"].to!string).empty) {
+ _a.some_image_lacks_alt = true;
+ }
+ }
+ }
+ }
+ return _a;
+ }
+ string _epub_a11y_metadata(D)(D doc) {
+ auto _a = _epub_a11y(doc);
+ bool _alt_throughout = _a.images && !(_a.some_image_lacks_alt);
+ string _m;
+ string _meta(string _property, string _value) {
+ return format(" <meta property=\"%s\">%s</meta>\n", _property, _value);
+ }
+ _m ~= _meta("schema:accessMode", "textual");
+ if (_a.images) { _m ~= _meta("schema:accessMode", "visual"); }
+ /+ ↓ a sufficient set is one a reader can get the whole publication
+ through. Text alone is sufficient where there are no images, and
+ where every image says what it is; where an image says nothing,
+ claiming it would be untrue. +/
+ if (!(_a.images) || _alt_throughout) {
+ _m ~= _meta("schema:accessModeSufficient", "textual");
+ } else {
+ _m ~= _meta("schema:accessModeSufficient", "textual,visual");
+ }
+ _m ~= _meta("schema:accessibilityFeature", "structuralNavigation");
+ _m ~= _meta("schema:accessibilityFeature", "tableOfContents");
+ _m ~= _meta("schema:accessibilityFeature", "readingOrder");
+ if (_a.bookindex) { _m ~= _meta("schema:accessibilityFeature", "index"); }
+ if (_alt_throughout) { _m ~= _meta("schema:accessibilityFeature", "alternativeText"); }
+ /+ ↓ nothing spine writes flashes, moves or makes a sound +/
+ _m ~= _meta("schema:accessibilityHazard", "none");
+ string _summary =
+ "Structured text with headings at every level, a navigation document,"
+ ~ " and a reading order the publication declares."
+ ~ " Every substantive object carries a citation number that is the same"
+ ~ " in every format this document is published in.";
+ if (!(_a.images)) {
+ _summary ~= " There are no images.";
+ } else if (_alt_throughout) {
+ _summary ~= " Every image carries alternative text.";
+ } else {
+ _summary ~= " Some images carry no alternative text.";
+ }
+ _m ~= _meta("schema:accessibilitySummary", _summary);
+ return _m;
+ }
/+ ↓ the publication identifier, as a real RFC 4122 UUID.
Version 5, name based: the name is the document's own uid, so the
identifier is stable across rebuilds and distinct per document,
@@ -153,13 +287,13 @@ template outputEPub3() {
<dc:identifier id="bookid">urn:uuid:%s</dc:identifier>
<dc:title id="title">%s</dc:title>
<meta refines="#title" property="title-type">main</meta>
- %s <dc:creator id="aut">%s</dc:creator>
+ %s <dc:creator id="aut">%s</dc:creator>
<meta refines="#aut" property="file-as">%s</meta>
<dc:language>%s</dc:language>
<dc:date id="published">%s</dc:date>
<dc:rights>Copyright: %s</dc:rights>
<meta property="dcterms:modified">%s</meta>
- </metadata>
+ %s </metadata>
<manifest>
<item id="css" href="%s" media-type="text/css"/>
<item id="nav" href="toc_nav.xhtml" media-type="application/xhtml+xml" properties="nav" />
@@ -181,6 +315,7 @@ template outputEPub3() {
(doc.matters.conf_make_meta.meta.rights_copyright.empty)
? "" : xhtml_format.special_characters_plain(doc.matters.conf_make_meta.meta.rights_copyright),
_epub_modified_utc(),
+ _epub_a11y_metadata(doc),
(pth_epub3.fn_oebps_css).chompPrefix("OEBPS/"),
);
content ~= parts["manifest_documents"];
@@ -241,7 +376,7 @@ template outputEPub3() {
<header>
<h1>Contents</h1>
</header>
- <nav epub:type="toc" id="toc">
+ <nav epub:type="toc" id="toc" role="doc-toc">
┃",
(doc.matters.conf_make_meta.meta.title_full).special_characters_text,
);
@@ -332,7 +467,9 @@ template outputEPub3() {
if (n == 0) {
_toc_nav_tail ~=" </nav>
</section>
- </body>
+ ";
+ _toc_nav_tail ~= _epub_landmarks(doc);
+ _toc_nav_tail ~=" </body>
</html>\n";
}
}
diff --git a/src/sisudoc/outputs/io_out/rgx.d b/src/sisudoc/outputs/io_out/rgx.d
index c915076..7ce87e0 100644
--- a/src/sisudoc/outputs/io_out/rgx.d
+++ b/src/sisudoc/outputs/io_out/rgx.d
@@ -100,6 +100,10 @@ static template spineRgxOut() {
static inline_image = ctRegex!(`(?P<pre>┥)☼(?P<imginf>(?P<img>[a-zA-Z0-9._-]+?\.(?:jpg|gif|png)),w(?P<width>\d+)h(?P<height>\d+))\s*(?P<post>.*?┝┤.*?├)`, "mg");
static inline_image_without_dimensions = ctRegex!(`(?P<pre>┥)☼(?P<imginf>(?P<img>[a-zA-Z0-9._-]+?\.(?:jpg|gif|png)),w(?P<width>0)h(?P<height>0))\s*(?P<post>.*?┝┤.*?├)`, "mg");
static inline_image_info = ctRegex!(`☼?(?P<img>[a-zA-Z0-9._-]+?\.(?:jpg|gif|png)),w(?P<width>\d+)h(?P<height>\d+)`, "mg");
+ /+ the image's alt text: what markup writes as { tux.png 64x80 "alt" }image,
+ which by this stage is a quoted string leading whatever follows the image
+ +/
+ static inline_image_alt = ctRegex!(`^\s*[“"](?P<alt>[^”"]*)[”"]`, "m");
static inline_link_anchor = ctRegex!(`┃(?P<anchor>\S+?)┃`, "mg"); // TODO *~text_link_anchor
static inline_link = ctRegex!(`┥(?P<text>.+?)┝┤(?P<link>#?(\S+?))├`, "mg");
static inline_link_empty = ctRegex!(`┥(?P<text>.+?)┝┤├`, "mg");
diff --git a/src/sisudoc/outputs/io_out/sqlite_collection_db.d b/src/sisudoc/outputs/io_out/sqlite_collection_db.d
index f1556c4..2a7fcd7 100644
--- a/src/sisudoc/outputs/io_out/sqlite_collection_db.d
+++ b/src/sisudoc/outputs/io_out/sqlite_collection_db.d
@@ -487,14 +487,40 @@ template SQLiteFormatAndLoadObject() {
_img_pth = "../../../image/";
}
if (_txt.match(rgx.inline_image)) {
- _txt = _txt.replaceAll( // TODO bug where image dimensions (w or h) not given & consequently set to 0; should not be used (calculate earlier, abstraction)
- rgx.inline_image,
- ("$1<img src=\""
- ~ _img_pth
- ~ "$3\" width=\"$4\" height=\"$5\" /> $6"));
+ _txt = replaceAll!(m =>
+ m["pre"].to!string
+ ~ _img_element(_img_pth ~ m["img"].to!string, m["width"].to!string,
+ m["height"].to!string, _image_alt(m["post"].to!string))
+ ~ " " ~ _image_rest(m["post"].to!string)
+ )(_txt, rgx.inline_image);
}
return _txt;
}
+ /+ ↓ the same treatment of an image as the html and epub writers give
+ it, because this is the html the cgi search renders: the markup's
+ description is the alt text, and dimensions that are not there are
+ not written as zeroes.
+ +/
+ string _image_alt(string _post) {
+ if (auto m = _post.matchFirst(rgx.inline_image_alt)) {
+ return m["alt"].to!string;
+ }
+ return "";
+ }
+ string _image_rest(string _post) {
+ if (auto m = _post.matchFirst(rgx.inline_image_alt)) {
+ return m.post.to!string;
+ }
+ return _post;
+ }
+ string _img_element(string _src, string _w, string _h, string _alt) {
+ string _dim = (_w == "0" && _h == "0")
+ ? ""
+ : " width=\"" ~ _w ~ "\" height=\"" ~ _h ~ "\"";
+ return "<img src=\"" ~ _src ~ "\""
+ ~ " alt=\"" ~ html_special_characters(_alt) ~ "\""
+ ~ _dim ~ " />";
+ }
string inline_links(M,O)(
M doc_matters,
const O obj,
diff --git a/src/sisudoc/outputs/io_out/xmls.d b/src/sisudoc/outputs/io_out/xmls.d
index c1e733e..4ce5434 100644
--- a/src/sisudoc/outputs/io_out/xmls.d
+++ b/src/sisudoc/outputs/io_out/xmls.d
@@ -78,13 +78,13 @@ template outputXHTMLs() {
delimit_ ~= "\n<div class=\"doc_title\">\n" ;
break;
case "toc":
- delimit_ ~= "\n<div class=\"doc_toc\">\n" ;
+ delimit_ ~= "\n<div class=\"doc_toc\"" ~ _section_role(section) ~ ">\n" ;
break;
case "bookindex":
- delimit_ ~= "\n<div class=\"doc_bookindex\">\n" ;
+ delimit_ ~= "\n<div class=\"doc_bookindex\"" ~ _section_role(section) ~ ">\n" ;
break;
default:
- delimit_ ~= "\n<div class=\"doc_" ~ section ~ "\">\n" ;
+ delimit_ ~= "\n<div class=\"doc_" ~ section ~ "\"" ~ _section_role(section) ~ ">\n" ;
break;
}
if (previous_section.length > 0) {
@@ -96,6 +96,40 @@ template outputXHTMLs() {
// you also need to close the last div, introduce a footer?
return delimit;
}
+ /+ ↓ the DPUB-ARIA role for a document section.
+ spine has always known what these sections are; the output never
+ said so, and "doc_endnotes" as a class name means something to a
+ stylesheet and nothing to a reading system or a screen reader.
+ These roles are what carry that structure across. They are plain
+ ARIA, valid in html as well as in the epub, so both get them.
+ .
+ A section with no role of its own returns nothing rather than an
+ invented one.
+ +/
+ /+ ↓ and the role of an individual object, where it has one of its own.
+ A gathered endnote is a note body, whatever the section around it
+ says; the reference to it carries doc-noteref at the other end.
+ .
+ Not doc-biblioentry for a bibliography entry: DPUB-ARIA 1.1
+ deprecated it, and the entries sit inside doc-bibliography, which
+ is what a reader needs.
+ +/
+ string _para_role(O)(O obj) {
+ switch (obj.metainfo.is_a) {
+ case "endnote": return " role=\"doc-footnote\"";
+ default: return "";
+ }
+ }
+ string _section_role(string _section) {
+ switch (_section) {
+ case "toc": return " role=\"doc-toc\"";
+ case "endnotes": return " role=\"doc-endnotes\"";
+ case "bibliography": return " role=\"doc-bibliography\"";
+ case "glossary": return " role=\"doc-glossary\"";
+ case "bookindex": return " role=\"doc-index\"";
+ default: return "";
+ }
+ }
string special_characters_text(string _txt) {
_txt = _txt
.replaceAll(rgx_xhtml.ampersand, "&amp;") // "&#38;"
@@ -216,12 +250,29 @@ template outputXHTMLs() {
if (_xml_type == "epub" && _lev == 0) { return 1; }
return (_lev > 6) ? 6 : _lev;
}
+ /+ ↓ the anchors an object carries.
+ "name" on an <a> is obsolete; "id" is what a link target is. The
+ two are not interchangeable here, because these tags are not all
+ link targets. A gathered endnote carries exactly one, "note_<n>",
+ which is unique in the document and is what every reference to it
+ points at, so that one becomes an id.
+ .
+ A book index entry carries the same marker on every object the
+ entry names - "loc" appears 296 times in one document - and those
+ are markers of where a term occurs, not distinct targets. Making
+ them ids would put the same id in a document hundreds of times,
+ so they stay as they are until the book index gives them
+ something unique to be.
+ +/
string _xhtml_anchor_tags(O)(O obj) {
string tags="";
+ bool _is_a_unique_target = (obj.metainfo.is_a == "endnote");
if (obj.tags.anchor_tags.length > 0) {
foreach (tag; obj.tags.anchor_tags) {
if (!(tag.empty)) {
- tags ~= "<a name=\"" ~ special_characters_text(tag) ~ "\"></a>"; /+ repeated markers, not unique targets: see below +/
+ tags ~= (_is_a_unique_target)
+ ? "<a id=\"" ~ special_characters_text(tag) ~ "\"></a>"
+ : "<a name=\"" ~ special_characters_text(tag) ~ "\"></a>";
}
}
}
@@ -552,17 +603,54 @@ string tail(M)(M doc_matters) {
default: break;
}
if (_txt.match(rgx.inline_image)) {
- _txt = _txt
- .replaceAll(rgx.inline_image,
- ("$1<img src=\""
- ~ _img_pth
- ~ "$3\" width=\"$4\" height=\"$5\" /> $6"))
- .replaceAll(
- rgx.inline_link_empty,
- ("$1"));
+ _txt = replaceAll!(m =>
+ m["pre"].to!string
+ ~ _img_element(_img_pth ~ m["img"].to!string, m["width"].to!string,
+ m["height"].to!string, _image_alt(m["post"].to!string))
+ ~ " " ~ _image_rest(m["post"].to!string)
+ )(_txt, rgx.inline_image);
+ _txt = _txt.replaceAll(rgx.inline_link_empty, ("$1"));
}
return _txt;
}
+ /+ ↓ an image's alt text, which markup carries and every writer was
+ throwing away: { tux.png 64x80 "a better way" }image. The project's
+ own markup reference calls it alt text, and it was being left in the
+ visible flow as a quoted string beside the image instead, so no <img>
+ spine wrote had an alt at all.
+ +/
+ string _image_alt(string _post) {
+ if (auto m = _post.matchFirst(rgx.inline_image_alt)) {
+ return m["alt"].to!string;
+ }
+ return "";
+ }
+ /+ ↓ and what is left of the markup once the alt text is taken out of it +/
+ string _image_rest(string _post) {
+ if (auto m = _post.matchFirst(rgx.inline_image_alt)) {
+ return m.post.to!string;
+ }
+ return _post;
+ }
+ /+ ↓ the <img> itself.
+ alt is required, and an image with nothing to say for itself takes
+ alt="", which is the way to say "decorative, skip me". No alt at all
+ is not that, it is an omission, and a reader has to announce the file
+ name instead.
+ .
+ width and height are dropped when there are none. They were written
+ as width="0" height="0", which is a perfectly valid pair of HTML
+ integers meaning zero by zero, so the image did not appear at all;
+ max-width in the stylesheet caps a width and cannot raise one.
+ +/
+ string _img_element(string _src, string _w, string _h, string _alt) {
+ string _dim = (_w == "0" && _h == "0")
+ ? ""
+ : " width=\"" ~ _w ~ "\" height=\"" ~ _h ~ "\"";
+ return "<img src=\"" ~ _src ~ "\""
+ ~ " alt=\"" ~ special_characters_plain(_alt) ~ "\""
+ ~ _dim ~ " />";
+ }
string inline_links(O,M)(
string _txt,
const O obj,
@@ -665,14 +753,14 @@ string tail(M)(M doc_matters) {
_txt = font_face(_txt);
_txt = _txt.replaceAll(
rgx.inline_notes_al_regular_number_note,
- ("<a href=\"#note_$1\"><span class=\"noteref\" id=\"noteref_$1\">&#160;<sup>$1</sup> </span></a>")
+ ("<a href=\"#note_$1\" role=\"doc-noteref\"><span class=\"noteref\" id=\"noteref_$1\">&#160;<sup>$1</sup> </span></a>")
);
}
if (obj.has.inline_notes_star) {
_txt = font_face(_txt);
_txt = _txt.replaceAll(
rgx.inline_notes_al_special_char_note,
- ("<a href=\"#note_$1\"><span class=\"noteref\" id=\"noteref_$1\">&#160;<sup>$1</sup> </span></a>")
+ ("<a href=\"#note_$1\" role=\"doc-noteref\"><span class=\"noteref\" id=\"noteref_$1\">&#160;<sup>$1</sup> </span></a>")
);
}
debug(markup_endnotes) {
@@ -699,7 +787,7 @@ string tail(M)(M doc_matters) {
foreach(m; _txt.matchAll(rgx.inline_notes_al_special_char_note)) {
_endnotes ~= format(
"%s%s%s%s\n %s%s%s%s%s %s\n%s",
- "<p class=\"endnote\">",
+ "<p class=\"endnote\" role=\"doc-footnote\">",
"<a href=\"#noteref_",
m.captures[1],
"\">",
@@ -714,7 +802,7 @@ string tail(M)(M doc_matters) {
}
_txt = _txt.replaceAll(
rgx.inline_notes_al_special_char_note,
- ("<a href=\"#note_$1\"><span class=\"noteref\" id=\"noteref_$1\">&#160;<sup>$1</sup> </span></a>")
+ ("<a href=\"#note_$1\" role=\"doc-noteref\"><span class=\"noteref\" id=\"noteref_$1\">&#160;<sup>$1</sup> </span></a>")
);
}
if (obj.has.inline_notes_reg) {
@@ -723,7 +811,7 @@ string tail(M)(M doc_matters) {
foreach(m; _txt.matchAll(rgx.inline_notes_al_regular_number_note)) {
_endnotes ~= format(
"%s%s%s%s\n %s%s%s%s%s %s\n%s",
- "<p class=\"endnote\">",
+ "<p class=\"endnote\" role=\"doc-footnote\">",
"<a href=\"#noteref_",
m.captures[1],
"\">",
@@ -738,7 +826,7 @@ string tail(M)(M doc_matters) {
}
_txt = _txt.replaceAll(
rgx.inline_notes_al_regular_number_note,
- ("<a href=\"#note_$1\"><span class=\"noteref\" id=\"noteref_$1\">&#160;<sup>$1</sup> </span></a>")
+ ("<a href=\"#note_$1\" role=\"doc-noteref\"><span class=\"noteref\" id=\"noteref_$1\">&#160;<sup>$1</sup> </span></a>")
);
} else if (_txt.match(rgx.inline_notes_al_regular_number_note)) {
debug(markup) {
@@ -1012,7 +1100,7 @@ string tail(M)(M doc_matters) {
if (!(obj.metainfo.identifier.empty)) {
o = format(q"┃ <div class="substance">
<label class="ocn"><a href="#%s" class="lnkocn">%s</a></label>
- <p class="%s" data-indent="h%si%s" id="%s">%s
+ <p class="%s" data-indent="h%si%s" id="%s"%s>%s
%s
</p>
</div>┃",
@@ -1026,18 +1114,20 @@ string tail(M)(M doc_matters) {
obj.attrib.indent_hang,
obj.attrib.indent_base,
obj.metainfo.identifier,
+ _para_role(obj),
tags,
_txt
);
} else {
o = format(q"┃ <div class="substance">
- <p class="%s" data-indent="h%si%s">%s
+ <p class="%s" data-indent="h%si%s"%s>%s
%s
</p>
</div>┃",
obj.metainfo.is_a,
obj.attrib.indent_hang,
obj.attrib.indent_base,
+ _para_role(obj),
tags,
_txt
);