diff options
Diffstat (limited to 'src/sisudoc/outputs')
| -rw-r--r-- | src/sisudoc/outputs/io_out/epub3.d | 145 | ||||
| -rw-r--r-- | src/sisudoc/outputs/io_out/rgx.d | 4 | ||||
| -rw-r--r-- | src/sisudoc/outputs/io_out/sqlite_collection_db.d | 36 | ||||
| -rw-r--r-- | src/sisudoc/outputs/io_out/xmls.d | 130 |
4 files changed, 286 insertions, 29 deletions
diff --git a/src/sisudoc/outputs/io_out/epub3.d b/src/sisudoc/outputs/io_out/epub3.d index 420c536..8a3baaf 100644 --- a/src/sisudoc/outputs/io_out/epub3.d +++ b/src/sisudoc/outputs/io_out/epub3.d @@ -106,6 +106,140 @@ template outputEPub3() { default: return "image/" ~ _ext.toLower; } } + /+ ↓ the landmarks navigation. + This is what EPUB3 offers in place of the EPUB2 <guide> that was + dropped: not a list of headings, which is what the toc nav is, but a + handful of named places a reader offers as buttons - start reading + here, the contents, the index. It is a nav in the navigation document + rather than a section of the package, and it is hidden, being + machinery rather than content. + . + Every entry names a file that is in the manifest, found the same way + the manifest found it: the first heading of the section that opens a + segment of its own. + +/ + string _epub_landmark_file(D)(D doc, string _section) { + if (!(_section in doc.abstraction)) { return ""; } + string _first; + foreach (obj; doc.abstraction[_section]) { + if (obj.metainfo.is_a != "heading" + || obj.metainfo.heading_lev_markup > 4 + || obj.tags.segment_anchor_tag_epub.length == 0 + ) { continue; } + /+ ↓ a level 4 heading is where the text of the section actually + starts. Above it are the part wrappers, which hold no text, so + a landmark that lands on one has sent the reader to a title and + not to the thing named. Take the level 4 where there is one. +/ + if (obj.metainfo.heading_lev_markup == 4) { + return obj.tags.segment_anchor_tag_epub ~ ".xhtml"; + } + if (_first.empty) { _first = obj.tags.segment_anchor_tag_epub ~ ".xhtml"; } + } + return _first; + } + string _epub_landmarks(D)(D doc) { + string[3][] _wanted = [ + ["toc", "toc", "Table of Contents"], + ["body", "bodymatter", "Start of Content"], + ["bookindex", "index", "Index"], + ["glossary", "glossary", "Glossary"], + ["bibliography", "bibliography", "Bibliography"], + ]; + string _items; + foreach (_w; _wanted) { + string _file = _epub_landmark_file(doc, _w[0]); + if (_file.empty) { continue; } + _items ~= format( + " <li><a epub:type=\"%s\" href=\"%s\">%s</a></li>\n", + _w[1], _file, _w[2]); + } + if (_items.empty) { return ""; } + return " <nav epub:type=\"landmarks\" id=\"landmarks\" hidden=\"\">\n" + ~ " <h2>Guide</h2>\n" + ~ " <ol>\n" + ~ _items + ~ " </ol>\n" + ~ " </nav>\n"; + } + /+ ↓ the accessibility metadata EPUB Accessibility 1.1 asks for. + accessMode, accessibilityFeature and accessibilityHazard are required + of every EPUB by that specification (W3C Recommendation, 2024-10-17), + and accessModeSufficient and accessibilitySummary are recommended. + epubcheck does not test for them, which is why their absence never + showed in the error count; a reader looking for a book it can use + does test for them, and so does anyone distributing into the EU under + Directive 2019/882. + . + Every value here is derived from the document rather than asserted. + Nothing claims conformance to WCAG: that is a claim about an + evaluation that has not been made, and dcterms:conformsTo is + deliberately absent. + . + "schema" is a reserved prefix in EPUB 3, so it needs no declaring. + +/ + struct ST_epubA11y { + bool images; + bool some_image_lacks_alt; + bool bookindex; + } + ST_epubA11y _epub_a11y(D)(D doc) { + auto xhtml_format = outputXHTMLs(); + ST_epubA11y _a; + foreach (section; doc.matters.has.keys_seq.seg) { + if (section == "tail") { continue; } + if (!(section in doc.abstraction)) { continue; } + if (section == "bookindex") { _a.bookindex = true; } + foreach (obj; doc.abstraction[section]) { + foreach (m; obj.text.matchAll(rgx.inline_image)) { + _a.images = true; + if (xhtml_format._image_alt(m["post"].to!string).empty) { + _a.some_image_lacks_alt = true; + } + } + } + } + return _a; + } + string _epub_a11y_metadata(D)(D doc) { + auto _a = _epub_a11y(doc); + bool _alt_throughout = _a.images && !(_a.some_image_lacks_alt); + string _m; + string _meta(string _property, string _value) { + return format(" <meta property=\"%s\">%s</meta>\n", _property, _value); + } + _m ~= _meta("schema:accessMode", "textual"); + if (_a.images) { _m ~= _meta("schema:accessMode", "visual"); } + /+ ↓ a sufficient set is one a reader can get the whole publication + through. Text alone is sufficient where there are no images, and + where every image says what it is; where an image says nothing, + claiming it would be untrue. +/ + if (!(_a.images) || _alt_throughout) { + _m ~= _meta("schema:accessModeSufficient", "textual"); + } else { + _m ~= _meta("schema:accessModeSufficient", "textual,visual"); + } + _m ~= _meta("schema:accessibilityFeature", "structuralNavigation"); + _m ~= _meta("schema:accessibilityFeature", "tableOfContents"); + _m ~= _meta("schema:accessibilityFeature", "readingOrder"); + if (_a.bookindex) { _m ~= _meta("schema:accessibilityFeature", "index"); } + if (_alt_throughout) { _m ~= _meta("schema:accessibilityFeature", "alternativeText"); } + /+ ↓ nothing spine writes flashes, moves or makes a sound +/ + _m ~= _meta("schema:accessibilityHazard", "none"); + string _summary = + "Structured text with headings at every level, a navigation document," + ~ " and a reading order the publication declares." + ~ " Every substantive object carries a citation number that is the same" + ~ " in every format this document is published in."; + if (!(_a.images)) { + _summary ~= " There are no images."; + } else if (_alt_throughout) { + _summary ~= " Every image carries alternative text."; + } else { + _summary ~= " Some images carry no alternative text."; + } + _m ~= _meta("schema:accessibilitySummary", _summary); + return _m; + } /+ ↓ the publication identifier, as a real RFC 4122 UUID. Version 5, name based: the name is the document's own uid, so the identifier is stable across rebuilds and distinct per document, @@ -153,13 +287,13 @@ template outputEPub3() { <dc:identifier id="bookid">urn:uuid:%s</dc:identifier> <dc:title id="title">%s</dc:title> <meta refines="#title" property="title-type">main</meta> - %s <dc:creator id="aut">%s</dc:creator> + %s <dc:creator id="aut">%s</dc:creator> <meta refines="#aut" property="file-as">%s</meta> <dc:language>%s</dc:language> <dc:date id="published">%s</dc:date> <dc:rights>Copyright: %s</dc:rights> <meta property="dcterms:modified">%s</meta> - </metadata> + %s </metadata> <manifest> <item id="css" href="%s" media-type="text/css"/> <item id="nav" href="toc_nav.xhtml" media-type="application/xhtml+xml" properties="nav" /> @@ -181,6 +315,7 @@ template outputEPub3() { (doc.matters.conf_make_meta.meta.rights_copyright.empty) ? "" : xhtml_format.special_characters_plain(doc.matters.conf_make_meta.meta.rights_copyright), _epub_modified_utc(), + _epub_a11y_metadata(doc), (pth_epub3.fn_oebps_css).chompPrefix("OEBPS/"), ); content ~= parts["manifest_documents"]; @@ -241,7 +376,7 @@ template outputEPub3() { <header> <h1>Contents</h1> </header> - <nav epub:type="toc" id="toc"> + <nav epub:type="toc" id="toc" role="doc-toc"> ┃", (doc.matters.conf_make_meta.meta.title_full).special_characters_text, ); @@ -332,7 +467,9 @@ template outputEPub3() { if (n == 0) { _toc_nav_tail ~=" </nav> </section> - </body> + "; + _toc_nav_tail ~= _epub_landmarks(doc); + _toc_nav_tail ~=" </body> </html>\n"; } } diff --git a/src/sisudoc/outputs/io_out/rgx.d b/src/sisudoc/outputs/io_out/rgx.d index c915076..7ce87e0 100644 --- a/src/sisudoc/outputs/io_out/rgx.d +++ b/src/sisudoc/outputs/io_out/rgx.d @@ -100,6 +100,10 @@ static template spineRgxOut() { static inline_image = ctRegex!(`(?P<pre>┥)☼(?P<imginf>(?P<img>[a-zA-Z0-9._-]+?\.(?:jpg|gif|png)),w(?P<width>\d+)h(?P<height>\d+))\s*(?P<post>.*?┝┤.*?├)`, "mg"); static inline_image_without_dimensions = ctRegex!(`(?P<pre>┥)☼(?P<imginf>(?P<img>[a-zA-Z0-9._-]+?\.(?:jpg|gif|png)),w(?P<width>0)h(?P<height>0))\s*(?P<post>.*?┝┤.*?├)`, "mg"); static inline_image_info = ctRegex!(`☼?(?P<img>[a-zA-Z0-9._-]+?\.(?:jpg|gif|png)),w(?P<width>\d+)h(?P<height>\d+)`, "mg"); + /+ the image's alt text: what markup writes as { tux.png 64x80 "alt" }image, + which by this stage is a quoted string leading whatever follows the image + +/ + static inline_image_alt = ctRegex!(`^\s*[“"](?P<alt>[^”"]*)[”"]`, "m"); static inline_link_anchor = ctRegex!(`┃(?P<anchor>\S+?)┃`, "mg"); // TODO *~text_link_anchor static inline_link = ctRegex!(`┥(?P<text>.+?)┝┤(?P<link>#?(\S+?))├`, "mg"); static inline_link_empty = ctRegex!(`┥(?P<text>.+?)┝┤├`, "mg"); diff --git a/src/sisudoc/outputs/io_out/sqlite_collection_db.d b/src/sisudoc/outputs/io_out/sqlite_collection_db.d index f1556c4..2a7fcd7 100644 --- a/src/sisudoc/outputs/io_out/sqlite_collection_db.d +++ b/src/sisudoc/outputs/io_out/sqlite_collection_db.d @@ -487,14 +487,40 @@ template SQLiteFormatAndLoadObject() { _img_pth = "../../../image/"; } if (_txt.match(rgx.inline_image)) { - _txt = _txt.replaceAll( // TODO bug where image dimensions (w or h) not given & consequently set to 0; should not be used (calculate earlier, abstraction) - rgx.inline_image, - ("$1<img src=\"" - ~ _img_pth - ~ "$3\" width=\"$4\" height=\"$5\" /> $6")); + _txt = replaceAll!(m => + m["pre"].to!string + ~ _img_element(_img_pth ~ m["img"].to!string, m["width"].to!string, + m["height"].to!string, _image_alt(m["post"].to!string)) + ~ " " ~ _image_rest(m["post"].to!string) + )(_txt, rgx.inline_image); } return _txt; } + /+ ↓ the same treatment of an image as the html and epub writers give + it, because this is the html the cgi search renders: the markup's + description is the alt text, and dimensions that are not there are + not written as zeroes. + +/ + string _image_alt(string _post) { + if (auto m = _post.matchFirst(rgx.inline_image_alt)) { + return m["alt"].to!string; + } + return ""; + } + string _image_rest(string _post) { + if (auto m = _post.matchFirst(rgx.inline_image_alt)) { + return m.post.to!string; + } + return _post; + } + string _img_element(string _src, string _w, string _h, string _alt) { + string _dim = (_w == "0" && _h == "0") + ? "" + : " width=\"" ~ _w ~ "\" height=\"" ~ _h ~ "\""; + return "<img src=\"" ~ _src ~ "\"" + ~ " alt=\"" ~ html_special_characters(_alt) ~ "\"" + ~ _dim ~ " />"; + } string inline_links(M,O)( M doc_matters, const O obj, diff --git a/src/sisudoc/outputs/io_out/xmls.d b/src/sisudoc/outputs/io_out/xmls.d index c1e733e..4ce5434 100644 --- a/src/sisudoc/outputs/io_out/xmls.d +++ b/src/sisudoc/outputs/io_out/xmls.d @@ -78,13 +78,13 @@ template outputXHTMLs() { delimit_ ~= "\n<div class=\"doc_title\">\n" ; break; case "toc": - delimit_ ~= "\n<div class=\"doc_toc\">\n" ; + delimit_ ~= "\n<div class=\"doc_toc\"" ~ _section_role(section) ~ ">\n" ; break; case "bookindex": - delimit_ ~= "\n<div class=\"doc_bookindex\">\n" ; + delimit_ ~= "\n<div class=\"doc_bookindex\"" ~ _section_role(section) ~ ">\n" ; break; default: - delimit_ ~= "\n<div class=\"doc_" ~ section ~ "\">\n" ; + delimit_ ~= "\n<div class=\"doc_" ~ section ~ "\"" ~ _section_role(section) ~ ">\n" ; break; } if (previous_section.length > 0) { @@ -96,6 +96,40 @@ template outputXHTMLs() { // you also need to close the last div, introduce a footer? return delimit; } + /+ ↓ the DPUB-ARIA role for a document section. + spine has always known what these sections are; the output never + said so, and "doc_endnotes" as a class name means something to a + stylesheet and nothing to a reading system or a screen reader. + These roles are what carry that structure across. They are plain + ARIA, valid in html as well as in the epub, so both get them. + . + A section with no role of its own returns nothing rather than an + invented one. + +/ + /+ ↓ and the role of an individual object, where it has one of its own. + A gathered endnote is a note body, whatever the section around it + says; the reference to it carries doc-noteref at the other end. + . + Not doc-biblioentry for a bibliography entry: DPUB-ARIA 1.1 + deprecated it, and the entries sit inside doc-bibliography, which + is what a reader needs. + +/ + string _para_role(O)(O obj) { + switch (obj.metainfo.is_a) { + case "endnote": return " role=\"doc-footnote\""; + default: return ""; + } + } + string _section_role(string _section) { + switch (_section) { + case "toc": return " role=\"doc-toc\""; + case "endnotes": return " role=\"doc-endnotes\""; + case "bibliography": return " role=\"doc-bibliography\""; + case "glossary": return " role=\"doc-glossary\""; + case "bookindex": return " role=\"doc-index\""; + default: return ""; + } + } string special_characters_text(string _txt) { _txt = _txt .replaceAll(rgx_xhtml.ampersand, "&") // "&" @@ -216,12 +250,29 @@ template outputXHTMLs() { if (_xml_type == "epub" && _lev == 0) { return 1; } return (_lev > 6) ? 6 : _lev; } + /+ ↓ the anchors an object carries. + "name" on an <a> is obsolete; "id" is what a link target is. The + two are not interchangeable here, because these tags are not all + link targets. A gathered endnote carries exactly one, "note_<n>", + which is unique in the document and is what every reference to it + points at, so that one becomes an id. + . + A book index entry carries the same marker on every object the + entry names - "loc" appears 296 times in one document - and those + are markers of where a term occurs, not distinct targets. Making + them ids would put the same id in a document hundreds of times, + so they stay as they are until the book index gives them + something unique to be. + +/ string _xhtml_anchor_tags(O)(O obj) { string tags=""; + bool _is_a_unique_target = (obj.metainfo.is_a == "endnote"); if (obj.tags.anchor_tags.length > 0) { foreach (tag; obj.tags.anchor_tags) { if (!(tag.empty)) { - tags ~= "<a name=\"" ~ special_characters_text(tag) ~ "\"></a>"; /+ repeated markers, not unique targets: see below +/ + tags ~= (_is_a_unique_target) + ? "<a id=\"" ~ special_characters_text(tag) ~ "\"></a>" + : "<a name=\"" ~ special_characters_text(tag) ~ "\"></a>"; } } } @@ -552,17 +603,54 @@ string tail(M)(M doc_matters) { default: break; } if (_txt.match(rgx.inline_image)) { - _txt = _txt - .replaceAll(rgx.inline_image, - ("$1<img src=\"" - ~ _img_pth - ~ "$3\" width=\"$4\" height=\"$5\" /> $6")) - .replaceAll( - rgx.inline_link_empty, - ("$1")); + _txt = replaceAll!(m => + m["pre"].to!string + ~ _img_element(_img_pth ~ m["img"].to!string, m["width"].to!string, + m["height"].to!string, _image_alt(m["post"].to!string)) + ~ " " ~ _image_rest(m["post"].to!string) + )(_txt, rgx.inline_image); + _txt = _txt.replaceAll(rgx.inline_link_empty, ("$1")); } return _txt; } + /+ ↓ an image's alt text, which markup carries and every writer was + throwing away: { tux.png 64x80 "a better way" }image. The project's + own markup reference calls it alt text, and it was being left in the + visible flow as a quoted string beside the image instead, so no <img> + spine wrote had an alt at all. + +/ + string _image_alt(string _post) { + if (auto m = _post.matchFirst(rgx.inline_image_alt)) { + return m["alt"].to!string; + } + return ""; + } + /+ ↓ and what is left of the markup once the alt text is taken out of it +/ + string _image_rest(string _post) { + if (auto m = _post.matchFirst(rgx.inline_image_alt)) { + return m.post.to!string; + } + return _post; + } + /+ ↓ the <img> itself. + alt is required, and an image with nothing to say for itself takes + alt="", which is the way to say "decorative, skip me". No alt at all + is not that, it is an omission, and a reader has to announce the file + name instead. + . + width and height are dropped when there are none. They were written + as width="0" height="0", which is a perfectly valid pair of HTML + integers meaning zero by zero, so the image did not appear at all; + max-width in the stylesheet caps a width and cannot raise one. + +/ + string _img_element(string _src, string _w, string _h, string _alt) { + string _dim = (_w == "0" && _h == "0") + ? "" + : " width=\"" ~ _w ~ "\" height=\"" ~ _h ~ "\""; + return "<img src=\"" ~ _src ~ "\"" + ~ " alt=\"" ~ special_characters_plain(_alt) ~ "\"" + ~ _dim ~ " />"; + } string inline_links(O,M)( string _txt, const O obj, @@ -665,14 +753,14 @@ string tail(M)(M doc_matters) { _txt = font_face(_txt); _txt = _txt.replaceAll( rgx.inline_notes_al_regular_number_note, - ("<a href=\"#note_$1\"><span class=\"noteref\" id=\"noteref_$1\"> <sup>$1</sup> </span></a>") + ("<a href=\"#note_$1\" role=\"doc-noteref\"><span class=\"noteref\" id=\"noteref_$1\"> <sup>$1</sup> </span></a>") ); } if (obj.has.inline_notes_star) { _txt = font_face(_txt); _txt = _txt.replaceAll( rgx.inline_notes_al_special_char_note, - ("<a href=\"#note_$1\"><span class=\"noteref\" id=\"noteref_$1\"> <sup>$1</sup> </span></a>") + ("<a href=\"#note_$1\" role=\"doc-noteref\"><span class=\"noteref\" id=\"noteref_$1\"> <sup>$1</sup> </span></a>") ); } debug(markup_endnotes) { @@ -699,7 +787,7 @@ string tail(M)(M doc_matters) { foreach(m; _txt.matchAll(rgx.inline_notes_al_special_char_note)) { _endnotes ~= format( "%s%s%s%s\n %s%s%s%s%s %s\n%s", - "<p class=\"endnote\">", + "<p class=\"endnote\" role=\"doc-footnote\">", "<a href=\"#noteref_", m.captures[1], "\">", @@ -714,7 +802,7 @@ string tail(M)(M doc_matters) { } _txt = _txt.replaceAll( rgx.inline_notes_al_special_char_note, - ("<a href=\"#note_$1\"><span class=\"noteref\" id=\"noteref_$1\"> <sup>$1</sup> </span></a>") + ("<a href=\"#note_$1\" role=\"doc-noteref\"><span class=\"noteref\" id=\"noteref_$1\"> <sup>$1</sup> </span></a>") ); } if (obj.has.inline_notes_reg) { @@ -723,7 +811,7 @@ string tail(M)(M doc_matters) { foreach(m; _txt.matchAll(rgx.inline_notes_al_regular_number_note)) { _endnotes ~= format( "%s%s%s%s\n %s%s%s%s%s %s\n%s", - "<p class=\"endnote\">", + "<p class=\"endnote\" role=\"doc-footnote\">", "<a href=\"#noteref_", m.captures[1], "\">", @@ -738,7 +826,7 @@ string tail(M)(M doc_matters) { } _txt = _txt.replaceAll( rgx.inline_notes_al_regular_number_note, - ("<a href=\"#note_$1\"><span class=\"noteref\" id=\"noteref_$1\"> <sup>$1</sup> </span></a>") + ("<a href=\"#note_$1\" role=\"doc-noteref\"><span class=\"noteref\" id=\"noteref_$1\"> <sup>$1</sup> </span></a>") ); } else if (_txt.match(rgx.inline_notes_al_regular_number_note)) { debug(markup) { @@ -1012,7 +1100,7 @@ string tail(M)(M doc_matters) { if (!(obj.metainfo.identifier.empty)) { o = format(q"┃ <div class="substance"> <label class="ocn"><a href="#%s" class="lnkocn">%s</a></label> - <p class="%s" data-indent="h%si%s" id="%s">%s + <p class="%s" data-indent="h%si%s" id="%s"%s>%s %s </p> </div>┃", @@ -1026,18 +1114,20 @@ string tail(M)(M doc_matters) { obj.attrib.indent_hang, obj.attrib.indent_base, obj.metainfo.identifier, + _para_role(obj), tags, _txt ); } else { o = format(q"┃ <div class="substance"> - <p class="%s" data-indent="h%si%s">%s + <p class="%s" data-indent="h%si%s"%s>%s %s </p> </div>┃", obj.metainfo.is_a, obj.attrib.indent_hang, obj.attrib.indent_base, + _para_role(obj), tags, _txt ); |
