Files
bookhoard/internal/services/media_scanner_opf.go
T
John O'Keefe 72d167005f fix(scanner): resolve EPUB covers via structured OPF parsing, Calibre chain
The cover lookup scraped the OPF with attribute-order-sensitive regexes.
Real books serialize attributes in any order - Grand Central's '3 Days to
Live' puts href before id on manifest items and content before name on
the cover meta - so all three regex paths missed and the book fell
through to filename guessing, extracting no cover at all. Attribute
order is meaningless in XML; the regexes were never safe.

Replace them with a structured parse (encoding/xml, namespace and
attribute-order agnostic; see the new media_scanner_opf.go) and follow
Calibre's read_raster_cover resolution order:

1. manifest item with properties=cover-image (non-(X)HTML media only)
2. <meta name=cover> resolved through the manifest, same media guard
3. first spine item that is itself a raster image (store manga)
4. NEW cover-page fallback: books declaring no raster cover at all -
   the classic EPUB2/Adobe cover.xhtml wrapper - are mined for
   <img src> / SVG <image xlink:href> references (Calibre renders the
   page with Qt; extracting the referenced image covers the practical
   cases without a rendering engine)
5. existing zip filename guessing stays as the last resort, and the old
   regex chain survives as findCoverInOPFLegacy for OPFs too malformed
   for a real XML parse.

Hrefs are now URL-decoded and posix-normalized against the OPF's own
path (path.Join semantics), so '../art/cover.jpg' from a nested cover
page and %20-encoded names resolve correctly.

Tests: attribute-order chaos modeled on the failing Patterson book,
SVG-wrapped cover pages via guide references, image-first spines, and
path resolution edge cases. Verified live against the real
'3 Days to Live' EPUB, which previously produced no cover.
2026-09-12 23:45:01 -04:00

200 lines
6.1 KiB
Go

package services
import (
"bytes"
"encoding/xml"
"strings"
)
// Calibre-modeled OPF parsing. The scanner previously scraped OPF content
// with attribute-order-sensitive regexes; real books serialize attributes in
// any order (e.g. Pragmatic/Pattinson EPUBs put id before properties and
// content before name), which silently defeated cover detection. Everything
// here is parsed with encoding/xml so attribute order and namespace prefix
// choices are irrelevant.
type opfDCValue struct {
ID string `xml:"id,attr"`
Value string `xml:",chardata"`
}
type opfIdentifier struct {
Scheme string `xml:"http://www.idpf.org/2007/opf scheme,attr"`
Value string `xml:",chardata"`
}
type opfMeta struct {
ID string `xml:"id,attr"`
Name string `xml:"name,attr"`
Content string `xml:"content,attr"`
Property string `xml:"property,attr"`
Refines string `xml:"refines,attr"`
Value string `xml:",chardata"`
}
type opfItem struct {
ID string `xml:"id,attr"`
Href string `xml:"href,attr"`
MediaType string `xml:"media-type,attr"`
Properties string `xml:"properties,attr"`
}
// opfDocument is a structured view of an OPF package document.
type opfDocument struct {
Metadata struct {
Titles []opfDCValue `xml:"http://purl.org/dc/elements/1.1/ title"`
Creators []string `xml:"http://purl.org/dc/elements/1.1/ creator"`
Subjects []string `xml:"http://purl.org/dc/elements/1.1/ subject"`
Descriptions []string `xml:"http://purl.org/dc/elements/1.1/ description"`
Publishers []string `xml:"http://purl.org/dc/elements/1.1/ publisher"`
Dates []string `xml:"http://purl.org/dc/elements/1.1/ date"`
Languages []string `xml:"http://purl.org/dc/elements/1.1/ language"`
Identifiers []opfIdentifier `xml:"http://purl.org/dc/elements/1.1/ identifier"`
Contributors []string `xml:"http://purl.org/dc/elements/1.1/ contributor"`
Metas []opfMeta `xml:"meta"`
} `xml:"metadata"`
Manifest struct {
Items []opfItem `xml:"item"`
} `xml:"manifest"`
Spine struct {
Itemrefs []struct {
IDRef string `xml:"idref,attr"`
} `xml:"itemref"`
} `xml:"spine"`
Guide struct {
References []struct {
Type string `xml:"type,attr"`
Href string `xml:"href,attr"`
} `xml:"reference"`
} `xml:"guide"`
}
func parseOPFXML(content []byte) (*opfDocument, error) {
var doc opfDocument
if err := xml.Unmarshal(content, &doc); err != nil {
return nil, err
}
return &doc, nil
}
// itemByID returns manifest items with id, href and media-type, keyed by id.
func (d *opfDocument) itemByID() map[string]opfItem {
m := make(map[string]opfItem, len(d.Manifest.Items))
for _, it := range d.Manifest.Items {
if it.ID != "" && it.Href != "" && it.MediaType != "" {
m[it.ID] = it
}
}
return m
}
// firstSpineItem returns the manifest item for the first spine idref.
func (d *opfDocument) firstSpineItem() (opfItem, bool) {
if len(d.Spine.Itemrefs) == 0 {
return opfItem{}, false
}
item, ok := d.itemByID()[d.Spine.Itemrefs[0].IDRef]
return item, ok
}
// isRasterMedia reports whether a manifest media-type is an image but not an
// (X)HTML document - Calibre's guard against cover *pages* masquerading as
// cover images.
func isRasterMedia(mediaType string) bool {
mt := strings.ToLower(strings.TrimSpace(mediaType))
if mt == "" {
return false
}
if strings.Contains(mt, "xml") || strings.Contains(mt, "html") {
return false
}
return strings.HasPrefix(mt, "image/")
}
// findRasterCoverInOPF ports Calibre's read_raster_cover resolution order:
// 1. manifest item with properties containing "cover-image"
// 2. <meta name="cover" content="ID"> resolved through the manifest
// 3. the first spine item being a raster image itself (store manga)
//
// Returns the OPF-relative href of the cover image, or "".
func (d *opfDocument) findRasterCoverInOPF() string {
// 1. properties="cover-image" (space-separated property list)
for _, it := range d.Manifest.Items {
for _, prop := range strings.Fields(it.Properties) {
if strings.EqualFold(prop, "cover-image") && isRasterMedia(it.MediaType) {
return it.Href
}
}
}
// 2. meta name="cover" content=<manifest image id>
byID := d.itemByID()
for _, m := range d.Metadata.Metas {
if !strings.EqualFold(m.Name, "cover") {
continue
}
if it, ok := byID[strings.TrimSpace(m.Content)]; ok && isRasterMedia(it.MediaType) {
return it.Href
}
}
// 3. first spine item is itself an image (jpeg/webp/png per Calibre)
if it, ok := d.firstSpineItem(); ok {
mt := strings.ToLower(it.MediaType)
if mt == "image/jpeg" || mt == "image/webp" || mt == "image/png" {
return it.Href
}
}
return ""
}
// coverPageHref returns the OPF-relative href of the cover *page* document to
// mine for an embedded image: the guide's type="cover" reference when
// present, otherwise the first spine item (Calibre renders the latter).
func (d *opfDocument) coverPageHref() string {
for _, ref := range d.Guide.References {
if strings.EqualFold(ref.Type, "cover") && ref.Href != "" {
return ref.Href
}
}
if it, ok := d.firstSpineItem(); ok {
if it.Href != "" && !isRasterMedia(it.MediaType) {
return it.Href
}
}
return ""
}
// findImageReferenceInPage extracts the first raster image reference from a
// cover (X)HTML page: <img src="..."> or SVG <image xlink:href="...">.
// Token-based parsing keeps it tolerant of mixed namespaces and fragments.
// Returns the reference relative to the page document, or "".
func findImageReferenceInPage(pageContent []byte) string {
decoder := xml.NewDecoder(bytes.NewReader(pageContent))
for {
tok, err := decoder.Token()
if err != nil {
return ""
}
start, ok := tok.(xml.StartElement)
if !ok {
continue
}
switch strings.ToLower(start.Name.Local) {
case "img":
for _, a := range start.Attr {
if strings.EqualFold(a.Name.Local, "src") && strings.TrimSpace(a.Value) != "" {
return strings.TrimSpace(a.Value)
}
}
case "image":
for _, a := range start.Attr {
if strings.EqualFold(a.Name.Local, "href") && strings.TrimSpace(a.Value) != "" {
return strings.TrimSpace(a.Value)
}
}
}
}
}