fix(scanner): resolve EPUB covers via structured OPF parsing, Calibre chain
The cover lookup scraped the OPF with attribute-order-sensitive regexes. Real books serialize attributes in any order - Grand Central's '3 Days to Live' puts href before id on manifest items and content before name on the cover meta - so all three regex paths missed and the book fell through to filename guessing, extracting no cover at all. Attribute order is meaningless in XML; the regexes were never safe. Replace them with a structured parse (encoding/xml, namespace and attribute-order agnostic; see the new media_scanner_opf.go) and follow Calibre's read_raster_cover resolution order: 1. manifest item with properties=cover-image (non-(X)HTML media only) 2. <meta name=cover> resolved through the manifest, same media guard 3. first spine item that is itself a raster image (store manga) 4. NEW cover-page fallback: books declaring no raster cover at all - the classic EPUB2/Adobe cover.xhtml wrapper - are mined for <img src> / SVG <image xlink:href> references (Calibre renders the page with Qt; extracting the referenced image covers the practical cases without a rendering engine) 5. existing zip filename guessing stays as the last resort, and the old regex chain survives as findCoverInOPFLegacy for OPFs too malformed for a real XML parse. Hrefs are now URL-decoded and posix-normalized against the OPF's own path (path.Join semantics), so '../art/cover.jpg' from a nested cover page and %20-encoded names resolve correctly. Tests: attribute-order chaos modeled on the failing Patterson book, SVG-wrapped cover pages via guide references, image-first spines, and path resolution edge cases. Verified live against the real '3 Days to Live' EPUB, which previously produced no cover.
This commit is contained in:
@@ -0,0 +1,199 @@
|
||||
package services
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/xml"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// Calibre-modeled OPF parsing. The scanner previously scraped OPF content
|
||||
// with attribute-order-sensitive regexes; real books serialize attributes in
|
||||
// any order (e.g. Pragmatic/Pattinson EPUBs put id before properties and
|
||||
// content before name), which silently defeated cover detection. Everything
|
||||
// here is parsed with encoding/xml so attribute order and namespace prefix
|
||||
// choices are irrelevant.
|
||||
|
||||
type opfDCValue struct {
|
||||
ID string `xml:"id,attr"`
|
||||
Value string `xml:",chardata"`
|
||||
}
|
||||
|
||||
type opfIdentifier struct {
|
||||
Scheme string `xml:"http://www.idpf.org/2007/opf scheme,attr"`
|
||||
Value string `xml:",chardata"`
|
||||
}
|
||||
|
||||
type opfMeta struct {
|
||||
ID string `xml:"id,attr"`
|
||||
Name string `xml:"name,attr"`
|
||||
Content string `xml:"content,attr"`
|
||||
Property string `xml:"property,attr"`
|
||||
Refines string `xml:"refines,attr"`
|
||||
Value string `xml:",chardata"`
|
||||
}
|
||||
|
||||
type opfItem struct {
|
||||
ID string `xml:"id,attr"`
|
||||
Href string `xml:"href,attr"`
|
||||
MediaType string `xml:"media-type,attr"`
|
||||
Properties string `xml:"properties,attr"`
|
||||
}
|
||||
|
||||
// opfDocument is a structured view of an OPF package document.
|
||||
type opfDocument struct {
|
||||
Metadata struct {
|
||||
Titles []opfDCValue `xml:"http://purl.org/dc/elements/1.1/ title"`
|
||||
Creators []string `xml:"http://purl.org/dc/elements/1.1/ creator"`
|
||||
Subjects []string `xml:"http://purl.org/dc/elements/1.1/ subject"`
|
||||
Descriptions []string `xml:"http://purl.org/dc/elements/1.1/ description"`
|
||||
Publishers []string `xml:"http://purl.org/dc/elements/1.1/ publisher"`
|
||||
Dates []string `xml:"http://purl.org/dc/elements/1.1/ date"`
|
||||
Languages []string `xml:"http://purl.org/dc/elements/1.1/ language"`
|
||||
Identifiers []opfIdentifier `xml:"http://purl.org/dc/elements/1.1/ identifier"`
|
||||
Contributors []string `xml:"http://purl.org/dc/elements/1.1/ contributor"`
|
||||
Metas []opfMeta `xml:"meta"`
|
||||
} `xml:"metadata"`
|
||||
Manifest struct {
|
||||
Items []opfItem `xml:"item"`
|
||||
} `xml:"manifest"`
|
||||
Spine struct {
|
||||
Itemrefs []struct {
|
||||
IDRef string `xml:"idref,attr"`
|
||||
} `xml:"itemref"`
|
||||
} `xml:"spine"`
|
||||
Guide struct {
|
||||
References []struct {
|
||||
Type string `xml:"type,attr"`
|
||||
Href string `xml:"href,attr"`
|
||||
} `xml:"reference"`
|
||||
} `xml:"guide"`
|
||||
}
|
||||
|
||||
func parseOPFXML(content []byte) (*opfDocument, error) {
|
||||
var doc opfDocument
|
||||
if err := xml.Unmarshal(content, &doc); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return &doc, nil
|
||||
}
|
||||
|
||||
// itemByID returns manifest items with id, href and media-type, keyed by id.
|
||||
func (d *opfDocument) itemByID() map[string]opfItem {
|
||||
m := make(map[string]opfItem, len(d.Manifest.Items))
|
||||
for _, it := range d.Manifest.Items {
|
||||
if it.ID != "" && it.Href != "" && it.MediaType != "" {
|
||||
m[it.ID] = it
|
||||
}
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
// firstSpineItem returns the manifest item for the first spine idref.
|
||||
func (d *opfDocument) firstSpineItem() (opfItem, bool) {
|
||||
if len(d.Spine.Itemrefs) == 0 {
|
||||
return opfItem{}, false
|
||||
}
|
||||
item, ok := d.itemByID()[d.Spine.Itemrefs[0].IDRef]
|
||||
return item, ok
|
||||
}
|
||||
|
||||
// isRasterMedia reports whether a manifest media-type is an image but not an
|
||||
// (X)HTML document - Calibre's guard against cover *pages* masquerading as
|
||||
// cover images.
|
||||
func isRasterMedia(mediaType string) bool {
|
||||
mt := strings.ToLower(strings.TrimSpace(mediaType))
|
||||
if mt == "" {
|
||||
return false
|
||||
}
|
||||
if strings.Contains(mt, "xml") || strings.Contains(mt, "html") {
|
||||
return false
|
||||
}
|
||||
return strings.HasPrefix(mt, "image/")
|
||||
}
|
||||
|
||||
// findRasterCoverInOPF ports Calibre's read_raster_cover resolution order:
|
||||
// 1. manifest item with properties containing "cover-image"
|
||||
// 2. <meta name="cover" content="ID"> resolved through the manifest
|
||||
// 3. the first spine item being a raster image itself (store manga)
|
||||
//
|
||||
// Returns the OPF-relative href of the cover image, or "".
|
||||
func (d *opfDocument) findRasterCoverInOPF() string {
|
||||
// 1. properties="cover-image" (space-separated property list)
|
||||
for _, it := range d.Manifest.Items {
|
||||
for _, prop := range strings.Fields(it.Properties) {
|
||||
if strings.EqualFold(prop, "cover-image") && isRasterMedia(it.MediaType) {
|
||||
return it.Href
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// 2. meta name="cover" content=<manifest image id>
|
||||
byID := d.itemByID()
|
||||
for _, m := range d.Metadata.Metas {
|
||||
if !strings.EqualFold(m.Name, "cover") {
|
||||
continue
|
||||
}
|
||||
if it, ok := byID[strings.TrimSpace(m.Content)]; ok && isRasterMedia(it.MediaType) {
|
||||
return it.Href
|
||||
}
|
||||
}
|
||||
|
||||
// 3. first spine item is itself an image (jpeg/webp/png per Calibre)
|
||||
if it, ok := d.firstSpineItem(); ok {
|
||||
mt := strings.ToLower(it.MediaType)
|
||||
if mt == "image/jpeg" || mt == "image/webp" || mt == "image/png" {
|
||||
return it.Href
|
||||
}
|
||||
}
|
||||
|
||||
return ""
|
||||
}
|
||||
|
||||
// coverPageHref returns the OPF-relative href of the cover *page* document to
|
||||
// mine for an embedded image: the guide's type="cover" reference when
|
||||
// present, otherwise the first spine item (Calibre renders the latter).
|
||||
func (d *opfDocument) coverPageHref() string {
|
||||
for _, ref := range d.Guide.References {
|
||||
if strings.EqualFold(ref.Type, "cover") && ref.Href != "" {
|
||||
return ref.Href
|
||||
}
|
||||
}
|
||||
if it, ok := d.firstSpineItem(); ok {
|
||||
if it.Href != "" && !isRasterMedia(it.MediaType) {
|
||||
return it.Href
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// findImageReferenceInPage extracts the first raster image reference from a
|
||||
// cover (X)HTML page: <img src="..."> or SVG <image xlink:href="...">.
|
||||
// Token-based parsing keeps it tolerant of mixed namespaces and fragments.
|
||||
// Returns the reference relative to the page document, or "".
|
||||
func findImageReferenceInPage(pageContent []byte) string {
|
||||
decoder := xml.NewDecoder(bytes.NewReader(pageContent))
|
||||
for {
|
||||
tok, err := decoder.Token()
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
start, ok := tok.(xml.StartElement)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
switch strings.ToLower(start.Name.Local) {
|
||||
case "img":
|
||||
for _, a := range start.Attr {
|
||||
if strings.EqualFold(a.Name.Local, "src") && strings.TrimSpace(a.Value) != "" {
|
||||
return strings.TrimSpace(a.Value)
|
||||
}
|
||||
}
|
||||
case "image":
|
||||
for _, a := range start.Attr {
|
||||
if strings.EqualFold(a.Name.Local, "href") && strings.TrimSpace(a.Value) != "" {
|
||||
return strings.TrimSpace(a.Value)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user