The cover lookup scraped the OPF with attribute-order-sensitive regexes. Real books serialize attributes in any order - Grand Central's '3 Days to Live' puts href before id on manifest items and content before name on the cover meta - so all three regex paths missed and the book fell through to filename guessing, extracting no cover at all. Attribute order is meaningless in XML; the regexes were never safe. Replace them with a structured parse (encoding/xml, namespace and attribute-order agnostic; see the new media_scanner_opf.go) and follow Calibre's read_raster_cover resolution order: 1. manifest item with properties=cover-image (non-(X)HTML media only) 2. <meta name=cover> resolved through the manifest, same media guard 3. first spine item that is itself a raster image (store manga) 4. NEW cover-page fallback: books declaring no raster cover at all - the classic EPUB2/Adobe cover.xhtml wrapper - are mined for <img src> / SVG <image xlink:href> references (Calibre renders the page with Qt; extracting the referenced image covers the practical cases without a rendering engine) 5. existing zip filename guessing stays as the last resort, and the old regex chain survives as findCoverInOPFLegacy for OPFs too malformed for a real XML parse. Hrefs are now URL-decoded and posix-normalized against the OPF's own path (path.Join semantics), so '../art/cover.jpg' from a nested cover page and %20-encoded names resolve correctly. Tests: attribute-order chaos modeled on the failing Patterson book, SVG-wrapped cover pages via guide references, image-first spines, and path resolution edge cases. Verified live against the real '3 Days to Live' EPUB, which previously produced no cover.
200 lines
6.1 KiB
Go
200 lines
6.1 KiB
Go
package services
|
|
|
|
import (
|
|
"bytes"
|
|
"encoding/xml"
|
|
"strings"
|
|
)
|
|
|
|
// Calibre-modeled OPF parsing. The scanner previously scraped OPF content
|
|
// with attribute-order-sensitive regexes; real books serialize attributes in
|
|
// any order (e.g. Pragmatic/Pattinson EPUBs put id before properties and
|
|
// content before name), which silently defeated cover detection. Everything
|
|
// here is parsed with encoding/xml so attribute order and namespace prefix
|
|
// choices are irrelevant.
|
|
|
|
type opfDCValue struct {
|
|
ID string `xml:"id,attr"`
|
|
Value string `xml:",chardata"`
|
|
}
|
|
|
|
type opfIdentifier struct {
|
|
Scheme string `xml:"http://www.idpf.org/2007/opf scheme,attr"`
|
|
Value string `xml:",chardata"`
|
|
}
|
|
|
|
type opfMeta struct {
|
|
ID string `xml:"id,attr"`
|
|
Name string `xml:"name,attr"`
|
|
Content string `xml:"content,attr"`
|
|
Property string `xml:"property,attr"`
|
|
Refines string `xml:"refines,attr"`
|
|
Value string `xml:",chardata"`
|
|
}
|
|
|
|
type opfItem struct {
|
|
ID string `xml:"id,attr"`
|
|
Href string `xml:"href,attr"`
|
|
MediaType string `xml:"media-type,attr"`
|
|
Properties string `xml:"properties,attr"`
|
|
}
|
|
|
|
// opfDocument is a structured view of an OPF package document.
|
|
type opfDocument struct {
|
|
Metadata struct {
|
|
Titles []opfDCValue `xml:"http://purl.org/dc/elements/1.1/ title"`
|
|
Creators []string `xml:"http://purl.org/dc/elements/1.1/ creator"`
|
|
Subjects []string `xml:"http://purl.org/dc/elements/1.1/ subject"`
|
|
Descriptions []string `xml:"http://purl.org/dc/elements/1.1/ description"`
|
|
Publishers []string `xml:"http://purl.org/dc/elements/1.1/ publisher"`
|
|
Dates []string `xml:"http://purl.org/dc/elements/1.1/ date"`
|
|
Languages []string `xml:"http://purl.org/dc/elements/1.1/ language"`
|
|
Identifiers []opfIdentifier `xml:"http://purl.org/dc/elements/1.1/ identifier"`
|
|
Contributors []string `xml:"http://purl.org/dc/elements/1.1/ contributor"`
|
|
Metas []opfMeta `xml:"meta"`
|
|
} `xml:"metadata"`
|
|
Manifest struct {
|
|
Items []opfItem `xml:"item"`
|
|
} `xml:"manifest"`
|
|
Spine struct {
|
|
Itemrefs []struct {
|
|
IDRef string `xml:"idref,attr"`
|
|
} `xml:"itemref"`
|
|
} `xml:"spine"`
|
|
Guide struct {
|
|
References []struct {
|
|
Type string `xml:"type,attr"`
|
|
Href string `xml:"href,attr"`
|
|
} `xml:"reference"`
|
|
} `xml:"guide"`
|
|
}
|
|
|
|
func parseOPFXML(content []byte) (*opfDocument, error) {
|
|
var doc opfDocument
|
|
if err := xml.Unmarshal(content, &doc); err != nil {
|
|
return nil, err
|
|
}
|
|
return &doc, nil
|
|
}
|
|
|
|
// itemByID returns manifest items with id, href and media-type, keyed by id.
|
|
func (d *opfDocument) itemByID() map[string]opfItem {
|
|
m := make(map[string]opfItem, len(d.Manifest.Items))
|
|
for _, it := range d.Manifest.Items {
|
|
if it.ID != "" && it.Href != "" && it.MediaType != "" {
|
|
m[it.ID] = it
|
|
}
|
|
}
|
|
return m
|
|
}
|
|
|
|
// firstSpineItem returns the manifest item for the first spine idref.
|
|
func (d *opfDocument) firstSpineItem() (opfItem, bool) {
|
|
if len(d.Spine.Itemrefs) == 0 {
|
|
return opfItem{}, false
|
|
}
|
|
item, ok := d.itemByID()[d.Spine.Itemrefs[0].IDRef]
|
|
return item, ok
|
|
}
|
|
|
|
// isRasterMedia reports whether a manifest media-type is an image but not an
|
|
// (X)HTML document - Calibre's guard against cover *pages* masquerading as
|
|
// cover images.
|
|
func isRasterMedia(mediaType string) bool {
|
|
mt := strings.ToLower(strings.TrimSpace(mediaType))
|
|
if mt == "" {
|
|
return false
|
|
}
|
|
if strings.Contains(mt, "xml") || strings.Contains(mt, "html") {
|
|
return false
|
|
}
|
|
return strings.HasPrefix(mt, "image/")
|
|
}
|
|
|
|
// findRasterCoverInOPF ports Calibre's read_raster_cover resolution order:
|
|
// 1. manifest item with properties containing "cover-image"
|
|
// 2. <meta name="cover" content="ID"> resolved through the manifest
|
|
// 3. the first spine item being a raster image itself (store manga)
|
|
//
|
|
// Returns the OPF-relative href of the cover image, or "".
|
|
func (d *opfDocument) findRasterCoverInOPF() string {
|
|
// 1. properties="cover-image" (space-separated property list)
|
|
for _, it := range d.Manifest.Items {
|
|
for _, prop := range strings.Fields(it.Properties) {
|
|
if strings.EqualFold(prop, "cover-image") && isRasterMedia(it.MediaType) {
|
|
return it.Href
|
|
}
|
|
}
|
|
}
|
|
|
|
// 2. meta name="cover" content=<manifest image id>
|
|
byID := d.itemByID()
|
|
for _, m := range d.Metadata.Metas {
|
|
if !strings.EqualFold(m.Name, "cover") {
|
|
continue
|
|
}
|
|
if it, ok := byID[strings.TrimSpace(m.Content)]; ok && isRasterMedia(it.MediaType) {
|
|
return it.Href
|
|
}
|
|
}
|
|
|
|
// 3. first spine item is itself an image (jpeg/webp/png per Calibre)
|
|
if it, ok := d.firstSpineItem(); ok {
|
|
mt := strings.ToLower(it.MediaType)
|
|
if mt == "image/jpeg" || mt == "image/webp" || mt == "image/png" {
|
|
return it.Href
|
|
}
|
|
}
|
|
|
|
return ""
|
|
}
|
|
|
|
// coverPageHref returns the OPF-relative href of the cover *page* document to
|
|
// mine for an embedded image: the guide's type="cover" reference when
|
|
// present, otherwise the first spine item (Calibre renders the latter).
|
|
func (d *opfDocument) coverPageHref() string {
|
|
for _, ref := range d.Guide.References {
|
|
if strings.EqualFold(ref.Type, "cover") && ref.Href != "" {
|
|
return ref.Href
|
|
}
|
|
}
|
|
if it, ok := d.firstSpineItem(); ok {
|
|
if it.Href != "" && !isRasterMedia(it.MediaType) {
|
|
return it.Href
|
|
}
|
|
}
|
|
return ""
|
|
}
|
|
|
|
// findImageReferenceInPage extracts the first raster image reference from a
|
|
// cover (X)HTML page: <img src="..."> or SVG <image xlink:href="...">.
|
|
// Token-based parsing keeps it tolerant of mixed namespaces and fragments.
|
|
// Returns the reference relative to the page document, or "".
|
|
func findImageReferenceInPage(pageContent []byte) string {
|
|
decoder := xml.NewDecoder(bytes.NewReader(pageContent))
|
|
for {
|
|
tok, err := decoder.Token()
|
|
if err != nil {
|
|
return ""
|
|
}
|
|
start, ok := tok.(xml.StartElement)
|
|
if !ok {
|
|
continue
|
|
}
|
|
switch strings.ToLower(start.Name.Local) {
|
|
case "img":
|
|
for _, a := range start.Attr {
|
|
if strings.EqualFold(a.Name.Local, "src") && strings.TrimSpace(a.Value) != "" {
|
|
return strings.TrimSpace(a.Value)
|
|
}
|
|
}
|
|
case "image":
|
|
for _, a := range start.Attr {
|
|
if strings.EqualFold(a.Name.Local, "href") && strings.TrimSpace(a.Value) != "" {
|
|
return strings.TrimSpace(a.Value)
|
|
}
|
|
}
|
|
}
|
|
}
|
|
}
|