package services import ( "bytes" "encoding/xml" "strings" ) // Calibre-modeled OPF parsing. The scanner previously scraped OPF content // with attribute-order-sensitive regexes; real books serialize attributes in // any order (e.g. Pragmatic/Pattinson EPUBs put id before properties and // content before name), which silently defeated cover detection. Everything // here is parsed with encoding/xml so attribute order and namespace prefix // choices are irrelevant. type opfDCValue struct { ID string `xml:"id,attr"` Value string `xml:",chardata"` } type opfIdentifier struct { Scheme string `xml:"http://www.idpf.org/2007/opf scheme,attr"` Value string `xml:",chardata"` } type opfMeta struct { ID string `xml:"id,attr"` Name string `xml:"name,attr"` Content string `xml:"content,attr"` Property string `xml:"property,attr"` Refines string `xml:"refines,attr"` Value string `xml:",chardata"` } type opfItem struct { ID string `xml:"id,attr"` Href string `xml:"href,attr"` MediaType string `xml:"media-type,attr"` Properties string `xml:"properties,attr"` } // opfDocument is a structured view of an OPF package document. type opfDocument struct { Metadata struct { Titles []opfDCValue `xml:"http://purl.org/dc/elements/1.1/ title"` Creators []string `xml:"http://purl.org/dc/elements/1.1/ creator"` Subjects []string `xml:"http://purl.org/dc/elements/1.1/ subject"` Descriptions []string `xml:"http://purl.org/dc/elements/1.1/ description"` Publishers []string `xml:"http://purl.org/dc/elements/1.1/ publisher"` Dates []string `xml:"http://purl.org/dc/elements/1.1/ date"` Languages []string `xml:"http://purl.org/dc/elements/1.1/ language"` Identifiers []opfIdentifier `xml:"http://purl.org/dc/elements/1.1/ identifier"` Contributors []string `xml:"http://purl.org/dc/elements/1.1/ contributor"` Metas []opfMeta `xml:"meta"` } `xml:"metadata"` Manifest struct { Items []opfItem `xml:"item"` } `xml:"manifest"` Spine struct { Itemrefs []struct { IDRef string `xml:"idref,attr"` } `xml:"itemref"` } `xml:"spine"` Guide struct { References []struct { Type string `xml:"type,attr"` Href string `xml:"href,attr"` } `xml:"reference"` } `xml:"guide"` } func parseOPFXML(content []byte) (*opfDocument, error) { var doc opfDocument if err := xml.Unmarshal(content, &doc); err != nil { return nil, err } return &doc, nil } // itemByID returns manifest items with id, href and media-type, keyed by id. func (d *opfDocument) itemByID() map[string]opfItem { m := make(map[string]opfItem, len(d.Manifest.Items)) for _, it := range d.Manifest.Items { if it.ID != "" && it.Href != "" && it.MediaType != "" { m[it.ID] = it } } return m } // firstSpineItem returns the manifest item for the first spine idref. func (d *opfDocument) firstSpineItem() (opfItem, bool) { if len(d.Spine.Itemrefs) == 0 { return opfItem{}, false } item, ok := d.itemByID()[d.Spine.Itemrefs[0].IDRef] return item, ok } // isRasterMedia reports whether a manifest media-type is an image but not an // (X)HTML document - Calibre's guard against cover *pages* masquerading as // cover images. func isRasterMedia(mediaType string) bool { mt := strings.ToLower(strings.TrimSpace(mediaType)) if mt == "" { return false } if strings.Contains(mt, "xml") || strings.Contains(mt, "html") { return false } return strings.HasPrefix(mt, "image/") } // findRasterCoverInOPF ports Calibre's read_raster_cover resolution order: // 1. manifest item with properties containing "cover-image" // 2. resolved through the manifest // 3. the first spine item being a raster image itself (store manga) // // Returns the OPF-relative href of the cover image, or "". func (d *opfDocument) findRasterCoverInOPF() string { // 1. properties="cover-image" (space-separated property list) for _, it := range d.Manifest.Items { for _, prop := range strings.Fields(it.Properties) { if strings.EqualFold(prop, "cover-image") && isRasterMedia(it.MediaType) { return it.Href } } } // 2. meta name="cover" content= byID := d.itemByID() for _, m := range d.Metadata.Metas { if !strings.EqualFold(m.Name, "cover") { continue } if it, ok := byID[strings.TrimSpace(m.Content)]; ok && isRasterMedia(it.MediaType) { return it.Href } } // 3. first spine item is itself an image (jpeg/webp/png per Calibre) if it, ok := d.firstSpineItem(); ok { mt := strings.ToLower(it.MediaType) if mt == "image/jpeg" || mt == "image/webp" || mt == "image/png" { return it.Href } } return "" } // coverPageHref returns the OPF-relative href of the cover *page* document to // mine for an embedded image: the guide's type="cover" reference when // present, otherwise the first spine item (Calibre renders the latter). func (d *opfDocument) coverPageHref() string { for _, ref := range d.Guide.References { if strings.EqualFold(ref.Type, "cover") && ref.Href != "" { return ref.Href } } if it, ok := d.firstSpineItem(); ok { if it.Href != "" && !isRasterMedia(it.MediaType) { return it.Href } } return "" } // findImageReferenceInPage extracts the first raster image reference from a // cover (X)HTML page: or SVG . // Token-based parsing keeps it tolerant of mixed namespaces and fragments. // Returns the reference relative to the page document, or "". func findImageReferenceInPage(pageContent []byte) string { decoder := xml.NewDecoder(bytes.NewReader(pageContent)) for { tok, err := decoder.Token() if err != nil { return "" } start, ok := tok.(xml.StartElement) if !ok { continue } switch strings.ToLower(start.Name.Local) { case "img": for _, a := range start.Attr { if strings.EqualFold(a.Name.Local, "src") && strings.TrimSpace(a.Value) != "" { return strings.TrimSpace(a.Value) } } case "image": for _, a := range start.Attr { if strings.EqualFold(a.Name.Local, "href") && strings.TrimSpace(a.Value) != "" { return strings.TrimSpace(a.Value) } } } } }