feat(scanner): Calibre-aligned OPF metadata extraction
Adopt Calibre's reading conventions for the Dublin Core metadata that
parseOPFContent now pulls from the structured OPF parse:
- Titles: EPUB3 title-type selection (prefer 'main', join a distinct
subtitle with ': ' exactly as Calibre stores it). There is no separate
subtitle column by design - Calibre-sidecar books arrive pre-joined,
so a column would stay empty for most libraries and force every client
to reimplement concatenation.
- Genre: first dc:subject, mirroring the existing processGenresAndTags
behavior of the Calibre-sidecar path; the embedded path never
populated Genre before. Subjects stay one-element-one-tag - Library
of Congress headings legitimately contain commas ("Holmes, Sherlock
(Fictitious character) -- Fiction") and must not be split.
- Identifiers: urn:isbn:/urn:asin: prefixed values parse in addition to
opf:scheme attributes, and the scheme-less fallback now requires an
ISBN-shaped value (10/13 digits, optional separators/trailing X) so
URIs like the Gutenberg identifiers cannot masquerade as ISBNs -
observed live on 'A Study in Scarlet'.
- Series: EPUB3 belongs-to-collection with collection-type=series and
group-position refines, ahead of the classic calibre:series metas.
- Audiobookshelf metadata.json sidecars join their subtitle field into
the title the same way.
Tests cover title-type main+subtitle joining, belongs-to-collection
series with fractional group-position, urn:isbn extraction, genre/tag
parity, and comma preservation inside subject headings.
This commit is contained in:
@@ -1025,6 +1025,7 @@ func extractAudiobookshelfSidecar(path string) *MediaMetadata {
|
|||||||
|
|
||||||
var sidecar struct {
|
var sidecar struct {
|
||||||
Title string `json:"title"`
|
Title string `json:"title"`
|
||||||
|
Subtitle string `json:"subtitle"`
|
||||||
Authors []string `json:"authors"`
|
Authors []string `json:"authors"`
|
||||||
Series []struct {
|
Series []struct {
|
||||||
Series string `json:"series"`
|
Series string `json:"series"`
|
||||||
@@ -1048,6 +1049,11 @@ func extractAudiobookshelfSidecar(path string) *MediaMetadata {
|
|||||||
metadata := &MediaMetadata{
|
metadata := &MediaMetadata{
|
||||||
Title: strings.TrimSpace(sidecar.Title),
|
Title: strings.TrimSpace(sidecar.Title),
|
||||||
}
|
}
|
||||||
|
// Subtitle joins the title Calibre-style ("Main: Subtitle") - there is no
|
||||||
|
// separate subtitle column, and Calibre-sidecar books arrive pre-joined.
|
||||||
|
if subtitle := strings.TrimSpace(sidecar.Subtitle); subtitle != "" && metadata.Title != "" {
|
||||||
|
metadata.Title = metadata.Title + ": " + subtitle
|
||||||
|
}
|
||||||
if len(sidecar.Authors) > 0 {
|
if len(sidecar.Authors) > 0 {
|
||||||
metadata.Author = strings.TrimSpace(sidecar.Authors[0])
|
metadata.Author = strings.TrimSpace(sidecar.Authors[0])
|
||||||
}
|
}
|
||||||
@@ -1721,80 +1727,75 @@ func (s *MediaScanner) parseCalibreMetadataOPF(opfPath string) (*MediaMetadata,
|
|||||||
|
|
||||||
// parseOPFContent parses an OPF document (Dublin Core metadata) into
|
// parseOPFContent parses an OPF document (Dublin Core metadata) into
|
||||||
// MediaMetadata. Used for both Calibre metadata.opf sidecars and the OPF
|
// MediaMetadata. Used for both Calibre metadata.opf sidecars and the OPF
|
||||||
// embedded inside an EPUB - the dc:* vocabulary is identical. Namespace-aware
|
// embedded inside an EPUB - the dc:* vocabulary is identical. Parsing is
|
||||||
// parsing means it tolerates wherever the xmlns:dc declaration lives.
|
// attribute-order agnostic and namespace aware (see media_scanner_opf.go);
|
||||||
|
// title selection and series detection follow Calibre's behavior.
|
||||||
func parseOPFContent(content []byte) (*MediaMetadata, error) {
|
func parseOPFContent(content []byte) (*MediaMetadata, error) {
|
||||||
// Define XML structure for parsing with full Dublin Core namespace URLs
|
opf, err := parseOPFXML(content)
|
||||||
var opf struct {
|
if err != nil {
|
||||||
XMLName xml.Name `xml:"package"`
|
|
||||||
Metadata struct {
|
|
||||||
XMLName xml.Name `xml:"metadata"`
|
|
||||||
Titles []string `xml:"http://purl.org/dc/elements/1.1/ title"`
|
|
||||||
Creators []string `xml:"http://purl.org/dc/elements/1.1/ creator"`
|
|
||||||
Subjects []string `xml:"http://purl.org/dc/elements/1.1/ subject"`
|
|
||||||
Desc []string `xml:"http://purl.org/dc/elements/1.1/ description"`
|
|
||||||
Publisher []string `xml:"http://purl.org/dc/elements/1.1/ publisher"`
|
|
||||||
Dates []string `xml:"http://purl.org/dc/elements/1.1/ date"`
|
|
||||||
Language []string `xml:"http://purl.org/dc/elements/1.1/ language"`
|
|
||||||
Identifiers []struct {
|
|
||||||
Scheme string `xml:"http://www.idpf.org/2007/opf scheme,attr"`
|
|
||||||
Value string `xml:",chardata"`
|
|
||||||
} `xml:"http://purl.org/dc/elements/1.1/ identifier"`
|
|
||||||
Contributors []string `xml:"http://purl.org/dc/elements/1.1/ contributor"`
|
|
||||||
// Calibre-specific meta tags - capture all, filter later
|
|
||||||
MetaTags []struct {
|
|
||||||
Name string `xml:"name,attr"`
|
|
||||||
Value string `xml:"content,attr"`
|
|
||||||
} `xml:"meta"`
|
|
||||||
} `xml:"metadata"`
|
|
||||||
}
|
|
||||||
// Parse XML
|
|
||||||
if err := xml.NewDecoder(bytes.NewReader(content)).Decode(&opf); err != nil {
|
|
||||||
return nil, fmt.Errorf("failed to parse OPF XML: %v", err)
|
return nil, fmt.Errorf("failed to parse OPF XML: %v", err)
|
||||||
}
|
}
|
||||||
|
md := opf.Metadata
|
||||||
|
|
||||||
// Map to MediaMetadata struct
|
// Map to MediaMetadata struct
|
||||||
metadata := &MediaMetadata{}
|
metadata := &MediaMetadata{}
|
||||||
// Title (required)
|
// Title: EPUB3 title-type main selection with Calibre-style subtitle join
|
||||||
if len(opf.Metadata.Titles) > 0 {
|
if title := opf.selectTitle(); title != "" {
|
||||||
metadata.Title = opf.Metadata.Titles[0]
|
metadata.Title = title
|
||||||
}
|
}
|
||||||
// Author (first creator)
|
// Author (first creator)
|
||||||
if len(opf.Metadata.Creators) > 0 {
|
if len(md.Creators) > 0 {
|
||||||
metadata.Author = opf.Metadata.Creators[0]
|
metadata.Author = md.Creators[0]
|
||||||
|
}
|
||||||
|
// Tags (all subjects; one element = one tag, commas are legal inside
|
||||||
|
// subject headings like "Holmes, Sherlock (Fictitious character)").
|
||||||
|
// Genre mirrors the sidecar path's processGenresAndTags: first subject.
|
||||||
|
if len(md.Subjects) > 0 {
|
||||||
|
metadata.Tags = utils.NormalizeTags(md.Subjects)
|
||||||
|
if len(metadata.Tags) > 0 {
|
||||||
|
metadata.Genre = metadata.Tags[0]
|
||||||
}
|
}
|
||||||
// Tags (all subjects)
|
|
||||||
if len(opf.Metadata.Subjects) > 0 {
|
|
||||||
metadata.Tags = utils.NormalizeTags(opf.Metadata.Subjects)
|
|
||||||
}
|
}
|
||||||
// Description
|
// Description
|
||||||
if len(opf.Metadata.Desc) > 0 {
|
if len(md.Descriptions) > 0 {
|
||||||
metadata.Description = opf.Metadata.Desc[0]
|
metadata.Description = md.Descriptions[0]
|
||||||
}
|
}
|
||||||
// Publisher
|
// Publisher
|
||||||
if len(opf.Metadata.Publisher) > 0 {
|
if len(md.Publishers) > 0 {
|
||||||
metadata.Publisher = opf.Metadata.Publisher[0]
|
metadata.Publisher = md.Publishers[0]
|
||||||
}
|
}
|
||||||
// Language
|
// Language
|
||||||
if len(opf.Metadata.Language) > 0 && opf.Metadata.Language[0] != "" {
|
if len(md.Languages) > 0 && md.Languages[0] != "" {
|
||||||
metadata.Language = opf.Metadata.Language[0]
|
metadata.Language = md.Languages[0]
|
||||||
}
|
}
|
||||||
// Publish date
|
// Publish date
|
||||||
if len(opf.Metadata.Dates) > 0 {
|
if len(md.Dates) > 0 {
|
||||||
if date, err := time.Parse("2006-01-02T15:04:05Z07:00", opf.Metadata.Dates[0]); err == nil {
|
if date, err := time.Parse("2006-01-02T15:04:05Z07:00", md.Dates[0]); err == nil {
|
||||||
metadata.PublishDate = date
|
metadata.PublishDate = date
|
||||||
} else if date, err := time.Parse("2006-01-02", opf.Metadata.Dates[0]); err == nil {
|
} else if date, err := time.Parse("2006-01-02", md.Dates[0]); err == nil {
|
||||||
metadata.PublishDate = date
|
metadata.PublishDate = date
|
||||||
} else {
|
} else {
|
||||||
// Try alternative date formats
|
// Try alternative date formats
|
||||||
if date, err := time.Parse("2006", opf.Metadata.Dates[0]); err == nil {
|
if date, err := time.Parse("2006", md.Dates[0]); err == nil {
|
||||||
metadata.PublishDate = date
|
metadata.PublishDate = date
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// Identifiers (ISBN, ASIN)
|
// Identifiers: opf:scheme attribute first, then URN-prefixed values
|
||||||
for _, id := range opf.Metadata.Identifiers {
|
// (urn:isbn:...), then bare values that normalize to a valid ISBN.
|
||||||
|
for _, id := range md.Identifiers {
|
||||||
value := strings.TrimSpace(id.Value)
|
value := strings.TrimSpace(id.Value)
|
||||||
switch strings.ToUpper(id.Scheme) {
|
scheme := strings.ToUpper(strings.TrimSpace(id.Scheme))
|
||||||
|
if scheme == "" {
|
||||||
|
if lower := strings.ToLower(value); strings.HasPrefix(lower, "urn:") {
|
||||||
|
rest := value[4:]
|
||||||
|
if prefix, val, ok := strings.Cut(rest, ":"); ok {
|
||||||
|
scheme = strings.ToUpper(prefix)
|
||||||
|
value = strings.TrimSpace(val)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
switch scheme {
|
||||||
case "ISBN":
|
case "ISBN":
|
||||||
metadata.ISBN = utils.NormalizeISBNSafe(value)
|
metadata.ISBN = utils.NormalizeISBNSafe(value)
|
||||||
case "ASIN":
|
case "ASIN":
|
||||||
@@ -1803,28 +1804,30 @@ func parseOPFContent(content []byte) (*MediaMetadata, error) {
|
|||||||
// Store UUID in hash info, not metadata
|
// Store UUID in hash info, not metadata
|
||||||
// Will be extracted by extractHashInfo()
|
// Will be extracted by extractHashInfo()
|
||||||
default:
|
default:
|
||||||
// EPUB3 identifiers often carry no opf:scheme attribute;
|
// EPUB3 identifiers often carry no opf:scheme attribute; accept a
|
||||||
// accept a bare value that normalizes to a valid ISBN.
|
// bare value only when it is ISBN-shaped (rejects the URIs and
|
||||||
if metadata.ISBN == "" && id.Scheme == "" {
|
// UUIDs that commonly share the identifier list).
|
||||||
if normalized := utils.NormalizeISBNSafe(value); normalized != "" {
|
if metadata.ISBN == "" && scheme == "" && isISBNLike(value) {
|
||||||
|
if normalized, err := utils.NormalizeISBN(value); err == nil {
|
||||||
metadata.ISBN = normalized
|
metadata.ISBN = normalized
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// Contributors
|
// Contributors
|
||||||
if len(opf.Metadata.Contributors) > 0 {
|
if len(md.Contributors) > 0 {
|
||||||
metadata.Contributors = utils.NormalizeContributors(opf.Metadata.Contributors)
|
metadata.Contributors = utils.NormalizeContributors(md.Contributors)
|
||||||
}
|
}
|
||||||
// Calibre-specific meta tags (filter by name attribute)
|
// Series: EPUB3 belongs-to-collection, then calibre:series metas
|
||||||
for _, meta := range opf.Metadata.MetaTags {
|
if series, index := opf.readSeries(); series != "" {
|
||||||
switch meta.Name {
|
metadata.Series = series
|
||||||
case "calibre:series":
|
if index > 0 {
|
||||||
metadata.Series = meta.Value
|
|
||||||
case "calibre:series_index":
|
|
||||||
if index, err := strconv.ParseFloat(meta.Value, 32); err == nil {
|
|
||||||
metadata.SeriesNumber = int32(index)
|
metadata.SeriesNumber = int32(index)
|
||||||
}
|
}
|
||||||
|
}
|
||||||
|
// Calibre-specific meta tags (filter by name attribute)
|
||||||
|
for _, meta := range md.Metas {
|
||||||
|
switch meta.Name {
|
||||||
case "calibre:rating":
|
case "calibre:rating":
|
||||||
// Not imported (ratings are per-user in Bookhoard)
|
// Not imported (ratings are per-user in Bookhoard)
|
||||||
case "calibre:title_sort":
|
case "calibre:title_sort":
|
||||||
|
|||||||
@@ -3,6 +3,7 @@ package services
|
|||||||
import (
|
import (
|
||||||
"bytes"
|
"bytes"
|
||||||
"encoding/xml"
|
"encoding/xml"
|
||||||
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -77,6 +78,112 @@ func parseOPFXML(content []byte) (*opfDocument, error) {
|
|||||||
return &doc, nil
|
return &doc, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// refinesFor maps an element id to its EPUB3 refining metas
|
||||||
|
// (those whose refines attribute starts with '#').
|
||||||
|
func (d *opfDocument) refinesFor(id string) []opfMeta {
|
||||||
|
var out []opfMeta
|
||||||
|
if id == "" {
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
for _, m := range d.Metadata.Metas {
|
||||||
|
if strings.HasPrefix(m.Refines, "#") && m.Refines[1:] == id {
|
||||||
|
out = append(out, m)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
// refinesProperty returns the value of the first refining meta carrying the
|
||||||
|
// given property (e.g. "title-type", "collection-type", "group-position").
|
||||||
|
func refinesProperty(metas []opfMeta, property string) (string, bool) {
|
||||||
|
for _, m := range metas {
|
||||||
|
if strings.EqualFold(m.Property, property) {
|
||||||
|
if v := strings.TrimSpace(m.Value); v != "" {
|
||||||
|
return v, true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
|
||||||
|
// selectTitle ports Calibre's read_title: prefer the dc:title refined as
|
||||||
|
// title-type "main"; fall back to the first non-empty title. A distinct
|
||||||
|
// subtitle (title-type containing "subtitle"/"sub-title") is joined onto the
|
||||||
|
// main title with ": ", exactly as Calibre stores it.
|
||||||
|
func (d *opfDocument) selectTitle() string {
|
||||||
|
var first, main, subtitle string
|
||||||
|
for _, t := range d.Metadata.Titles {
|
||||||
|
v := strings.TrimSpace(t.Value)
|
||||||
|
if v == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if first == "" {
|
||||||
|
first = v
|
||||||
|
}
|
||||||
|
tt, ok := refinesProperty(d.refinesFor(t.ID), "title-type")
|
||||||
|
if !ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
switch strings.ToLower(tt) {
|
||||||
|
case "main":
|
||||||
|
if main == "" {
|
||||||
|
main = v
|
||||||
|
}
|
||||||
|
default:
|
||||||
|
l := strings.ToLower(tt)
|
||||||
|
if strings.Contains(l, "subtitle") || strings.Contains(l, "sub-title") {
|
||||||
|
if subtitle == "" {
|
||||||
|
subtitle = v
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
title := main
|
||||||
|
if title == "" {
|
||||||
|
title = first
|
||||||
|
}
|
||||||
|
if subtitle != "" && subtitle != title {
|
||||||
|
title = title + ": " + subtitle
|
||||||
|
}
|
||||||
|
return title
|
||||||
|
}
|
||||||
|
|
||||||
|
// readSeries ports Calibre's read_series: EPUB3 belongs-to-collection (with a
|
||||||
|
// collection-type=series refine and group-position index) first, then the
|
||||||
|
// classic calibre:series / calibre:series_index metas.
|
||||||
|
func (d *opfDocument) readSeries() (series string, index float64) {
|
||||||
|
for _, m := range d.Metadata.Metas {
|
||||||
|
if !strings.EqualFold(m.Property, "belongs-to-collection") {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
name := strings.TrimSpace(m.Value)
|
||||||
|
if name == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
refines := d.refinesFor(m.ID)
|
||||||
|
if ct, ok := refinesProperty(refines, "collection-type"); !ok || !strings.EqualFold(ct, "series") {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if gp, ok := refinesProperty(refines, "group-position"); ok {
|
||||||
|
if v, err := strconv.ParseFloat(strings.TrimSpace(gp), 64); err == nil {
|
||||||
|
index = v
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return name, index
|
||||||
|
}
|
||||||
|
for _, m := range d.Metadata.Metas {
|
||||||
|
switch m.Name {
|
||||||
|
case "calibre:series":
|
||||||
|
series = m.Content
|
||||||
|
case "calibre:series_index":
|
||||||
|
if v, err := strconv.ParseFloat(strings.TrimSpace(m.Content), 64); err == nil {
|
||||||
|
index = v
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return series, index
|
||||||
|
}
|
||||||
|
|
||||||
// itemByID returns manifest items with id, href and media-type, keyed by id.
|
// itemByID returns manifest items with id, href and media-type, keyed by id.
|
||||||
func (d *opfDocument) itemByID() map[string]opfItem {
|
func (d *opfDocument) itemByID() map[string]opfItem {
|
||||||
m := make(map[string]opfItem, len(d.Manifest.Items))
|
m := make(map[string]opfItem, len(d.Manifest.Items))
|
||||||
@@ -166,6 +273,27 @@ func (d *opfDocument) coverPageHref() string {
|
|||||||
return ""
|
return ""
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// isISBNLike reports whether a bare identifier value is shaped like an ISBN
|
||||||
|
// (digits, optional hyphens/spaces, optional trailing X; 10 or 13
|
||||||
|
// significant characters). Guards the scheme-less dc:identifier fallback
|
||||||
|
// against URLs and UUIDs sharing the same slot.
|
||||||
|
func isISBNLike(v string) bool {
|
||||||
|
digits := 0
|
||||||
|
for i, r := range v {
|
||||||
|
switch {
|
||||||
|
case r >= '0' && r <= '9':
|
||||||
|
digits++
|
||||||
|
case r == '-' || r == ' ':
|
||||||
|
// separator
|
||||||
|
case (r == 'X' || r == 'x') && i == len(v)-1:
|
||||||
|
digits++ // ISBN-10 check character
|
||||||
|
default:
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return digits == 10 || digits == 13
|
||||||
|
}
|
||||||
|
|
||||||
// findImageReferenceInPage extracts the first raster image reference from a
|
// findImageReferenceInPage extracts the first raster image reference from a
|
||||||
// cover (X)HTML page: <img src="..."> or SVG <image xlink:href="...">.
|
// cover (X)HTML page: <img src="..."> or SVG <image xlink:href="...">.
|
||||||
// Token-based parsing keeps it tolerant of mixed namespaces and fragments.
|
// Token-based parsing keeps it tolerant of mixed namespaces and fragments.
|
||||||
|
|||||||
@@ -145,6 +145,65 @@ func TestFindCoverInOPFImageFirstSpine(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestParseOPFContentTitleTypeAndSeries covers EPUB3 refines-based title
|
||||||
|
// selection (main + subtitle joined Calibre-style) and belongs-to-collection
|
||||||
|
// series with collection-type and group-position refines.
|
||||||
|
func TestParseOPFContentTitleTypeAndSeries(t *testing.T) {
|
||||||
|
opf := `<?xml version="1.0"?>
|
||||||
|
<package xmlns="http://www.idpf.org/2007/opf" version="3.0" unique-identifier="pub-id">
|
||||||
|
<metadata xmlns:dc="http://purl.org/dc/elements/1.1/">
|
||||||
|
<dc:title id="t1">The Main Title</dc:title>
|
||||||
|
<dc:title id="t2">The Subtitle</dc:title>
|
||||||
|
<meta refines="#t1" property="title-type">main</meta>
|
||||||
|
<meta refines="#t2" property="title-type">subtitle</meta>
|
||||||
|
<dc:subject>Programming</dc:subject>
|
||||||
|
<dc:subject>Algorithms</dc:subject>
|
||||||
|
<dc:identifier>urn:isbn:978-3-16-148410-0</dc:identifier>
|
||||||
|
<meta id="coll1" property="belongs-to-collection">Great Series</meta>
|
||||||
|
<meta refines="#coll1" property="collection-type">series</meta>
|
||||||
|
<meta refines="#coll1" property="group-position">4.5</meta>
|
||||||
|
</metadata>
|
||||||
|
</package>`
|
||||||
|
metadata, err := parseOPFContent([]byte(opf))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parseOPFContent() error: %v", err)
|
||||||
|
}
|
||||||
|
if want := "The Main Title: The Subtitle"; metadata.Title != want {
|
||||||
|
t.Errorf("Title = %q, want %q", metadata.Title, want)
|
||||||
|
}
|
||||||
|
if metadata.Series != "Great Series" || metadata.SeriesNumber != 4 {
|
||||||
|
t.Errorf("Series = %q/%d, want Great Series/4", metadata.Series, metadata.SeriesNumber)
|
||||||
|
}
|
||||||
|
if metadata.ISBN == "" {
|
||||||
|
t.Error("urn:isbn: identifier not extracted")
|
||||||
|
}
|
||||||
|
if metadata.Genre != "Programming" {
|
||||||
|
t.Errorf("Genre = %q, want first subject %q", metadata.Genre, "Programming")
|
||||||
|
}
|
||||||
|
if len(metadata.Tags) != 2 {
|
||||||
|
t.Errorf("Tags = %v, want both subjects", metadata.Tags)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestParseOPFContentSubjectsWithCommas verifies subject headings keep their
|
||||||
|
// embedded commas as single tags (Library of Congress style headings).
|
||||||
|
func TestParseOPFContentSubjectsWithCommas(t *testing.T) {
|
||||||
|
opf := `<?xml version="1.0"?>
|
||||||
|
<package xmlns="http://www.idpf.org/2007/opf" version="2.0">
|
||||||
|
<metadata xmlns:dc="http://purl.org/dc/elements/1.1/">
|
||||||
|
<dc:title>A Study in Scarlet</dc:title>
|
||||||
|
<dc:subject>Holmes, Sherlock (Fictitious character) -- Fiction</dc:subject>
|
||||||
|
</metadata>
|
||||||
|
</package>`
|
||||||
|
metadata, err := parseOPFContent([]byte(opf))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parseOPFContent() error: %v", err)
|
||||||
|
}
|
||||||
|
if len(metadata.Tags) != 1 {
|
||||||
|
t.Errorf("Tags = %v, want exactly 1 unsplit subject heading", metadata.Tags)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestResolveOPFPath checks URL decoding and posix normalization of
|
// TestResolveOPFPath checks URL decoding and posix normalization of
|
||||||
// OPF-relative hrefs.
|
// OPF-relative hrefs.
|
||||||
func TestResolveOPFPath(t *testing.T) {
|
func TestResolveOPFPath(t *testing.T) {
|
||||||
|
|||||||
Reference in New Issue
Block a user