package sync // Ingest-side position authority for reading progress. // // Clients submit (percentage, context_text, epubcfi). The epubcfi is a // standard wrapped CFI — epubcfi(/6/N!/…) — resolvable against the same // document this package parses, so every submission is INDEPENDENTLY // verified: the text at the resolved anchor is extracted and compared // with the submitted context. A mismatch (or an unresolvable anchor) // heals the position by text search, with the submitted percentage // disambiguating repeated phrases, instead of trusting a client-side // projection. This keeps buggy clients from poisoning stored positions: // a resolver that silently returns "wherever I'm currently scrolled" // fails the context check and gets healed to the true location. // // The anchor's block element is also derived as a cssSelector plus a // block-relative character offset and served back — the readium-native // handle clients scroll to, so they never have to parse or trust CFIs // themselves. // // Note: readium-based clients number their readingOrder excluding // linear="no" spine items, while this package's spine index follows the // OPF spine as written. The spine index is therefore internal-only; // client-facing responses carry the anchor document's href instead. import ( "fmt" "strings" "golang.org/x/net/html" ) // IsConvertibleFormat reports whether a format group uses standard CFIs // as its structural locator currency (i.e. whether CFI verification and // cssSelector derivation apply to it). func IsConvertibleFormat(formatGroup string) bool { return isConvertible(string(formatGroup)) } func truncateRunes(s string, n int) string { r := []rune(s) if len(r) <= n { return s } return string(r[:n]) } // anchorBlock returns the nearest non-inline (block-level) ancestor of a // resolved text/element node — the server-side equivalent of the // readers' computed-style block walk. func anchorBlock(node *html.Node) *html.Node { if node == nil { return nil } if node.Type == html.ElementNode && !isInlineFormatting(node) { return node } return findBlockParent(node) } // blockContextText extracts the normalized text from (node, runeOff) to // the end of the anchor block — the same excerpt rule the readers use for // context_text, ≤100 chars. func blockContextText(block, node *html.Node, runeOff int) string { segments := collectInlineText(block) var sb strings.Builder started := false for _, seg := range segments { if !started && seg.node == node { started = true runes := []rune(string(seg.runes)) if runeOff < len(runes) { sb.WriteString(string(runes[runeOff:])) } continue } if started { sb.WriteString(string(seg.runes)) } } return truncateRunes(normalizeWhitespace(sb.String()), 100) } // contextMatches reports whether a server-extracted context and a client- // submitted context describe the same anchor. Both are suffixes of the // same block text when the anchors share a block, so containment in // either direction verifies; empty or very short contexts never match. func contextMatches(serverCtx, submitted string) bool { s := truncateRunes(normalizeWhitespace(submitted), 100) t := truncateRunes(normalizeWhitespace(serverCtx), 100) if s == "" || t == "" { return false } short, long := s, t if len([]rune(short)) > len([]rune(long)) { short, long = long, short } if len([]rune(short)) < 12 { return false } return strings.Contains(long, short) } // cssSelectorFor mirrors the readers' selOf: a body-relative // tag:nth-child(k) chain (k = 1-based position among element siblings). func cssSelectorFor(block *html.Node) string { var segs []string n := block for n != nil && n.Type == html.ElementNode && n.Data != "body" { k := 1 sib := n.Parent.FirstChild for sib != nil && sib != n { if sib.Type == html.ElementNode { k++ } sib = sib.NextSibling } segs = append([]string{n.Data + ":nth-child(" + fmt.Sprintf("%d", k) + ")"}, segs...) n = n.Parent } return "body>" + strings.Join(segs, ">") } // blockCharOffset computes the rune offset of (node, runeOff) within the // concatenated text of its block — the client-side scroll target. func blockCharOffset(block, node *html.Node, runeOff int) int { segments := collectInlineText(block) s := 0 for _, seg := range segments { if seg.node == node { return s + runeOff } s += len(seg.runes) } return s + runeOff } // ProgressAnchor is the full server-computed apply handle for a stored // standard CFI: the anchor block's cssSelector, the character offset // within that block's text, and the spine document's href. type ProgressAnchor struct { CSSSelector string CharOffset int Href string HealedCFI string Healed bool HealedPct *float64 } // VerifyProgressAnchor resolves a client-submitted standard CFI against // the EPUB, cross-checks the submitted context text, and heals the anchor // by text search on any mismatch. func VerifyProgressAnchor(epubPath, epubcfi, contextText string, percentage float64) (finalCFI string, cssSelector string, anchorHref string, charOffset *int, healedPct *float64, healed bool, err error) { finalCFI = epubcfi anchorHref = "" spineIndex, localSteps, err := parseEPUBCFI(epubcfi) if err != nil { cfi, sel, href, off, pct, healedFlag, herr := healFromContext(epubPath, contextText, percentage) return cfi, sel, href, off, pct, healedFlag, herr } conv := cachedConverter(epubPath) doc, docHref, err := conv.getContentDoc(spineIndex + 1) if err != nil { return healFromContext(epubPath, contextText, percentage) } node, runeOff, rerr := resolveCFIToNode(doc, localSteps) if rerr != nil { return healFromContext(epubPath, contextText, percentage) } anchorHref = docHref block := anchorBlock(node) serverCtx := blockContextText(block, node, runeOff) if contextMatches(serverCtx, contextText) { off := blockCharOffset(block, node, runeOff) return finalCFI, cssSelectorFor(block), anchorHref, &off, nil, false, nil } // Mismatch: heal by text search. hCFI, hPct, hSel, hHref, herr := healAnchorByText(epubPath, contextText, percentage, spineIndex) if herr != nil { return finalCFI, "", hHref, nil, nil, false, fmt.Errorf("context mismatch (server %q vs client %q) and heal failed: %w", truncateRunes(serverCtx, 40), truncateRunes(contextText, 40), herr) } hOff := blockCharOffsetFor(epubPath, hCFI) return hCFI, hSel, hHref, &hOff, &hPct, true, nil } func healFromContext(epubPath, contextText string, percentage float64) (string, string, string, *int, *float64, bool, error) { cfi, healedPct, sel, href, err := healAnchorByText(epubPath, contextText, percentage, -1) if err != nil { return "", "", "", nil, nil, false, err } off := blockCharOffsetFor(epubPath, cfi) return cfi, sel, href, &off, &healedPct, true, nil } func healAnchorByText(epubPath, contextText string, percentage float64, spineIndex int) (string, float64, string, string, error) { if epubPath == "" { return "", 0, "", "", fmt.Errorf("no epub available for text anchoring") } conv := cachedConverter(epubPath) spine, err := conv.loadSpine() if err != nil { return "", 0, "", "", err } needle := truncateRunes(normalizeWhitespace(contextText), 40) if len([]rune(needle)) < 12 { return "", 0, "", "", fmt.Errorf("context too short to anchor") } type match struct { spine int node *html.Node off int } var matches []match totalChars := 0 charsBefore := make([]int, len(spine.items)) for i := range spine.items { doc, _, derr := conv.getContentDoc(i + 1) if derr != nil { continue } body := findBody(doc) if body == nil { continue } charsBefore[i] = totalChars totalChars += countTextChars(body) if n, runeOff := findTextInNode(body, needle); n != nil { matches = append(matches, match{spine: i, node: n, off: runeOff}) } } if len(matches) == 0 || totalChars <= 0 { return "", 0, "", "", fmt.Errorf("context not found in book") } best := matches[0] bestDist := -1.0 for _, m := range matches { frac := (float64(charsBefore[m.spine]) + float64(countTextCharsBefore(m.node)+m.off)) / float64(totalChars) d := frac - percentage if d < 0 { d = -d } if bestDist < 0 || d < bestDist { best = m bestDist = d } } healedPct := (float64(charsBefore[best.spine]) + float64(countTextCharsBefore(best.node)+best.off)) / float64(totalChars) cfi, err := buildCFI(best.spine, best.node, best.off) if err != nil { return "", 0, "", "", err } sel := "" if block := anchorBlock(best.node); block != nil { sel = cssSelectorFor(block) } return cfi, healedPct, sel, spine.items[best.spine].href, nil } func blockCharOffsetFor(epubPath, cfi string) int { n, off, _, herr := cfiTextAtAnchorInternal(epubPath, cfi) if herr != nil { return 0 } block := anchorBlock(n) if block == nil { return 0 } return blockCharOffset(block, n, off) } // cfiTextAtAnchorInternal resolves a stored standard CFI to its anchor // text node, rune offset, and containing document. func cfiTextAtAnchorInternal(epubPath, cfi string) (*html.Node, int, *html.Node, error) { spineIndex, localSteps, err := parseEPUBCFI(cfi) if err != nil { return nil, 0, nil, err } conv := cachedConverter(epubPath) doc, _, err := conv.getContentDoc(spineIndex + 1) if err != nil { return nil, 0, nil, err } node, off, err := resolveCFIToNode(doc, localSteps) return node, off, doc, err } // ProgressCSSSelector derives the anchor block's cssSelector from a stored // standard CFI. func ProgressCSSSelector(epubPath, epubcfi string) (string, error) { spineIndex, localSteps, err := parseEPUBCFI(epubcfi) if err != nil { return "", err } conv := cachedConverter(epubPath) doc, _, err := conv.getContentDoc(spineIndex + 1) if err != nil { return "", err } node, _, err := resolveCFIToNode(doc, localSteps) if err != nil { return "", err } block := anchorBlock(node) if block == nil { return "", fmt.Errorf("no block ancestor for anchor") } return cssSelectorFor(block), nil }