Progress submissions now carry (percentage, context_text, epubcfi) and the server becomes the position authority: - VerifyProgressAnchor resolves the submitted standard CFI against the book's own XHTML, extracts the text at the anchor, and cross-checks it with the submitted context_text. A mismatch heals the anchor by text search (percentage disambiguates repeats) instead of storing a bad position. - The anchor's block element is derived as a cssSelector plus a block- relative character offset, and served on progress GET alongside the anchor document's href — readium-native handles that let clients re-open a book without parsing CFIs themselves. - context_text-only submissions (no CFI — the dumb-client tier) are anchored structurally from the context text. Motivation: cross-client progress sync (web foliate CFIs, KOReader CRE xpointers, readium-native apps) previously trusted each client's own locator math; the app's EPUB restore drifted ±pages because readium's paginator does not lay out far-from-viewport columns and the foliate- ported CFI walk ran against readium's mutated WebView DOM. Server-side verification heals both classes at ingest.
316 lines
9.8 KiB
Go
316 lines
9.8 KiB
Go
package sync
|
|
|
|
// Ingest-side position authority for reading progress.
|
|
//
|
|
// Clients submit (percentage, context_text, epubcfi). The epubcfi is a
|
|
// standard wrapped CFI — epubcfi(/6/N!/…) — resolvable against the same
|
|
// document this package parses, so every submission is INDEPENDENTLY
|
|
// verified: the text at the resolved anchor is extracted and compared
|
|
// with the submitted context. A mismatch (or an unresolvable anchor)
|
|
// heals the position by text search, with the submitted percentage
|
|
// disambiguating repeated phrases, instead of trusting a client-side
|
|
// projection. This keeps buggy clients from poisoning stored positions:
|
|
// a resolver that silently returns "wherever I'm currently scrolled"
|
|
// fails the context check and gets healed to the true location.
|
|
//
|
|
// The anchor's block element is also derived as a cssSelector plus a
|
|
// block-relative character offset and served back — the readium-native
|
|
// handle clients scroll to, so they never have to parse or trust CFIs
|
|
// themselves.
|
|
//
|
|
// Note: readium-based clients number their readingOrder excluding
|
|
// linear="no" spine items, while this package's spine index follows the
|
|
// OPF spine as written. The spine index is therefore internal-only;
|
|
// client-facing responses carry the anchor document's href instead.
|
|
|
|
import (
|
|
"fmt"
|
|
"strings"
|
|
|
|
"golang.org/x/net/html"
|
|
)
|
|
|
|
// IsConvertibleFormat reports whether a format group uses standard CFIs
|
|
// as its structural locator currency (i.e. whether CFI verification and
|
|
// cssSelector derivation apply to it).
|
|
func IsConvertibleFormat(formatGroup string) bool {
|
|
return isConvertible(string(formatGroup))
|
|
}
|
|
|
|
func truncateRunes(s string, n int) string {
|
|
r := []rune(s)
|
|
if len(r) <= n {
|
|
return s
|
|
}
|
|
return string(r[:n])
|
|
}
|
|
|
|
// anchorBlock returns the nearest non-inline (block-level) ancestor of a
|
|
// resolved text/element node — the server-side equivalent of the
|
|
// readers' computed-style block walk.
|
|
func anchorBlock(node *html.Node) *html.Node {
|
|
if node == nil {
|
|
return nil
|
|
}
|
|
if node.Type == html.ElementNode && !isInlineFormatting(node) {
|
|
return node
|
|
}
|
|
return findBlockParent(node)
|
|
}
|
|
|
|
// blockContextText extracts the normalized text from (node, runeOff) to
|
|
// the end of the anchor block — the same excerpt rule the readers use for
|
|
// context_text, ≤100 chars.
|
|
func blockContextText(block, node *html.Node, runeOff int) string {
|
|
segments := collectInlineText(block)
|
|
var sb strings.Builder
|
|
started := false
|
|
for _, seg := range segments {
|
|
if !started && seg.node == node {
|
|
started = true
|
|
runes := []rune(string(seg.runes))
|
|
if runeOff < len(runes) {
|
|
sb.WriteString(string(runes[runeOff:]))
|
|
}
|
|
continue
|
|
}
|
|
if started {
|
|
sb.WriteString(string(seg.runes))
|
|
}
|
|
}
|
|
return truncateRunes(normalizeWhitespace(sb.String()), 100)
|
|
}
|
|
|
|
// contextMatches reports whether a server-extracted context and a client-
|
|
// submitted context describe the same anchor. Both are suffixes of the
|
|
// same block text when the anchors share a block, so containment in
|
|
// either direction verifies; empty or very short contexts never match.
|
|
func contextMatches(serverCtx, submitted string) bool {
|
|
s := truncateRunes(normalizeWhitespace(submitted), 100)
|
|
t := truncateRunes(normalizeWhitespace(serverCtx), 100)
|
|
if s == "" || t == "" {
|
|
return false
|
|
}
|
|
short, long := s, t
|
|
if len([]rune(short)) > len([]rune(long)) {
|
|
short, long = long, short
|
|
}
|
|
if len([]rune(short)) < 12 {
|
|
return false
|
|
}
|
|
return strings.Contains(long, short)
|
|
}
|
|
|
|
// cssSelectorFor mirrors the readers' selOf: a body-relative
|
|
// tag:nth-child(k) chain (k = 1-based position among element siblings).
|
|
func cssSelectorFor(block *html.Node) string {
|
|
var segs []string
|
|
n := block
|
|
for n != nil && n.Type == html.ElementNode && n.Data != "body" {
|
|
k := 1
|
|
sib := n.Parent.FirstChild
|
|
for sib != nil && sib != n {
|
|
if sib.Type == html.ElementNode {
|
|
k++
|
|
}
|
|
sib = sib.NextSibling
|
|
}
|
|
segs = append([]string{n.Data + ":nth-child(" + fmt.Sprintf("%d", k) + ")"}, segs...)
|
|
n = n.Parent
|
|
}
|
|
return "body>" + strings.Join(segs, ">")
|
|
}
|
|
|
|
// blockCharOffset computes the rune offset of (node, runeOff) within the
|
|
// concatenated text of its block — the client-side scroll target.
|
|
func blockCharOffset(block, node *html.Node, runeOff int) int {
|
|
segments := collectInlineText(block)
|
|
s := 0
|
|
for _, seg := range segments {
|
|
if seg.node == node {
|
|
return s + runeOff
|
|
}
|
|
s += len(seg.runes)
|
|
}
|
|
return s + runeOff
|
|
}
|
|
|
|
// ProgressAnchor is the full server-computed apply handle for a stored
|
|
// standard CFI: the anchor block's cssSelector, the character offset
|
|
// within that block's text, and the spine document's href.
|
|
type ProgressAnchor struct {
|
|
CSSSelector string
|
|
CharOffset int
|
|
Href string
|
|
HealedCFI string
|
|
Healed bool
|
|
HealedPct *float64
|
|
}
|
|
|
|
// VerifyProgressAnchor resolves a client-submitted standard CFI against
|
|
// the EPUB, cross-checks the submitted context text, and heals the anchor
|
|
// by text search on any mismatch.
|
|
func VerifyProgressAnchor(epubPath, epubcfi, contextText string, percentage float64) (finalCFI string, cssSelector string, anchorHref string, charOffset *int, healedPct *float64, healed bool, err error) {
|
|
finalCFI = epubcfi
|
|
anchorHref = ""
|
|
|
|
spineIndex, localSteps, err := parseEPUBCFI(epubcfi)
|
|
if err != nil {
|
|
cfi, sel, href, off, pct, healedFlag, herr := healFromContext(epubPath, contextText, percentage)
|
|
return cfi, sel, href, off, pct, healedFlag, herr
|
|
}
|
|
conv := cachedConverter(epubPath)
|
|
doc, docHref, err := conv.getContentDoc(spineIndex + 1)
|
|
if err != nil {
|
|
return healFromContext(epubPath, contextText, percentage)
|
|
}
|
|
node, runeOff, rerr := resolveCFIToNode(doc, localSteps)
|
|
if rerr != nil {
|
|
return healFromContext(epubPath, contextText, percentage)
|
|
}
|
|
anchorHref = docHref
|
|
|
|
block := anchorBlock(node)
|
|
serverCtx := blockContextText(block, node, runeOff)
|
|
if contextMatches(serverCtx, contextText) {
|
|
off := blockCharOffset(block, node, runeOff)
|
|
return finalCFI, cssSelectorFor(block), anchorHref, &off, nil, false, nil
|
|
}
|
|
|
|
// Mismatch: heal by text search.
|
|
hCFI, hPct, hSel, hHref, herr := healAnchorByText(epubPath, contextText, percentage, spineIndex)
|
|
if herr != nil {
|
|
return finalCFI, "", hHref, nil, nil, false, fmt.Errorf("context mismatch (server %q vs client %q) and heal failed: %w",
|
|
truncateRunes(serverCtx, 40), truncateRunes(contextText, 40), herr)
|
|
}
|
|
hOff := blockCharOffsetFor(epubPath, hCFI)
|
|
return hCFI, hSel, hHref, &hOff, &hPct, true, nil
|
|
}
|
|
|
|
func healFromContext(epubPath, contextText string, percentage float64) (string, string, string, *int, *float64, bool, error) {
|
|
cfi, healedPct, sel, href, err := healAnchorByText(epubPath, contextText, percentage, -1)
|
|
if err != nil {
|
|
return "", "", "", nil, nil, false, err
|
|
}
|
|
off := blockCharOffsetFor(epubPath, cfi)
|
|
return cfi, sel, href, &off, &healedPct, true, nil
|
|
}
|
|
|
|
func healAnchorByText(epubPath, contextText string, percentage float64, spineIndex int) (string, float64, string, string, error) {
|
|
if epubPath == "" {
|
|
return "", 0, "", "", fmt.Errorf("no epub available for text anchoring")
|
|
}
|
|
conv := cachedConverter(epubPath)
|
|
spine, err := conv.loadSpine()
|
|
if err != nil {
|
|
return "", 0, "", "", err
|
|
}
|
|
needle := truncateRunes(normalizeWhitespace(contextText), 40)
|
|
if len([]rune(needle)) < 12 {
|
|
return "", 0, "", "", fmt.Errorf("context too short to anchor")
|
|
}
|
|
|
|
type match struct {
|
|
spine int
|
|
node *html.Node
|
|
off int
|
|
}
|
|
var matches []match
|
|
totalChars := 0
|
|
charsBefore := make([]int, len(spine.items))
|
|
for i := range spine.items {
|
|
doc, _, derr := conv.getContentDoc(i + 1)
|
|
if derr != nil {
|
|
continue
|
|
}
|
|
body := findBody(doc)
|
|
if body == nil {
|
|
continue
|
|
}
|
|
charsBefore[i] = totalChars
|
|
totalChars += countTextChars(body)
|
|
if n, runeOff := findTextInNode(body, needle); n != nil {
|
|
matches = append(matches, match{spine: i, node: n, off: runeOff})
|
|
}
|
|
}
|
|
if len(matches) == 0 || totalChars <= 0 {
|
|
return "", 0, "", "", fmt.Errorf("context not found in book")
|
|
}
|
|
|
|
best := matches[0]
|
|
bestDist := -1.0
|
|
for _, m := range matches {
|
|
frac := (float64(charsBefore[m.spine]) + float64(countTextCharsBefore(m.node)+m.off)) / float64(totalChars)
|
|
d := frac - percentage
|
|
if d < 0 {
|
|
d = -d
|
|
}
|
|
if bestDist < 0 || d < bestDist {
|
|
best = m
|
|
bestDist = d
|
|
}
|
|
}
|
|
|
|
healedPct := (float64(charsBefore[best.spine]) + float64(countTextCharsBefore(best.node)+best.off)) / float64(totalChars)
|
|
cfi, err := buildCFI(best.spine, best.node, best.off)
|
|
if err != nil {
|
|
return "", 0, "", "", err
|
|
}
|
|
sel := ""
|
|
if block := anchorBlock(best.node); block != nil {
|
|
sel = cssSelectorFor(block)
|
|
}
|
|
return cfi, healedPct, sel, spine.items[best.spine].href, nil
|
|
}
|
|
|
|
func blockCharOffsetFor(epubPath, cfi string) int {
|
|
n, off, _, herr := cfiTextAtAnchorInternal(epubPath, cfi)
|
|
if herr != nil {
|
|
return 0
|
|
}
|
|
block := anchorBlock(n)
|
|
if block == nil {
|
|
return 0
|
|
}
|
|
return blockCharOffset(block, n, off)
|
|
}
|
|
|
|
// cfiTextAtAnchorInternal resolves a stored standard CFI to its anchor
|
|
// text node, rune offset, and containing document.
|
|
func cfiTextAtAnchorInternal(epubPath, cfi string) (*html.Node, int, *html.Node, error) {
|
|
spineIndex, localSteps, err := parseEPUBCFI(cfi)
|
|
if err != nil {
|
|
return nil, 0, nil, err
|
|
}
|
|
conv := cachedConverter(epubPath)
|
|
doc, _, err := conv.getContentDoc(spineIndex + 1)
|
|
if err != nil {
|
|
return nil, 0, nil, err
|
|
}
|
|
node, off, err := resolveCFIToNode(doc, localSteps)
|
|
return node, off, doc, err
|
|
}
|
|
|
|
// ProgressCSSSelector derives the anchor block's cssSelector from a stored
|
|
// standard CFI.
|
|
func ProgressCSSSelector(epubPath, epubcfi string) (string, error) {
|
|
spineIndex, localSteps, err := parseEPUBCFI(epubcfi)
|
|
if err != nil {
|
|
return "", err
|
|
}
|
|
conv := cachedConverter(epubPath)
|
|
doc, _, err := conv.getContentDoc(spineIndex + 1)
|
|
if err != nil {
|
|
return "", err
|
|
}
|
|
node, _, err := resolveCFIToNode(doc, localSteps)
|
|
if err != nil {
|
|
return "", err
|
|
}
|
|
block := anchorBlock(node)
|
|
if block == nil {
|
|
return "", fmt.Errorf("no block ancestor for anchor")
|
|
}
|
|
return cssSelectorFor(block), nil
|
|
}
|