feat(sync): server-side position authority — verify/heal progress anchors
Progress submissions now carry (percentage, context_text, epubcfi) and the server becomes the position authority: - VerifyProgressAnchor resolves the submitted standard CFI against the book's own XHTML, extracts the text at the anchor, and cross-checks it with the submitted context_text. A mismatch heals the anchor by text search (percentage disambiguates repeats) instead of storing a bad position. - The anchor's block element is derived as a cssSelector plus a block- relative character offset, and served on progress GET alongside the anchor document's href — readium-native handles that let clients re-open a book without parsing CFIs themselves. - context_text-only submissions (no CFI — the dumb-client tier) are anchored structurally from the context text. Motivation: cross-client progress sync (web foliate CFIs, KOReader CRE xpointers, readium-native apps) previously trusted each client's own locator math; the app's EPUB restore drifted ±pages because readium's paginator does not lay out far-from-viewport columns and the foliate- ported CFI walk ran against readium's mutated WebView DOM. Server-side verification heals both classes at ingest.
This commit is contained in:
@@ -0,0 +1,315 @@
|
||||
package sync
|
||||
|
||||
// Ingest-side position authority for reading progress.
|
||||
//
|
||||
// Clients submit (percentage, context_text, epubcfi). The epubcfi is a
|
||||
// standard wrapped CFI — epubcfi(/6/N!/…) — resolvable against the same
|
||||
// document this package parses, so every submission is INDEPENDENTLY
|
||||
// verified: the text at the resolved anchor is extracted and compared
|
||||
// with the submitted context. A mismatch (or an unresolvable anchor)
|
||||
// heals the position by text search, with the submitted percentage
|
||||
// disambiguating repeated phrases, instead of trusting a client-side
|
||||
// projection. This keeps buggy clients from poisoning stored positions:
|
||||
// a resolver that silently returns "wherever I'm currently scrolled"
|
||||
// fails the context check and gets healed to the true location.
|
||||
//
|
||||
// The anchor's block element is also derived as a cssSelector plus a
|
||||
// block-relative character offset and served back — the readium-native
|
||||
// handle clients scroll to, so they never have to parse or trust CFIs
|
||||
// themselves.
|
||||
//
|
||||
// Note: readium-based clients number their readingOrder excluding
|
||||
// linear="no" spine items, while this package's spine index follows the
|
||||
// OPF spine as written. The spine index is therefore internal-only;
|
||||
// client-facing responses carry the anchor document's href instead.
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
"golang.org/x/net/html"
|
||||
)
|
||||
|
||||
// IsConvertibleFormat reports whether a format group uses standard CFIs
|
||||
// as its structural locator currency (i.e. whether CFI verification and
|
||||
// cssSelector derivation apply to it).
|
||||
func IsConvertibleFormat(formatGroup string) bool {
|
||||
return isConvertible(string(formatGroup))
|
||||
}
|
||||
|
||||
func truncateRunes(s string, n int) string {
|
||||
r := []rune(s)
|
||||
if len(r) <= n {
|
||||
return s
|
||||
}
|
||||
return string(r[:n])
|
||||
}
|
||||
|
||||
// anchorBlock returns the nearest non-inline (block-level) ancestor of a
|
||||
// resolved text/element node — the server-side equivalent of the
|
||||
// readers' computed-style block walk.
|
||||
func anchorBlock(node *html.Node) *html.Node {
|
||||
if node == nil {
|
||||
return nil
|
||||
}
|
||||
if node.Type == html.ElementNode && !isInlineFormatting(node) {
|
||||
return node
|
||||
}
|
||||
return findBlockParent(node)
|
||||
}
|
||||
|
||||
// blockContextText extracts the normalized text from (node, runeOff) to
|
||||
// the end of the anchor block — the same excerpt rule the readers use for
|
||||
// context_text, ≤100 chars.
|
||||
func blockContextText(block, node *html.Node, runeOff int) string {
|
||||
segments := collectInlineText(block)
|
||||
var sb strings.Builder
|
||||
started := false
|
||||
for _, seg := range segments {
|
||||
if !started && seg.node == node {
|
||||
started = true
|
||||
runes := []rune(string(seg.runes))
|
||||
if runeOff < len(runes) {
|
||||
sb.WriteString(string(runes[runeOff:]))
|
||||
}
|
||||
continue
|
||||
}
|
||||
if started {
|
||||
sb.WriteString(string(seg.runes))
|
||||
}
|
||||
}
|
||||
return truncateRunes(normalizeWhitespace(sb.String()), 100)
|
||||
}
|
||||
|
||||
// contextMatches reports whether a server-extracted context and a client-
|
||||
// submitted context describe the same anchor. Both are suffixes of the
|
||||
// same block text when the anchors share a block, so containment in
|
||||
// either direction verifies; empty or very short contexts never match.
|
||||
func contextMatches(serverCtx, submitted string) bool {
|
||||
s := truncateRunes(normalizeWhitespace(submitted), 100)
|
||||
t := truncateRunes(normalizeWhitespace(serverCtx), 100)
|
||||
if s == "" || t == "" {
|
||||
return false
|
||||
}
|
||||
short, long := s, t
|
||||
if len([]rune(short)) > len([]rune(long)) {
|
||||
short, long = long, short
|
||||
}
|
||||
if len([]rune(short)) < 12 {
|
||||
return false
|
||||
}
|
||||
return strings.Contains(long, short)
|
||||
}
|
||||
|
||||
// cssSelectorFor mirrors the readers' selOf: a body-relative
|
||||
// tag:nth-child(k) chain (k = 1-based position among element siblings).
|
||||
func cssSelectorFor(block *html.Node) string {
|
||||
var segs []string
|
||||
n := block
|
||||
for n != nil && n.Type == html.ElementNode && n.Data != "body" {
|
||||
k := 1
|
||||
sib := n.Parent.FirstChild
|
||||
for sib != nil && sib != n {
|
||||
if sib.Type == html.ElementNode {
|
||||
k++
|
||||
}
|
||||
sib = sib.NextSibling
|
||||
}
|
||||
segs = append([]string{n.Data + ":nth-child(" + fmt.Sprintf("%d", k) + ")"}, segs...)
|
||||
n = n.Parent
|
||||
}
|
||||
return "body>" + strings.Join(segs, ">")
|
||||
}
|
||||
|
||||
// blockCharOffset computes the rune offset of (node, runeOff) within the
|
||||
// concatenated text of its block — the client-side scroll target.
|
||||
func blockCharOffset(block, node *html.Node, runeOff int) int {
|
||||
segments := collectInlineText(block)
|
||||
s := 0
|
||||
for _, seg := range segments {
|
||||
if seg.node == node {
|
||||
return s + runeOff
|
||||
}
|
||||
s += len(seg.runes)
|
||||
}
|
||||
return s + runeOff
|
||||
}
|
||||
|
||||
// ProgressAnchor is the full server-computed apply handle for a stored
|
||||
// standard CFI: the anchor block's cssSelector, the character offset
|
||||
// within that block's text, and the spine document's href.
|
||||
type ProgressAnchor struct {
|
||||
CSSSelector string
|
||||
CharOffset int
|
||||
Href string
|
||||
HealedCFI string
|
||||
Healed bool
|
||||
HealedPct *float64
|
||||
}
|
||||
|
||||
// VerifyProgressAnchor resolves a client-submitted standard CFI against
|
||||
// the EPUB, cross-checks the submitted context text, and heals the anchor
|
||||
// by text search on any mismatch.
|
||||
func VerifyProgressAnchor(epubPath, epubcfi, contextText string, percentage float64) (finalCFI string, cssSelector string, anchorHref string, charOffset *int, healedPct *float64, healed bool, err error) {
|
||||
finalCFI = epubcfi
|
||||
anchorHref = ""
|
||||
|
||||
spineIndex, localSteps, err := parseEPUBCFI(epubcfi)
|
||||
if err != nil {
|
||||
cfi, sel, href, off, pct, healedFlag, herr := healFromContext(epubPath, contextText, percentage)
|
||||
return cfi, sel, href, off, pct, healedFlag, herr
|
||||
}
|
||||
conv := cachedConverter(epubPath)
|
||||
doc, docHref, err := conv.getContentDoc(spineIndex + 1)
|
||||
if err != nil {
|
||||
return healFromContext(epubPath, contextText, percentage)
|
||||
}
|
||||
node, runeOff, rerr := resolveCFIToNode(doc, localSteps)
|
||||
if rerr != nil {
|
||||
return healFromContext(epubPath, contextText, percentage)
|
||||
}
|
||||
anchorHref = docHref
|
||||
|
||||
block := anchorBlock(node)
|
||||
serverCtx := blockContextText(block, node, runeOff)
|
||||
if contextMatches(serverCtx, contextText) {
|
||||
off := blockCharOffset(block, node, runeOff)
|
||||
return finalCFI, cssSelectorFor(block), anchorHref, &off, nil, false, nil
|
||||
}
|
||||
|
||||
// Mismatch: heal by text search.
|
||||
hCFI, hPct, hSel, hHref, herr := healAnchorByText(epubPath, contextText, percentage, spineIndex)
|
||||
if herr != nil {
|
||||
return finalCFI, "", hHref, nil, nil, false, fmt.Errorf("context mismatch (server %q vs client %q) and heal failed: %w",
|
||||
truncateRunes(serverCtx, 40), truncateRunes(contextText, 40), herr)
|
||||
}
|
||||
hOff := blockCharOffsetFor(epubPath, hCFI)
|
||||
return hCFI, hSel, hHref, &hOff, &hPct, true, nil
|
||||
}
|
||||
|
||||
func healFromContext(epubPath, contextText string, percentage float64) (string, string, string, *int, *float64, bool, error) {
|
||||
cfi, healedPct, sel, href, err := healAnchorByText(epubPath, contextText, percentage, -1)
|
||||
if err != nil {
|
||||
return "", "", "", nil, nil, false, err
|
||||
}
|
||||
off := blockCharOffsetFor(epubPath, cfi)
|
||||
return cfi, sel, href, &off, &healedPct, true, nil
|
||||
}
|
||||
|
||||
func healAnchorByText(epubPath, contextText string, percentage float64, spineIndex int) (string, float64, string, string, error) {
|
||||
if epubPath == "" {
|
||||
return "", 0, "", "", fmt.Errorf("no epub available for text anchoring")
|
||||
}
|
||||
conv := cachedConverter(epubPath)
|
||||
spine, err := conv.loadSpine()
|
||||
if err != nil {
|
||||
return "", 0, "", "", err
|
||||
}
|
||||
needle := truncateRunes(normalizeWhitespace(contextText), 40)
|
||||
if len([]rune(needle)) < 12 {
|
||||
return "", 0, "", "", fmt.Errorf("context too short to anchor")
|
||||
}
|
||||
|
||||
type match struct {
|
||||
spine int
|
||||
node *html.Node
|
||||
off int
|
||||
}
|
||||
var matches []match
|
||||
totalChars := 0
|
||||
charsBefore := make([]int, len(spine.items))
|
||||
for i := range spine.items {
|
||||
doc, _, derr := conv.getContentDoc(i + 1)
|
||||
if derr != nil {
|
||||
continue
|
||||
}
|
||||
body := findBody(doc)
|
||||
if body == nil {
|
||||
continue
|
||||
}
|
||||
charsBefore[i] = totalChars
|
||||
totalChars += countTextChars(body)
|
||||
if n, runeOff := findTextInNode(body, needle); n != nil {
|
||||
matches = append(matches, match{spine: i, node: n, off: runeOff})
|
||||
}
|
||||
}
|
||||
if len(matches) == 0 || totalChars <= 0 {
|
||||
return "", 0, "", "", fmt.Errorf("context not found in book")
|
||||
}
|
||||
|
||||
best := matches[0]
|
||||
bestDist := -1.0
|
||||
for _, m := range matches {
|
||||
frac := (float64(charsBefore[m.spine]) + float64(countTextCharsBefore(m.node)+m.off)) / float64(totalChars)
|
||||
d := frac - percentage
|
||||
if d < 0 {
|
||||
d = -d
|
||||
}
|
||||
if bestDist < 0 || d < bestDist {
|
||||
best = m
|
||||
bestDist = d
|
||||
}
|
||||
}
|
||||
|
||||
healedPct := (float64(charsBefore[best.spine]) + float64(countTextCharsBefore(best.node)+best.off)) / float64(totalChars)
|
||||
cfi, err := buildCFI(best.spine, best.node, best.off)
|
||||
if err != nil {
|
||||
return "", 0, "", "", err
|
||||
}
|
||||
sel := ""
|
||||
if block := anchorBlock(best.node); block != nil {
|
||||
sel = cssSelectorFor(block)
|
||||
}
|
||||
return cfi, healedPct, sel, spine.items[best.spine].href, nil
|
||||
}
|
||||
|
||||
func blockCharOffsetFor(epubPath, cfi string) int {
|
||||
n, off, _, herr := cfiTextAtAnchorInternal(epubPath, cfi)
|
||||
if herr != nil {
|
||||
return 0
|
||||
}
|
||||
block := anchorBlock(n)
|
||||
if block == nil {
|
||||
return 0
|
||||
}
|
||||
return blockCharOffset(block, n, off)
|
||||
}
|
||||
|
||||
// cfiTextAtAnchorInternal resolves a stored standard CFI to its anchor
|
||||
// text node, rune offset, and containing document.
|
||||
func cfiTextAtAnchorInternal(epubPath, cfi string) (*html.Node, int, *html.Node, error) {
|
||||
spineIndex, localSteps, err := parseEPUBCFI(cfi)
|
||||
if err != nil {
|
||||
return nil, 0, nil, err
|
||||
}
|
||||
conv := cachedConverter(epubPath)
|
||||
doc, _, err := conv.getContentDoc(spineIndex + 1)
|
||||
if err != nil {
|
||||
return nil, 0, nil, err
|
||||
}
|
||||
node, off, err := resolveCFIToNode(doc, localSteps)
|
||||
return node, off, doc, err
|
||||
}
|
||||
|
||||
// ProgressCSSSelector derives the anchor block's cssSelector from a stored
|
||||
// standard CFI.
|
||||
func ProgressCSSSelector(epubPath, epubcfi string) (string, error) {
|
||||
spineIndex, localSteps, err := parseEPUBCFI(epubcfi)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
conv := cachedConverter(epubPath)
|
||||
doc, _, err := conv.getContentDoc(spineIndex + 1)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
node, _, err := resolveCFIToNode(doc, localSteps)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
block := anchorBlock(node)
|
||||
if block == nil {
|
||||
return "", fmt.Errorf("no block ancestor for anchor")
|
||||
}
|
||||
return cssSelectorFor(block), nil
|
||||
}
|
||||
Reference in New Issue
Block a user