Offset currency policy, now explicit: EPUB CFI terminals, CRE text() offsets and the served char_offset handle are UTF-16 code units (the EPUB CFI spec, and what foliate/readium/KOReader/Kobo clients actually observe), while internal arithmetic — the book-wide character_offset column and percentage fractions — stays rune-based, consistent with TotalCharacters. For all-BMP books the currencies are identical, so no stored value changes; astral-plane text (emoji, rare CJK) no longer drifts. Boundaries converted: resolveCFIToNode interprets incoming CFI terminal offsets as UTF-16; textNodeAtUTF16Offset (née textNodeAtRuneOffset) interprets CRE text() offsets as UTF-16; buildCFI and buildCREXPointer emit UTF-16 terminals; blockCharOffset (the served char_offset) is UTF-16. Also fixes two character_offset column defects: heals wrote a BLOCK- relative offset into the book-wide column, and verified-but-unhealed saves (e.g. KOReader pushes) never refreshed it, leaving it stale behind the anchor. VerifyProgressAnchor now returns the verified book- wide rune offset and SaveProgress refreshes the column on every verified save. Tests: astral currency round trip (offset after an emoji must shift by one unit between currencies, in both heal and exact-verify directions) and book-offset ordering. The cmd/server/tests integration harness failures under docker (library folder 400 during setup) reproduce on the pre-change tree and are unrelated.
350 lines
11 KiB
Go
350 lines
11 KiB
Go
package sync
|
|
|
|
// Ingest-side position authority for reading progress.
|
|
//
|
|
// Clients submit (percentage, context_text, epubcfi). The epubcfi is a
|
|
// standard wrapped CFI — epubcfi(/6/N!/…) — resolvable against the same
|
|
// document this package parses, so every submission is INDEPENDENTLY
|
|
// verified: the text at the resolved anchor is extracted and compared
|
|
// with the submitted context. A mismatch (or an unresolvable anchor)
|
|
// heals the position by text search, with the submitted percentage
|
|
// disambiguating repeated phrases, instead of trusting a client-side
|
|
// projection. This keeps buggy clients from poisoning stored positions:
|
|
// a resolver that silently returns "wherever I'm currently scrolled"
|
|
// fails the context check and gets healed to the true location.
|
|
//
|
|
// The anchor's block element is also derived as a cssSelector plus a
|
|
// block-relative character offset and served back — the readium-native
|
|
// handle clients scroll to, so they never have to parse or trust CFIs
|
|
// themselves.
|
|
//
|
|
// Note: readium-based clients number their readingOrder excluding
|
|
// linear="no" spine items, while this package's spine index follows the
|
|
// OPF spine as written. The spine index is therefore internal-only;
|
|
// client-facing responses carry the anchor document's href instead.
|
|
|
|
import (
|
|
"fmt"
|
|
"strings"
|
|
"unicode/utf8"
|
|
|
|
"golang.org/x/net/html"
|
|
)
|
|
|
|
// IsConvertibleFormat reports whether a format group uses standard CFIs
|
|
// as its structural locator currency (i.e. whether CFI verification and
|
|
// cssSelector derivation apply to it).
|
|
func IsConvertibleFormat(formatGroup string) bool {
|
|
return isConvertible(string(formatGroup))
|
|
}
|
|
|
|
func truncateRunes(s string, n int) string {
|
|
r := []rune(s)
|
|
if len(r) <= n {
|
|
return s
|
|
}
|
|
return string(r[:n])
|
|
}
|
|
|
|
// anchorBlock returns the nearest non-inline (block-level) ancestor of a
|
|
// resolved text/element node — the server-side equivalent of the
|
|
// readers' computed-style block walk.
|
|
func anchorBlock(node *html.Node) *html.Node {
|
|
if node == nil {
|
|
return nil
|
|
}
|
|
if node.Type == html.ElementNode && !isInlineFormatting(node) {
|
|
return node
|
|
}
|
|
return findBlockParent(node)
|
|
}
|
|
|
|
// blockContextText extracts the normalized text from (node, runeOff) to
|
|
// the end of the anchor block — the same excerpt rule the readers use for
|
|
// context_text, ≤100 chars.
|
|
func blockContextText(block, node *html.Node, runeOff int) string {
|
|
segments := collectInlineText(block)
|
|
var sb strings.Builder
|
|
started := false
|
|
for _, seg := range segments {
|
|
if !started && seg.node == node {
|
|
started = true
|
|
runes := []rune(string(seg.runes))
|
|
if runeOff < len(runes) {
|
|
sb.WriteString(string(runes[runeOff:]))
|
|
}
|
|
continue
|
|
}
|
|
if started {
|
|
sb.WriteString(string(seg.runes))
|
|
}
|
|
}
|
|
return truncateRunes(normalizeWhitespace(sb.String()), 100)
|
|
}
|
|
|
|
// contextMatches reports whether a server-extracted context and a client-
|
|
// submitted context describe the same anchor. Both are suffixes of the
|
|
// same block text when the anchors share a block, so containment in
|
|
// either direction verifies; empty or very short contexts never match.
|
|
func contextMatches(serverCtx, submitted string) bool {
|
|
s := truncateRunes(normalizeWhitespace(submitted), 100)
|
|
t := truncateRunes(normalizeWhitespace(serverCtx), 100)
|
|
if s == "" || t == "" {
|
|
return false
|
|
}
|
|
short, long := s, t
|
|
if len([]rune(short)) > len([]rune(long)) {
|
|
short, long = long, short
|
|
}
|
|
if len([]rune(short)) < 12 {
|
|
return false
|
|
}
|
|
return strings.Contains(long, short)
|
|
}
|
|
|
|
// cssSelectorFor mirrors the readers' selOf: a body-relative
|
|
// tag:nth-child(k) chain (k = 1-based position among element siblings).
|
|
func cssSelectorFor(block *html.Node) string {
|
|
var segs []string
|
|
n := block
|
|
for n != nil && n.Type == html.ElementNode && n.Data != "body" {
|
|
k := 1
|
|
sib := n.Parent.FirstChild
|
|
for sib != nil && sib != n {
|
|
if sib.Type == html.ElementNode {
|
|
k++
|
|
}
|
|
sib = sib.NextSibling
|
|
}
|
|
segs = append([]string{n.Data + ":nth-child(" + fmt.Sprintf("%d", k) + ")"}, segs...)
|
|
n = n.Parent
|
|
}
|
|
return "body>" + strings.Join(segs, ">")
|
|
}
|
|
|
|
// blockCharOffset computes the offset of (node, runeOff) within the
|
|
// concatenated text of its block, in UTF-16 code units — the client-side
|
|
// scroll-target currency (JavaScript .length semantics).
|
|
func blockCharOffset(block, node *html.Node, runeOff int) int {
|
|
segments := collectInlineText(block)
|
|
var sb strings.Builder
|
|
for _, seg := range segments {
|
|
if seg.node == node {
|
|
runes := seg.runes
|
|
if runeOff < len(runes) {
|
|
runes = runes[:runeOff]
|
|
}
|
|
sb.WriteString(string(runes))
|
|
return runeToUTF16Index(sb.String(), utf8.RuneCountInString(sb.String()))
|
|
}
|
|
sb.WriteString(string(seg.runes))
|
|
}
|
|
return utf16Len(sb.String())
|
|
}
|
|
|
|
// bookCharOffset computes the book-wide rune offset of (node, runeOff) —
|
|
// the reading_progress.character_offset column's currency, consistent with
|
|
// TotalCharacters and the percentage derivations.
|
|
func bookCharOffset(conv *CFIConverter, spineIndex int, node *html.Node, runeOff int) int {
|
|
spine, err := conv.loadSpine()
|
|
if err != nil {
|
|
return 0
|
|
}
|
|
before := 0
|
|
for i := 0; i < spineIndex && i < len(spine.items); i++ {
|
|
doc, _, derr := conv.getContentDoc(i + 1)
|
|
if derr != nil {
|
|
continue
|
|
}
|
|
if b := findBody(doc); b != nil {
|
|
before += countTextChars(b)
|
|
}
|
|
}
|
|
return before + countTextCharsBefore(node) + runeOff
|
|
}
|
|
|
|
// ProgressAnchor is the full server-computed apply handle for a stored
|
|
// standard CFI: the anchor block's cssSelector, the character offset
|
|
// within that block's text, and the spine document's href.
|
|
type ProgressAnchor struct {
|
|
CSSSelector string
|
|
CharOffset int
|
|
Href string
|
|
HealedCFI string
|
|
Healed bool
|
|
HealedPct *float64
|
|
}
|
|
|
|
// VerifyProgressAnchor resolves a client-submitted standard CFI against
|
|
// the EPUB, cross-checks the submitted context text, and heals the anchor
|
|
// by text search on any mismatch. charOffset is the anchor's block-
|
|
// relative UTF-16 offset (the served char_offset handle); bookOffset is
|
|
// the anchor's book-wide rune offset (the character_offset column's
|
|
// currency) — callers refresh the column from it on every verified save
|
|
// so it never goes stale behind the anchor.
|
|
func VerifyProgressAnchor(epubPath, epubcfi, contextText string, percentage float64) (finalCFI string, cssSelector string, anchorHref string, charOffset *int, bookOffset *int, healedPct *float64, healed bool, err error) {
|
|
finalCFI = epubcfi
|
|
anchorHref = ""
|
|
|
|
spineIndex, localSteps, err := parseEPUBCFI(epubcfi)
|
|
if err != nil {
|
|
cfi, sel, href, off, book, pct, healedFlag, herr := healFromContext(epubPath, contextText, percentage)
|
|
return cfi, sel, href, off, book, pct, healedFlag, herr
|
|
}
|
|
conv := cachedConverter(epubPath)
|
|
doc, docHref, err := conv.getContentDoc(spineIndex + 1)
|
|
if err != nil {
|
|
return healFromContext(epubPath, contextText, percentage)
|
|
}
|
|
node, runeOff, rerr := resolveCFIToNode(doc, localSteps)
|
|
if rerr != nil {
|
|
return healFromContext(epubPath, contextText, percentage)
|
|
}
|
|
anchorHref = docHref
|
|
|
|
block := anchorBlock(node)
|
|
serverCtx := blockContextText(block, node, runeOff)
|
|
if contextMatches(serverCtx, contextText) {
|
|
off := blockCharOffset(block, node, runeOff)
|
|
book := bookCharOffset(conv, spineIndex, node, runeOff)
|
|
return finalCFI, cssSelectorFor(block), anchorHref, &off, &book, nil, false, nil
|
|
}
|
|
|
|
// Mismatch: heal by text search.
|
|
hCFI, hPct, hSel, hHref, hBook, herr := healAnchorByText(epubPath, contextText, percentage, spineIndex)
|
|
if herr != nil {
|
|
return finalCFI, "", hHref, nil, nil, nil, false, fmt.Errorf("context mismatch (server %q vs client %q) and heal failed: %w",
|
|
truncateRunes(serverCtx, 40), truncateRunes(contextText, 40), herr)
|
|
}
|
|
hOff := blockCharOffsetFor(epubPath, hCFI)
|
|
return hCFI, hSel, hHref, &hOff, &hBook, &hPct, true, nil
|
|
}
|
|
|
|
func healFromContext(epubPath, contextText string, percentage float64) (string, string, string, *int, *int, *float64, bool, error) {
|
|
cfi, healedPct, sel, href, book, err := healAnchorByText(epubPath, contextText, percentage, -1)
|
|
if err != nil {
|
|
return "", "", "", nil, nil, nil, false, err
|
|
}
|
|
off := blockCharOffsetFor(epubPath, cfi)
|
|
return cfi, sel, href, &off, &book, &healedPct, true, nil
|
|
}
|
|
|
|
func healAnchorByText(epubPath, contextText string, percentage float64, spineIndex int) (string, float64, string, string, int, error) {
|
|
if epubPath == "" {
|
|
return "", 0, "", "", 0, fmt.Errorf("no epub available for text anchoring")
|
|
}
|
|
conv := cachedConverter(epubPath)
|
|
spine, err := conv.loadSpine()
|
|
if err != nil {
|
|
return "", 0, "", "", 0, err
|
|
}
|
|
needle := truncateRunes(normalizeWhitespace(contextText), 40)
|
|
if len([]rune(needle)) < 12 {
|
|
return "", 0, "", "", 0, fmt.Errorf("context too short to anchor")
|
|
}
|
|
|
|
type match struct {
|
|
spine int
|
|
node *html.Node
|
|
off int
|
|
}
|
|
var matches []match
|
|
totalChars := 0
|
|
charsBefore := make([]int, len(spine.items))
|
|
for i := range spine.items {
|
|
doc, _, derr := conv.getContentDoc(i + 1)
|
|
if derr != nil {
|
|
continue
|
|
}
|
|
body := findBody(doc)
|
|
if body == nil {
|
|
continue
|
|
}
|
|
charsBefore[i] = totalChars
|
|
totalChars += countTextChars(body)
|
|
if n, runeOff := findTextInNode(body, needle); n != nil {
|
|
matches = append(matches, match{spine: i, node: n, off: runeOff})
|
|
}
|
|
}
|
|
if len(matches) == 0 || totalChars <= 0 {
|
|
return "", 0, "", "", 0, fmt.Errorf("context not found in book")
|
|
}
|
|
|
|
best := matches[0]
|
|
bestDist := -1.0
|
|
for _, m := range matches {
|
|
frac := (float64(charsBefore[m.spine]) + float64(countTextCharsBefore(m.node)+m.off)) / float64(totalChars)
|
|
d := frac - percentage
|
|
if d < 0 {
|
|
d = -d
|
|
}
|
|
if bestDist < 0 || d < bestDist {
|
|
best = m
|
|
bestDist = d
|
|
}
|
|
}
|
|
|
|
bookOff := charsBefore[best.spine] + countTextCharsBefore(best.node) + best.off
|
|
healedPct := float64(bookOff) / float64(totalChars)
|
|
cfi, err := buildCFI(best.spine, best.node, best.off)
|
|
if err != nil {
|
|
return "", 0, "", "", 0, err
|
|
}
|
|
sel := ""
|
|
if block := anchorBlock(best.node); block != nil {
|
|
sel = cssSelectorFor(block)
|
|
}
|
|
return cfi, healedPct, sel, spine.items[best.spine].href, bookOff, nil
|
|
}
|
|
|
|
func blockCharOffsetFor(epubPath, cfi string) int {
|
|
n, off, _, herr := cfiTextAtAnchorInternal(epubPath, cfi)
|
|
if herr != nil {
|
|
return 0
|
|
}
|
|
block := anchorBlock(n)
|
|
if block == nil {
|
|
return 0
|
|
}
|
|
return blockCharOffset(block, n, off)
|
|
}
|
|
|
|
// cfiTextAtAnchorInternal resolves a stored standard CFI to its anchor
|
|
// text node, rune offset, and containing document.
|
|
func cfiTextAtAnchorInternal(epubPath, cfi string) (*html.Node, int, *html.Node, error) {
|
|
spineIndex, localSteps, err := parseEPUBCFI(cfi)
|
|
if err != nil {
|
|
return nil, 0, nil, err
|
|
}
|
|
conv := cachedConverter(epubPath)
|
|
doc, _, err := conv.getContentDoc(spineIndex + 1)
|
|
if err != nil {
|
|
return nil, 0, nil, err
|
|
}
|
|
node, off, err := resolveCFIToNode(doc, localSteps)
|
|
return node, off, doc, err
|
|
}
|
|
|
|
// ProgressCSSSelector derives the anchor block's cssSelector from a stored
|
|
// standard CFI.
|
|
func ProgressCSSSelector(epubPath, epubcfi string) (string, error) {
|
|
spineIndex, localSteps, err := parseEPUBCFI(epubcfi)
|
|
if err != nil {
|
|
return "", err
|
|
}
|
|
conv := cachedConverter(epubPath)
|
|
doc, _, err := conv.getContentDoc(spineIndex + 1)
|
|
if err != nil {
|
|
return "", err
|
|
}
|
|
node, _, err := resolveCFIToNode(doc, localSteps)
|
|
if err != nil {
|
|
return "", err
|
|
}
|
|
block := anchorBlock(node)
|
|
if block == nil {
|
|
return "", fmt.Errorf("no block ancestor for anchor")
|
|
}
|
|
return cssSelectorFor(block), nil
|
|
}
|