fix(sync): normalize character offsets to UTF-16 at the wire; refresh book offset on every verified save
Offset currency policy, now explicit: EPUB CFI terminals, CRE text() offsets and the served char_offset handle are UTF-16 code units (the EPUB CFI spec, and what foliate/readium/KOReader/Kobo clients actually observe), while internal arithmetic — the book-wide character_offset column and percentage fractions — stays rune-based, consistent with TotalCharacters. For all-BMP books the currencies are identical, so no stored value changes; astral-plane text (emoji, rare CJK) no longer drifts. Boundaries converted: resolveCFIToNode interprets incoming CFI terminal offsets as UTF-16; textNodeAtUTF16Offset (née textNodeAtRuneOffset) interprets CRE text() offsets as UTF-16; buildCFI and buildCREXPointer emit UTF-16 terminals; blockCharOffset (the served char_offset) is UTF-16. Also fixes two character_offset column defects: heals wrote a BLOCK- relative offset into the book-wide column, and verified-but-unhealed saves (e.g. KOReader pushes) never refreshed it, leaving it stale behind the anchor. VerifyProgressAnchor now returns the verified book- wide rune offset and SaveProgress refreshes the column on every verified save. Tests: astral currency round trip (offset after an emoji must shift by one unit between currencies, in both heal and exact-verify directions) and book-offset ordering. The cmd/server/tests integration harness failures under docker (library folder 400 during setup) reproduce on the pre-change tree and are unrelated.
This commit is contained in:
@@ -967,7 +967,7 @@ func (mh *MediaHandler) GetMediaReadingProgress(c *echo.Context) error {
|
|||||||
}
|
}
|
||||||
if mediaItem, err := mh.db.GetMediaItem(c.Request().Context(), pgtype.UUID{Bytes: mediaUUID, Valid: true}); err == nil {
|
if mediaItem, err := mh.db.GetMediaItem(c.Request().Context(), pgtype.UUID{Bytes: mediaUUID, Valid: true}); err == nil {
|
||||||
if path, perr := mh.libraryService.ResolveMediaPath(c.Request().Context(), mediaItem.LibraryID, mediaItem.FilePath); perr == nil && path != "" {
|
if path, perr := mh.libraryService.ResolveMediaPath(c.Request().Context(), mediaItem.LibraryID, mediaItem.FilePath); perr == nil && path != "" {
|
||||||
if finalCFI, sel, href, charOff, _, _, verr := wsync.VerifyProgressAnchor(path, progress.Epubcfi.String, contextText, pct); verr == nil && sel != "" {
|
if finalCFI, sel, href, charOff, _, _, _, verr := wsync.VerifyProgressAnchor(path, progress.Epubcfi.String, contextText, pct); verr == nil && sel != "" {
|
||||||
resp["css_selector"] = sel
|
resp["css_selector"] = sel
|
||||||
resp["anchor_href"] = href
|
resp["anchor_href"] = href
|
||||||
if charOff != nil {
|
if charOff != nil {
|
||||||
|
|||||||
+107
-23
@@ -431,7 +431,9 @@ func firstTextDescendant(n *html.Node) *html.Node {
|
|||||||
// textNodeAtRuneOffset walks text nodes under elem in document order and
|
// textNodeAtRuneOffset walks text nodes under elem in document order and
|
||||||
// returns the node containing the rune offset plus the local offset within
|
// returns the node containing the rune offset plus the local offset within
|
||||||
// that node. Offsets beyond the end clamp to the last node.
|
// that node. Offsets beyond the end clamp to the last node.
|
||||||
func textNodeAtRuneOffset(elem *html.Node, offset int) (*html.Node, int) {
|
// textNodeAtUTF16Offset resolves a crengine text().N offset — UTF-16 code
|
||||||
|
// units — to (text node, rune offset) within elem's text.
|
||||||
|
func textNodeAtUTF16Offset(elem *html.Node, offset int) (*html.Node, int) {
|
||||||
if offset < 0 {
|
if offset < 0 {
|
||||||
offset = 0
|
offset = 0
|
||||||
}
|
}
|
||||||
@@ -441,15 +443,15 @@ func textNodeAtRuneOffset(elem *html.Node, offset int) (*html.Node, int) {
|
|||||||
var walk func(*html.Node) bool
|
var walk func(*html.Node) bool
|
||||||
walk = func(node *html.Node) bool {
|
walk = func(node *html.Node) bool {
|
||||||
if node.Type == html.TextNode {
|
if node.Type == html.TextNode {
|
||||||
length := utf8.RuneCountInString(node.Data)
|
length := utf16Len(node.Data)
|
||||||
if remaining < length {
|
if remaining < length {
|
||||||
target = node
|
target = node
|
||||||
local = remaining
|
local = utf16ToRuneIndex(node.Data, remaining)
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
remaining -= length
|
remaining -= length
|
||||||
target = node
|
target = node
|
||||||
local = length
|
local = utf8.RuneCountInString(node.Data)
|
||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
for child := node.FirstChild; child != nil; child = child.NextSibling {
|
for child := node.FirstChild; child != nil; child = child.NextSibling {
|
||||||
@@ -507,7 +509,7 @@ func (c *CFIConverter) convertByStructuralPath(body *html.Node, xp *CREXPointer,
|
|||||||
var textNode *html.Node
|
var textNode *html.Node
|
||||||
var localOffset int
|
var localOffset int
|
||||||
if xp.CharOffset > 0 {
|
if xp.CharOffset > 0 {
|
||||||
textNode, localOffset = textNodeAtRuneOffset(elem, xp.CharOffset)
|
textNode, localOffset = textNodeAtUTF16Offset(elem, xp.CharOffset)
|
||||||
} else {
|
} else {
|
||||||
textNode = firstTextDescendant(elem)
|
textNode = firstTextDescendant(elem)
|
||||||
localOffset = 0
|
localOffset = 0
|
||||||
@@ -1058,24 +1060,96 @@ func indexChildNodes(parent *html.Node) []indexedNode {
|
|||||||
return nodes
|
return nodes
|
||||||
}
|
}
|
||||||
|
|
||||||
func findTextChunkIndex(parent *html.Node, textNode *html.Node) (int, int) {
|
// Character-offset currency policy: the EPUB CFI spec and crengine both
|
||||||
|
// count UTF-16 code units (JavaScript `.length` semantics — what foliate,
|
||||||
|
// readium, KOReader and Kobo clients all observe), so every offset that
|
||||||
|
// CROSSES the wire — CFI terminals, CRE text() offsets, the served
|
||||||
|
// char_offset handle — is UTF-16. Internal arithmetic (book-level
|
||||||
|
// character_offset, percentage fractions) stays rune-based, consistent
|
||||||
|
// with TotalCharacters. These helpers convert at the boundaries; for
|
||||||
|
// all-BMP text the two currencies are identical, so ASCII books are
|
||||||
|
// unaffected.
|
||||||
|
|
||||||
|
// utf16Len returns the UTF-16 code-unit length of s.
|
||||||
|
func utf16Len(s string) int {
|
||||||
|
n := 0
|
||||||
|
for _, r := range s {
|
||||||
|
if r >= 0x10000 {
|
||||||
|
n += 2
|
||||||
|
} else {
|
||||||
|
n++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return n
|
||||||
|
}
|
||||||
|
|
||||||
|
// utf16ToRuneIndex converts a UTF-16 code-unit offset within s to a rune
|
||||||
|
// index (clamped to len(runes)).
|
||||||
|
func utf16ToRuneIndex(s string, u16 int) int {
|
||||||
|
if u16 <= 0 {
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
units := 0
|
||||||
|
i := 0
|
||||||
|
for _, r := range s {
|
||||||
|
if units >= u16 {
|
||||||
|
return i
|
||||||
|
}
|
||||||
|
if r >= 0x10000 {
|
||||||
|
units += 2
|
||||||
|
} else {
|
||||||
|
units++
|
||||||
|
}
|
||||||
|
i++
|
||||||
|
}
|
||||||
|
return i
|
||||||
|
}
|
||||||
|
|
||||||
|
// runeToUTF16Index converts a rune index within s to a UTF-16 code-unit
|
||||||
|
// offset (clamped to the string's unit length).
|
||||||
|
func runeToUTF16Index(s string, runeIdx int) int {
|
||||||
|
if runeIdx <= 0 {
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
units := 0
|
||||||
|
i := 0
|
||||||
|
for _, r := range s {
|
||||||
|
if i >= runeIdx {
|
||||||
|
return units
|
||||||
|
}
|
||||||
|
if r >= 0x10000 {
|
||||||
|
units += 2
|
||||||
|
} else {
|
||||||
|
units++
|
||||||
|
}
|
||||||
|
i++
|
||||||
|
}
|
||||||
|
return units
|
||||||
|
}
|
||||||
|
|
||||||
|
// findTextChunk locates the indexed text chunk containing textNode and
|
||||||
|
// returns its chunk index, the rune offset of the node within the chunk,
|
||||||
|
// and the chunk text up to and including the node (for UTF-16 conversion
|
||||||
|
// of chunk-relative offsets).
|
||||||
|
func findTextChunk(parent *html.Node, textNode *html.Node) (int, int, string) {
|
||||||
indexed := indexChildNodes(parent)
|
indexed := indexChildNodes(parent)
|
||||||
for i, node := range indexed {
|
for i, node := range indexed {
|
||||||
if node.isTextChunk() {
|
if node.isTextChunk() {
|
||||||
for j, tn := range node.textChunk {
|
var sb strings.Builder
|
||||||
if tn == textNode {
|
|
||||||
chunkOffset := 0
|
chunkOffset := 0
|
||||||
for k := 0; k < j; k++ {
|
for _, tn := range node.textChunk {
|
||||||
chunkOffset += utf8.RuneCountInString(node.textChunk[k].Data)
|
if tn == textNode {
|
||||||
|
return i, chunkOffset, sb.String() + textNode.Data
|
||||||
}
|
}
|
||||||
return i, chunkOffset
|
chunkOffset += utf8.RuneCountInString(tn.Data)
|
||||||
|
sb.WriteString(tn.Data)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
return -1, 0, ""
|
||||||
return -1, 0
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
func findElementCFIIndex(parent *html.Node, element *html.Node) int {
|
func findElementCFIIndex(parent *html.Node, element *html.Node) int {
|
||||||
indexed := indexChildNodes(parent)
|
indexed := indexChildNodes(parent)
|
||||||
for i, node := range indexed {
|
for i, node := range indexed {
|
||||||
@@ -1110,15 +1184,17 @@ func buildCFI(spineIndex int, textNode *html.Node, charOffset int) (string, erro
|
|||||||
return "", fmt.Errorf("text node has no parent")
|
return "", fmt.Errorf("text node has no parent")
|
||||||
}
|
}
|
||||||
|
|
||||||
chunkIdx, chunkOffset := findTextChunkIndex(parent, textNode)
|
chunkIdx, chunkOffset, chunkText := findTextChunk(parent, textNode)
|
||||||
if chunkIdx == -1 {
|
if chunkIdx == -1 {
|
||||||
return "", fmt.Errorf("text node not found in parent's indexed children")
|
return "", fmt.Errorf("text node not found in parent's indexed children")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The CFI terminal offset is UTF-16 code units (spec currency); the
|
||||||
|
// internal charOffset is runes. Convert over the chunk text.
|
||||||
totalOffset := chunkOffset + charOffset
|
totalOffset := chunkOffset + charOffset
|
||||||
|
|
||||||
var parts []string
|
var parts []string
|
||||||
parts = append(parts, fmt.Sprintf("/%d:%d", chunkIdx, totalOffset))
|
parts = append(parts, fmt.Sprintf("/%d:%d", chunkIdx, runeToUTF16Index(chunkText, totalOffset)))
|
||||||
|
|
||||||
current := parent
|
current := parent
|
||||||
for current != nil {
|
for current != nil {
|
||||||
@@ -1519,24 +1595,31 @@ func resolveCFIToNode(doc *html.Node, steps []cfiStep) (*html.Node, int, error)
|
|||||||
entry := indexed[lastStep.Index]
|
entry := indexed[lastStep.Index]
|
||||||
|
|
||||||
if entry.isTextChunk() {
|
if entry.isTextChunk() {
|
||||||
textOffset := 0
|
// The CFI terminal offset arrives in UTF-16 code units (spec
|
||||||
|
// currency — foliate/readium/KOReader/Kobo all emit UTF-16).
|
||||||
|
// Walk the chunk in UTF-16 units, then convert the hit position
|
||||||
|
// to the internal rune offset.
|
||||||
|
u16Remaining := 0
|
||||||
if lastStep.HasOffset {
|
if lastStep.HasOffset {
|
||||||
textOffset = lastStep.Offset
|
u16Remaining = lastStep.Offset
|
||||||
}
|
}
|
||||||
|
|
||||||
var targetNode *html.Node
|
var targetNode *html.Node
|
||||||
remainingOffset := textOffset
|
runeIntoTarget := 0
|
||||||
for _, tn := range entry.textChunk {
|
for _, tn := range entry.textChunk {
|
||||||
textLen := utf8.RuneCountInString(tn.Data)
|
units := utf16Len(tn.Data)
|
||||||
if remainingOffset < textLen || (remainingOffset == textLen && targetNode == nil) {
|
if u16Remaining < units || (u16Remaining == units && targetNode == nil) {
|
||||||
targetNode = tn
|
targetNode = tn
|
||||||
|
runeIntoTarget = utf16ToRuneIndex(tn.Data, u16Remaining)
|
||||||
break
|
break
|
||||||
}
|
}
|
||||||
remainingOffset -= textLen
|
u16Remaining -= units
|
||||||
targetNode = tn
|
targetNode = tn
|
||||||
|
runeIntoTarget = utf8.RuneCountInString(tn.Data)
|
||||||
}
|
}
|
||||||
if targetNode == nil && len(entry.textChunk) > 0 {
|
if targetNode == nil && len(entry.textChunk) > 0 {
|
||||||
targetNode = entry.textChunk[len(entry.textChunk)-1]
|
targetNode = entry.textChunk[len(entry.textChunk)-1]
|
||||||
|
runeIntoTarget = utf8.RuneCountInString(targetNode.Data)
|
||||||
}
|
}
|
||||||
|
|
||||||
parent := targetNode.Parent
|
parent := targetNode.Parent
|
||||||
@@ -1547,7 +1630,7 @@ func resolveCFIToNode(doc *html.Node, steps []cfiStep) (*html.Node, int, error)
|
|||||||
}
|
}
|
||||||
totalOffset += countTextChars(c)
|
totalOffset += countTextChars(c)
|
||||||
}
|
}
|
||||||
totalOffset += remainingOffset
|
totalOffset += runeIntoTarget
|
||||||
|
|
||||||
return targetNode, totalOffset, nil
|
return targetNode, totalOffset, nil
|
||||||
}
|
}
|
||||||
@@ -1606,7 +1689,8 @@ func buildCREXPointer(spineIndex int, node *html.Node, charOffset int) (string,
|
|||||||
xpointer := fmt.Sprintf("/body/DocFragment[%d]/body%s", fragIndex, strings.Join(parts, ""))
|
xpointer := fmt.Sprintf("/body/DocFragment[%d]/body%s", fragIndex, strings.Join(parts, ""))
|
||||||
|
|
||||||
if charOffset > 0 || (node.Type == html.TextNode) {
|
if charOffset > 0 || (node.Type == html.TextNode) {
|
||||||
xpointer += fmt.Sprintf("/text().%d", charOffset)
|
// crengine counts UTF-16 code units; charOffset is internal runes.
|
||||||
|
xpointer += fmt.Sprintf("/text().%d", runeToUTF16Index(node.Data, charOffset))
|
||||||
}
|
}
|
||||||
|
|
||||||
return xpointer, nil
|
return xpointer, nil
|
||||||
|
|||||||
@@ -133,7 +133,7 @@ func writeTestEPUB(t *testing.T) string {
|
|||||||
}
|
}
|
||||||
docs := []spineDoc{
|
docs := []spineDoc{
|
||||||
{"doc1.xhtml", "<body><div><p>Chapter one opening page.</p></div></body>"},
|
{"doc1.xhtml", "<body><div><p>Chapter one opening page.</p></div></body>"},
|
||||||
{"doc2.xhtml", "<body><div><p>The family of Dashwood had long been settled in Sussex.</p><p>Their estate was large, and their residence was at Norland Park.</p></div></body>"},
|
{"doc2.xhtml", "<body><div><p>The family of Dashwood had long been settled in Sussex.</p><p>Their estate was large, and their residence was at Norland Park.</p><p>The family crest shows a globe \U0001F30D and a rocket \U0001F680 flying onward.</p></div></body>"},
|
||||||
{"doc3.xhtml", "<body><div><p>Chapter three contents.</p></div></body>"},
|
{"doc3.xhtml", "<body><div><p>Chapter three contents.</p></div></body>"},
|
||||||
{"doc4.xhtml", "<body><div><p>Chapter four contents.</p></div></body>"},
|
{"doc4.xhtml", "<body><div><p>Chapter four contents.</p></div></body>"},
|
||||||
{"doc5.xhtml", "<body><div><p>Chapter five contents.</p></div></body>"},
|
{"doc5.xhtml", "<body><div><p>Chapter five contents.</p></div></body>"},
|
||||||
|
|||||||
+60
-26
@@ -26,6 +26,7 @@ package sync
|
|||||||
import (
|
import (
|
||||||
"fmt"
|
"fmt"
|
||||||
"strings"
|
"strings"
|
||||||
|
"unicode/utf8"
|
||||||
|
|
||||||
"golang.org/x/net/html"
|
"golang.org/x/net/html"
|
||||||
)
|
)
|
||||||
@@ -121,18 +122,45 @@ func cssSelectorFor(block *html.Node) string {
|
|||||||
return "body>" + strings.Join(segs, ">")
|
return "body>" + strings.Join(segs, ">")
|
||||||
}
|
}
|
||||||
|
|
||||||
// blockCharOffset computes the rune offset of (node, runeOff) within the
|
// blockCharOffset computes the offset of (node, runeOff) within the
|
||||||
// concatenated text of its block — the client-side scroll target.
|
// concatenated text of its block, in UTF-16 code units — the client-side
|
||||||
|
// scroll-target currency (JavaScript .length semantics).
|
||||||
func blockCharOffset(block, node *html.Node, runeOff int) int {
|
func blockCharOffset(block, node *html.Node, runeOff int) int {
|
||||||
segments := collectInlineText(block)
|
segments := collectInlineText(block)
|
||||||
s := 0
|
var sb strings.Builder
|
||||||
for _, seg := range segments {
|
for _, seg := range segments {
|
||||||
if seg.node == node {
|
if seg.node == node {
|
||||||
return s + runeOff
|
runes := seg.runes
|
||||||
|
if runeOff < len(runes) {
|
||||||
|
runes = runes[:runeOff]
|
||||||
}
|
}
|
||||||
s += len(seg.runes)
|
sb.WriteString(string(runes))
|
||||||
|
return runeToUTF16Index(sb.String(), utf8.RuneCountInString(sb.String()))
|
||||||
}
|
}
|
||||||
return s + runeOff
|
sb.WriteString(string(seg.runes))
|
||||||
|
}
|
||||||
|
return utf16Len(sb.String())
|
||||||
|
}
|
||||||
|
|
||||||
|
// bookCharOffset computes the book-wide rune offset of (node, runeOff) —
|
||||||
|
// the reading_progress.character_offset column's currency, consistent with
|
||||||
|
// TotalCharacters and the percentage derivations.
|
||||||
|
func bookCharOffset(conv *CFIConverter, spineIndex int, node *html.Node, runeOff int) int {
|
||||||
|
spine, err := conv.loadSpine()
|
||||||
|
if err != nil {
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
before := 0
|
||||||
|
for i := 0; i < spineIndex && i < len(spine.items); i++ {
|
||||||
|
doc, _, derr := conv.getContentDoc(i + 1)
|
||||||
|
if derr != nil {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if b := findBody(doc); b != nil {
|
||||||
|
before += countTextChars(b)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return before + countTextCharsBefore(node) + runeOff
|
||||||
}
|
}
|
||||||
|
|
||||||
// ProgressAnchor is the full server-computed apply handle for a stored
|
// ProgressAnchor is the full server-computed apply handle for a stored
|
||||||
@@ -149,15 +177,19 @@ type ProgressAnchor struct {
|
|||||||
|
|
||||||
// VerifyProgressAnchor resolves a client-submitted standard CFI against
|
// VerifyProgressAnchor resolves a client-submitted standard CFI against
|
||||||
// the EPUB, cross-checks the submitted context text, and heals the anchor
|
// the EPUB, cross-checks the submitted context text, and heals the anchor
|
||||||
// by text search on any mismatch.
|
// by text search on any mismatch. charOffset is the anchor's block-
|
||||||
func VerifyProgressAnchor(epubPath, epubcfi, contextText string, percentage float64) (finalCFI string, cssSelector string, anchorHref string, charOffset *int, healedPct *float64, healed bool, err error) {
|
// relative UTF-16 offset (the served char_offset handle); bookOffset is
|
||||||
|
// the anchor's book-wide rune offset (the character_offset column's
|
||||||
|
// currency) — callers refresh the column from it on every verified save
|
||||||
|
// so it never goes stale behind the anchor.
|
||||||
|
func VerifyProgressAnchor(epubPath, epubcfi, contextText string, percentage float64) (finalCFI string, cssSelector string, anchorHref string, charOffset *int, bookOffset *int, healedPct *float64, healed bool, err error) {
|
||||||
finalCFI = epubcfi
|
finalCFI = epubcfi
|
||||||
anchorHref = ""
|
anchorHref = ""
|
||||||
|
|
||||||
spineIndex, localSteps, err := parseEPUBCFI(epubcfi)
|
spineIndex, localSteps, err := parseEPUBCFI(epubcfi)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
cfi, sel, href, off, pct, healedFlag, herr := healFromContext(epubPath, contextText, percentage)
|
cfi, sel, href, off, book, pct, healedFlag, herr := healFromContext(epubPath, contextText, percentage)
|
||||||
return cfi, sel, href, off, pct, healedFlag, herr
|
return cfi, sel, href, off, book, pct, healedFlag, herr
|
||||||
}
|
}
|
||||||
conv := cachedConverter(epubPath)
|
conv := cachedConverter(epubPath)
|
||||||
doc, docHref, err := conv.getContentDoc(spineIndex + 1)
|
doc, docHref, err := conv.getContentDoc(spineIndex + 1)
|
||||||
@@ -174,40 +206,41 @@ func VerifyProgressAnchor(epubPath, epubcfi, contextText string, percentage floa
|
|||||||
serverCtx := blockContextText(block, node, runeOff)
|
serverCtx := blockContextText(block, node, runeOff)
|
||||||
if contextMatches(serverCtx, contextText) {
|
if contextMatches(serverCtx, contextText) {
|
||||||
off := blockCharOffset(block, node, runeOff)
|
off := blockCharOffset(block, node, runeOff)
|
||||||
return finalCFI, cssSelectorFor(block), anchorHref, &off, nil, false, nil
|
book := bookCharOffset(conv, spineIndex, node, runeOff)
|
||||||
|
return finalCFI, cssSelectorFor(block), anchorHref, &off, &book, nil, false, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// Mismatch: heal by text search.
|
// Mismatch: heal by text search.
|
||||||
hCFI, hPct, hSel, hHref, herr := healAnchorByText(epubPath, contextText, percentage, spineIndex)
|
hCFI, hPct, hSel, hHref, hBook, herr := healAnchorByText(epubPath, contextText, percentage, spineIndex)
|
||||||
if herr != nil {
|
if herr != nil {
|
||||||
return finalCFI, "", hHref, nil, nil, false, fmt.Errorf("context mismatch (server %q vs client %q) and heal failed: %w",
|
return finalCFI, "", hHref, nil, nil, nil, false, fmt.Errorf("context mismatch (server %q vs client %q) and heal failed: %w",
|
||||||
truncateRunes(serverCtx, 40), truncateRunes(contextText, 40), herr)
|
truncateRunes(serverCtx, 40), truncateRunes(contextText, 40), herr)
|
||||||
}
|
}
|
||||||
hOff := blockCharOffsetFor(epubPath, hCFI)
|
hOff := blockCharOffsetFor(epubPath, hCFI)
|
||||||
return hCFI, hSel, hHref, &hOff, &hPct, true, nil
|
return hCFI, hSel, hHref, &hOff, &hBook, &hPct, true, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func healFromContext(epubPath, contextText string, percentage float64) (string, string, string, *int, *float64, bool, error) {
|
func healFromContext(epubPath, contextText string, percentage float64) (string, string, string, *int, *int, *float64, bool, error) {
|
||||||
cfi, healedPct, sel, href, err := healAnchorByText(epubPath, contextText, percentage, -1)
|
cfi, healedPct, sel, href, book, err := healAnchorByText(epubPath, contextText, percentage, -1)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return "", "", "", nil, nil, false, err
|
return "", "", "", nil, nil, nil, false, err
|
||||||
}
|
}
|
||||||
off := blockCharOffsetFor(epubPath, cfi)
|
off := blockCharOffsetFor(epubPath, cfi)
|
||||||
return cfi, sel, href, &off, &healedPct, true, nil
|
return cfi, sel, href, &off, &book, &healedPct, true, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func healAnchorByText(epubPath, contextText string, percentage float64, spineIndex int) (string, float64, string, string, error) {
|
func healAnchorByText(epubPath, contextText string, percentage float64, spineIndex int) (string, float64, string, string, int, error) {
|
||||||
if epubPath == "" {
|
if epubPath == "" {
|
||||||
return "", 0, "", "", fmt.Errorf("no epub available for text anchoring")
|
return "", 0, "", "", 0, fmt.Errorf("no epub available for text anchoring")
|
||||||
}
|
}
|
||||||
conv := cachedConverter(epubPath)
|
conv := cachedConverter(epubPath)
|
||||||
spine, err := conv.loadSpine()
|
spine, err := conv.loadSpine()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return "", 0, "", "", err
|
return "", 0, "", "", 0, err
|
||||||
}
|
}
|
||||||
needle := truncateRunes(normalizeWhitespace(contextText), 40)
|
needle := truncateRunes(normalizeWhitespace(contextText), 40)
|
||||||
if len([]rune(needle)) < 12 {
|
if len([]rune(needle)) < 12 {
|
||||||
return "", 0, "", "", fmt.Errorf("context too short to anchor")
|
return "", 0, "", "", 0, fmt.Errorf("context too short to anchor")
|
||||||
}
|
}
|
||||||
|
|
||||||
type match struct {
|
type match struct {
|
||||||
@@ -234,7 +267,7 @@ func healAnchorByText(epubPath, contextText string, percentage float64, spineInd
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
if len(matches) == 0 || totalChars <= 0 {
|
if len(matches) == 0 || totalChars <= 0 {
|
||||||
return "", 0, "", "", fmt.Errorf("context not found in book")
|
return "", 0, "", "", 0, fmt.Errorf("context not found in book")
|
||||||
}
|
}
|
||||||
|
|
||||||
best := matches[0]
|
best := matches[0]
|
||||||
@@ -251,16 +284,17 @@ func healAnchorByText(epubPath, contextText string, percentage float64, spineInd
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
healedPct := (float64(charsBefore[best.spine]) + float64(countTextCharsBefore(best.node)+best.off)) / float64(totalChars)
|
bookOff := charsBefore[best.spine] + countTextCharsBefore(best.node) + best.off
|
||||||
|
healedPct := float64(bookOff) / float64(totalChars)
|
||||||
cfi, err := buildCFI(best.spine, best.node, best.off)
|
cfi, err := buildCFI(best.spine, best.node, best.off)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return "", 0, "", "", err
|
return "", 0, "", "", 0, err
|
||||||
}
|
}
|
||||||
sel := ""
|
sel := ""
|
||||||
if block := anchorBlock(best.node); block != nil {
|
if block := anchorBlock(best.node); block != nil {
|
||||||
sel = cssSelectorFor(block)
|
sel = cssSelectorFor(block)
|
||||||
}
|
}
|
||||||
return cfi, healedPct, sel, spine.items[best.spine].href, nil
|
return cfi, healedPct, sel, spine.items[best.spine].href, bookOff, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func blockCharOffsetFor(epubPath, cfi string) int {
|
func blockCharOffsetFor(epubPath, cfi string) int {
|
||||||
|
|||||||
@@ -12,7 +12,7 @@ func TestManualRealBookAnchor(t *testing.T) {
|
|||||||
}
|
}
|
||||||
cfi := "epubcfi(/6/2!/4[x1984]/8[_idContainer003]/62/1:456)"
|
cfi := "epubcfi(/6/2!/4[x1984]/8[_idContainer003]/62/1:456)"
|
||||||
ctx := "was at war with one of these Powers it was generally at peace with the other"
|
ctx := "was at war with one of these Powers it was generally at peace with the other"
|
||||||
finalCFI, sel, href, charOff, healedPct, healed, err := VerifyProgressAnchor(path, cfi, ctx, 0.0415)
|
finalCFI, sel, href, charOff, _, healedPct, healed, err := VerifyProgressAnchor(path, cfi, ctx, 0.0415)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("VerifyProgressAnchor error: %v", err)
|
t.Fatalf("VerifyProgressAnchor error: %v", err)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,6 +1,7 @@
|
|||||||
package sync
|
package sync
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"fmt"
|
||||||
"testing"
|
"testing"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -21,7 +22,7 @@ const (
|
|||||||
|
|
||||||
func TestVerifyProgressAnchorAcceptsExact(t *testing.T) {
|
func TestVerifyProgressAnchorAcceptsExact(t *testing.T) {
|
||||||
path := writeTestEPUB(t)
|
path := writeTestEPUB(t)
|
||||||
finalCFI, sel, _, charOff, healedPct, healed, err := VerifyProgressAnchor(path, dashwoodCFI, dashwoodText, 0.3)
|
finalCFI, sel, _, charOff, _, healedPct, healed, err := VerifyProgressAnchor(path, dashwoodCFI, dashwoodText, 0.3)
|
||||||
_ = charOff
|
_ = charOff
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("verify error: %v", err)
|
t.Fatalf("verify error: %v", err)
|
||||||
@@ -44,7 +45,7 @@ func TestVerifyProgressAnchorHealsMismatch(t *testing.T) {
|
|||||||
path := writeTestEPUB(t)
|
path := writeTestEPUB(t)
|
||||||
// Anchored at p1 but the context is p2's text: the classic
|
// Anchored at p1 but the context is p2's text: the classic
|
||||||
// client-projection bug — the server must heal to the true location.
|
// client-projection bug — the server must heal to the true location.
|
||||||
finalCFI, _, _, charOff, healedPct, healed, err := VerifyProgressAnchor(path, dashwoodCFI, estateText, 0.3)
|
finalCFI, _, _, charOff, _, healedPct, healed, err := VerifyProgressAnchor(path, dashwoodCFI, estateText, 0.3)
|
||||||
_ = charOff
|
_ = charOff
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("verify error: %v", err)
|
t.Fatalf("verify error: %v", err)
|
||||||
@@ -60,7 +61,7 @@ func TestVerifyProgressAnchorHealsMismatch(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// Healing must converge: re-verifying the healed anchor is a no-op.
|
// Healing must converge: re-verifying the healed anchor is a no-op.
|
||||||
finalCFI2, _, _, _, healedPct2, healed2, err := VerifyProgressAnchor(path, finalCFI, estateText, 0.3)
|
finalCFI2, _, _, _, _, healedPct2, healed2, err := VerifyProgressAnchor(path, finalCFI, estateText, 0.3)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("re-verify error: %v", err)
|
t.Fatalf("re-verify error: %v", err)
|
||||||
}
|
}
|
||||||
@@ -76,7 +77,7 @@ func TestVerifyProgressAnchorHealsUnresolvable(t *testing.T) {
|
|||||||
path := writeTestEPUB(t)
|
path := writeTestEPUB(t)
|
||||||
// Element index 99 is out of range in doc2 — the anchor cannot resolve.
|
// Element index 99 is out of range in doc2 — the anchor cannot resolve.
|
||||||
bogus := "epubcfi(/6/4!/4/2/99/1:0)"
|
bogus := "epubcfi(/6/4!/4/2/99/1:0)"
|
||||||
finalCFI, _, _, charOff, healedPct, healed, err := VerifyProgressAnchor(path, bogus, dashwoodText, 0.05)
|
finalCFI, _, _, charOff, _, healedPct, healed, err := VerifyProgressAnchor(path, bogus, dashwoodText, 0.05)
|
||||||
_ = charOff
|
_ = charOff
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("verify error: %v", err)
|
t.Fatalf("verify error: %v", err)
|
||||||
@@ -96,7 +97,7 @@ func TestVerifyProgressAnchorDumbClient(t *testing.T) {
|
|||||||
path := writeTestEPUB(t)
|
path := writeTestEPUB(t)
|
||||||
// No CFI at all — a client that only knows percentage + context is
|
// No CFI at all — a client that only knows percentage + context is
|
||||||
// fully supported: the server anchors structurally from the text.
|
// fully supported: the server anchors structurally from the text.
|
||||||
finalCFI, sel, _, charOff, healedPct, healed, err := VerifyProgressAnchor(path, "", estateText, 0.3)
|
finalCFI, sel, _, charOff, _, healedPct, healed, err := VerifyProgressAnchor(path, "", estateText, 0.3)
|
||||||
_ = charOff
|
_ = charOff
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("verify error: %v", err)
|
t.Fatalf("verify error: %v", err)
|
||||||
@@ -156,3 +157,92 @@ func TestParseStandardCFIRange(t *testing.T) {
|
|||||||
t.Errorf("start-arm step = %+v, want /4:0", last)
|
t.Errorf("start-arm step = %+v, want /4:0", last)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The emoji in fixture doc2's third paragraph are astral plane runes:
|
||||||
|
// one rune, two UTF-16 code units. Every wire offset (CFI terminals,
|
||||||
|
// the served char_offset) must therefore be UTF-16, while the book-wide
|
||||||
|
// character_offset column stays rune-based.
|
||||||
|
//
|
||||||
|
// Case 1: the context starts at rune 31 of the paragraph text ("The
|
||||||
|
// family crest shows a globe " = 31 BMP runes), so its UTF-16 offset is
|
||||||
|
// also 31 — the currencies agree.
|
||||||
|
//
|
||||||
|
// Case 2: the context starts at rune 33 ("and a rocket ..."), with the
|
||||||
|
// astral 🌍 (rune 31, units 31-32) BEFORE the offset — the UTF-16 offset
|
||||||
|
// is 34, one more than the rune offset. That +1 is the whole point of
|
||||||
|
// the boundary conversion.
|
||||||
|
//
|
||||||
|
// Local path: p3 of doc2's div = /4/2/6, text chunk /1.
|
||||||
|
func TestAstralOffsetCurrency(t *testing.T) {
|
||||||
|
path := writeTestEPUB(t)
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
context string
|
||||||
|
wantTerm int
|
||||||
|
}{
|
||||||
|
{"before any emoji", "🌍 and a rocket 🚀 flying onward.", 31},
|
||||||
|
{"after one emoji", "and a rocket 🚀 flying onward.", 34},
|
||||||
|
}
|
||||||
|
for _, tc := range cases {
|
||||||
|
t.Run(tc.name, func(t *testing.T) {
|
||||||
|
wantCFI := fmt.Sprintf("epubcfi(/6/4!/4/2/6/1:%d)", tc.wantTerm)
|
||||||
|
|
||||||
|
finalCFI, sel, _, charOff, bookOff, _, healed, err := VerifyProgressAnchor(path, "", tc.context, 0.3)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("heal error: %v", err)
|
||||||
|
}
|
||||||
|
if !healed {
|
||||||
|
t.Fatalf("expected heal from context-only submission")
|
||||||
|
}
|
||||||
|
if finalCFI != wantCFI {
|
||||||
|
t.Errorf("healed CFI = %q, want %q (UTF-16 terminal)", finalCFI, wantCFI)
|
||||||
|
}
|
||||||
|
if sel != "body>div:nth-child(1)>p:nth-child(3)" {
|
||||||
|
t.Errorf("cssSelector = %q", sel)
|
||||||
|
}
|
||||||
|
if charOff == nil || *charOff != tc.wantTerm {
|
||||||
|
t.Errorf("block char_offset = %v, want %d UTF-16 units", charOff, tc.wantTerm)
|
||||||
|
}
|
||||||
|
if bookOff == nil || *bookOff <= 0 {
|
||||||
|
t.Errorf("book offset = %v, want a positive book-wide rune offset", bookOff)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Round trip: the healed CFI must verify exactly, with the
|
||||||
|
// same UTF-16 block handle and a stable book offset.
|
||||||
|
finalCFI2, _, _, charOff2, bookOff2, healedPct2, healed2, err := VerifyProgressAnchor(path, finalCFI, tc.context, 0.3)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("re-verify error: %v", err)
|
||||||
|
}
|
||||||
|
if healed2 || healedPct2 != nil {
|
||||||
|
t.Errorf("exact round trip should not heal (healed=%v)", healed2)
|
||||||
|
}
|
||||||
|
if finalCFI2 != wantCFI {
|
||||||
|
t.Errorf("re-verified CFI = %q, want %q", finalCFI2, wantCFI)
|
||||||
|
}
|
||||||
|
if charOff2 == nil || *charOff2 != tc.wantTerm {
|
||||||
|
t.Errorf("re-verified block char_offset = %v, want %d", charOff2, tc.wantTerm)
|
||||||
|
}
|
||||||
|
if bookOff2 == nil || bookOff == nil || *bookOff2 != *bookOff {
|
||||||
|
t.Errorf("book offset unstable: %v vs %v", bookOff2, bookOff)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The book-wide offset must order anchors the way the book orders them:
|
||||||
|
// a later paragraph in the same document has a strictly larger
|
||||||
|
// character_offset.
|
||||||
|
func TestBookOffsetOrdering(t *testing.T) {
|
||||||
|
path := writeTestEPUB(t)
|
||||||
|
_, _, _, _, book1, _, _, err1 := VerifyProgressAnchor(path, dashwoodCFI, dashwoodText, 0.3)
|
||||||
|
_, _, _, _, book2, _, _, err2 := VerifyProgressAnchor(path, estateCFI, estateText, 0.3)
|
||||||
|
if err1 != nil || err2 != nil {
|
||||||
|
t.Fatalf("verify errors: %v %v", err1, err2)
|
||||||
|
}
|
||||||
|
if book1 == nil || book2 == nil {
|
||||||
|
t.Fatalf("book offsets missing: %v %v", book1, book2)
|
||||||
|
}
|
||||||
|
if *book2 <= *book1 {
|
||||||
|
t.Errorf("estate offset %d should exceed dashwood offset %d", *book2, *book1)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -462,7 +462,7 @@ func (s *ProgressService) SaveProgress(ctx context.Context, req SaveProgressRequ
|
|||||||
if params.Percentage.Valid {
|
if params.Percentage.Valid {
|
||||||
submittedPct = params.Percentage.Float64
|
submittedPct = params.Percentage.Float64
|
||||||
}
|
}
|
||||||
if finalCFI, _, _, charOff, healedPct, healed, verr := VerifyProgressAnchor(path, submittedCFI, params.ContextText.String, submittedPct); verr != nil {
|
if finalCFI, _, _, _, bookOff, healedPct, healed, verr := VerifyProgressAnchor(path, submittedCFI, params.ContextText.String, submittedPct); verr != nil {
|
||||||
log.Printf("Bookhoard: progress anchor verify failed for %s: %v", req.MediaItemID.String(), verr)
|
log.Printf("Bookhoard: progress anchor verify failed for %s: %v", req.MediaItemID.String(), verr)
|
||||||
} else {
|
} else {
|
||||||
if healed || (!params.Epubcfi.Valid && finalCFI != "") {
|
if healed || (!params.Epubcfi.Valid && finalCFI != "") {
|
||||||
@@ -470,9 +470,13 @@ func (s *ProgressService) SaveProgress(ctx context.Context, req SaveProgressRequ
|
|||||||
if healedPct != nil {
|
if healedPct != nil {
|
||||||
params.Percentage = pgtype.Float8{Float64: *healedPct, Valid: true}
|
params.Percentage = pgtype.Float8{Float64: *healedPct, Valid: true}
|
||||||
}
|
}
|
||||||
if charOff != nil {
|
|
||||||
params.CharacterOffset = pgtype.Int8{Int64: int64(*charOff), Valid: true}
|
|
||||||
}
|
}
|
||||||
|
// The verified book-wide offset refreshes the column on
|
||||||
|
// EVERY verified save — not only heals — so it never goes
|
||||||
|
// stale behind the anchor (and a block-relative offset is
|
||||||
|
// never written into the book-level column).
|
||||||
|
if bookOff != nil {
|
||||||
|
params.CharacterOffset = pgtype.Int8{Int64: int64(*bookOff), Valid: true}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user