fix(sync): normalize character offsets to UTF-16 at the wire; refresh book offset on every verified save

Offset currency policy, now explicit: EPUB CFI terminals, CRE text()
offsets and the served char_offset handle are UTF-16 code units (the
EPUB CFI spec, and what foliate/readium/KOReader/Kobo clients actually
observe), while internal arithmetic — the book-wide character_offset
column and percentage fractions — stays rune-based, consistent with
TotalCharacters. For all-BMP books the currencies are identical, so no
stored value changes; astral-plane text (emoji, rare CJK) no longer
drifts.

Boundaries converted: resolveCFIToNode interprets incoming CFI terminal
offsets as UTF-16; textNodeAtUTF16Offset (née textNodeAtRuneOffset)
interprets CRE text() offsets as UTF-16; buildCFI and buildCREXPointer
emit UTF-16 terminals; blockCharOffset (the served char_offset) is
UTF-16.

Also fixes two character_offset column defects: heals wrote a BLOCK-
relative offset into the book-wide column, and verified-but-unhealed
saves (e.g. KOReader pushes) never refreshed it, leaving it stale
behind the anchor. VerifyProgressAnchor now returns the verified book-
wide rune offset and SaveProgress refreshes the column on every
verified save.

Tests: astral currency round trip (offset after an emoji must shift by
one unit between currencies, in both heal and exact-verify directions)
and book-offset ordering. The cmd/server/tests integration harness
failures under docker (library folder 400 during setup) reproduce on
the pre-change tree and are unrelated.
This commit is contained in:
John O'Keefe
2026-09-26 20:18:48 -04:00
parent aec226af1a
commit 8c3273a0fc
7 changed files with 273 additions and 61 deletions
+107 -23
View File
@@ -431,7 +431,9 @@ func firstTextDescendant(n *html.Node) *html.Node {
// textNodeAtRuneOffset walks text nodes under elem in document order and
// returns the node containing the rune offset plus the local offset within
// that node. Offsets beyond the end clamp to the last node.
func textNodeAtRuneOffset(elem *html.Node, offset int) (*html.Node, int) {
// textNodeAtUTF16Offset resolves a crengine text().N offset — UTF-16 code
// units — to (text node, rune offset) within elem's text.
func textNodeAtUTF16Offset(elem *html.Node, offset int) (*html.Node, int) {
if offset < 0 {
offset = 0
}
@@ -441,15 +443,15 @@ func textNodeAtRuneOffset(elem *html.Node, offset int) (*html.Node, int) {
var walk func(*html.Node) bool
walk = func(node *html.Node) bool {
if node.Type == html.TextNode {
length := utf8.RuneCountInString(node.Data)
length := utf16Len(node.Data)
if remaining < length {
target = node
local = remaining
local = utf16ToRuneIndex(node.Data, remaining)
return true
}
remaining -= length
target = node
local = length
local = utf8.RuneCountInString(node.Data)
return false
}
for child := node.FirstChild; child != nil; child = child.NextSibling {
@@ -507,7 +509,7 @@ func (c *CFIConverter) convertByStructuralPath(body *html.Node, xp *CREXPointer,
var textNode *html.Node
var localOffset int
if xp.CharOffset > 0 {
textNode, localOffset = textNodeAtRuneOffset(elem, xp.CharOffset)
textNode, localOffset = textNodeAtUTF16Offset(elem, xp.CharOffset)
} else {
textNode = firstTextDescendant(elem)
localOffset = 0
@@ -1058,24 +1060,96 @@ func indexChildNodes(parent *html.Node) []indexedNode {
return nodes
}
func findTextChunkIndex(parent *html.Node, textNode *html.Node) (int, int) {
// Character-offset currency policy: the EPUB CFI spec and crengine both
// count UTF-16 code units (JavaScript `.length` semantics — what foliate,
// readium, KOReader and Kobo clients all observe), so every offset that
// CROSSES the wire — CFI terminals, CRE text() offsets, the served
// char_offset handle — is UTF-16. Internal arithmetic (book-level
// character_offset, percentage fractions) stays rune-based, consistent
// with TotalCharacters. These helpers convert at the boundaries; for
// all-BMP text the two currencies are identical, so ASCII books are
// unaffected.
// utf16Len returns the UTF-16 code-unit length of s.
func utf16Len(s string) int {
n := 0
for _, r := range s {
if r >= 0x10000 {
n += 2
} else {
n++
}
}
return n
}
// utf16ToRuneIndex converts a UTF-16 code-unit offset within s to a rune
// index (clamped to len(runes)).
func utf16ToRuneIndex(s string, u16 int) int {
if u16 <= 0 {
return 0
}
units := 0
i := 0
for _, r := range s {
if units >= u16 {
return i
}
if r >= 0x10000 {
units += 2
} else {
units++
}
i++
}
return i
}
// runeToUTF16Index converts a rune index within s to a UTF-16 code-unit
// offset (clamped to the string's unit length).
func runeToUTF16Index(s string, runeIdx int) int {
if runeIdx <= 0 {
return 0
}
units := 0
i := 0
for _, r := range s {
if i >= runeIdx {
return units
}
if r >= 0x10000 {
units += 2
} else {
units++
}
i++
}
return units
}
// findTextChunk locates the indexed text chunk containing textNode and
// returns its chunk index, the rune offset of the node within the chunk,
// and the chunk text up to and including the node (for UTF-16 conversion
// of chunk-relative offsets).
func findTextChunk(parent *html.Node, textNode *html.Node) (int, int, string) {
indexed := indexChildNodes(parent)
for i, node := range indexed {
if node.isTextChunk() {
for j, tn := range node.textChunk {
var sb strings.Builder
chunkOffset := 0
for _, tn := range node.textChunk {
if tn == textNode {
chunkOffset := 0
for k := 0; k < j; k++ {
chunkOffset += utf8.RuneCountInString(node.textChunk[k].Data)
}
return i, chunkOffset
return i, chunkOffset, sb.String() + textNode.Data
}
chunkOffset += utf8.RuneCountInString(tn.Data)
sb.WriteString(tn.Data)
}
}
}
return -1, 0
return -1, 0, ""
}
func findElementCFIIndex(parent *html.Node, element *html.Node) int {
indexed := indexChildNodes(parent)
for i, node := range indexed {
@@ -1110,15 +1184,17 @@ func buildCFI(spineIndex int, textNode *html.Node, charOffset int) (string, erro
return "", fmt.Errorf("text node has no parent")
}
chunkIdx, chunkOffset := findTextChunkIndex(parent, textNode)
chunkIdx, chunkOffset, chunkText := findTextChunk(parent, textNode)
if chunkIdx == -1 {
return "", fmt.Errorf("text node not found in parent's indexed children")
}
// The CFI terminal offset is UTF-16 code units (spec currency); the
// internal charOffset is runes. Convert over the chunk text.
totalOffset := chunkOffset + charOffset
var parts []string
parts = append(parts, fmt.Sprintf("/%d:%d", chunkIdx, totalOffset))
parts = append(parts, fmt.Sprintf("/%d:%d", chunkIdx, runeToUTF16Index(chunkText, totalOffset)))
current := parent
for current != nil {
@@ -1519,24 +1595,31 @@ func resolveCFIToNode(doc *html.Node, steps []cfiStep) (*html.Node, int, error)
entry := indexed[lastStep.Index]
if entry.isTextChunk() {
textOffset := 0
// The CFI terminal offset arrives in UTF-16 code units (spec
// currency — foliate/readium/KOReader/Kobo all emit UTF-16).
// Walk the chunk in UTF-16 units, then convert the hit position
// to the internal rune offset.
u16Remaining := 0
if lastStep.HasOffset {
textOffset = lastStep.Offset
u16Remaining = lastStep.Offset
}
var targetNode *html.Node
remainingOffset := textOffset
runeIntoTarget := 0
for _, tn := range entry.textChunk {
textLen := utf8.RuneCountInString(tn.Data)
if remainingOffset < textLen || (remainingOffset == textLen && targetNode == nil) {
units := utf16Len(tn.Data)
if u16Remaining < units || (u16Remaining == units && targetNode == nil) {
targetNode = tn
runeIntoTarget = utf16ToRuneIndex(tn.Data, u16Remaining)
break
}
remainingOffset -= textLen
u16Remaining -= units
targetNode = tn
runeIntoTarget = utf8.RuneCountInString(tn.Data)
}
if targetNode == nil && len(entry.textChunk) > 0 {
targetNode = entry.textChunk[len(entry.textChunk)-1]
runeIntoTarget = utf8.RuneCountInString(targetNode.Data)
}
parent := targetNode.Parent
@@ -1547,7 +1630,7 @@ func resolveCFIToNode(doc *html.Node, steps []cfiStep) (*html.Node, int, error)
}
totalOffset += countTextChars(c)
}
totalOffset += remainingOffset
totalOffset += runeIntoTarget
return targetNode, totalOffset, nil
}
@@ -1606,7 +1689,8 @@ func buildCREXPointer(spineIndex int, node *html.Node, charOffset int) (string,
xpointer := fmt.Sprintf("/body/DocFragment[%d]/body%s", fragIndex, strings.Join(parts, ""))
if charOffset > 0 || (node.Type == html.TextNode) {
xpointer += fmt.Sprintf("/text().%d", charOffset)
// crengine counts UTF-16 code units; charOffset is internal runes.
xpointer += fmt.Sprintf("/text().%d", runeToUTF16Index(node.Data, charOffset))
}
return xpointer, nil