fix(sync): normalize character offsets to UTF-16 at the wire; refresh book offset on every verified save
Offset currency policy, now explicit: EPUB CFI terminals, CRE text() offsets and the served char_offset handle are UTF-16 code units (the EPUB CFI spec, and what foliate/readium/KOReader/Kobo clients actually observe), while internal arithmetic — the book-wide character_offset column and percentage fractions — stays rune-based, consistent with TotalCharacters. For all-BMP books the currencies are identical, so no stored value changes; astral-plane text (emoji, rare CJK) no longer drifts. Boundaries converted: resolveCFIToNode interprets incoming CFI terminal offsets as UTF-16; textNodeAtUTF16Offset (née textNodeAtRuneOffset) interprets CRE text() offsets as UTF-16; buildCFI and buildCREXPointer emit UTF-16 terminals; blockCharOffset (the served char_offset) is UTF-16. Also fixes two character_offset column defects: heals wrote a BLOCK- relative offset into the book-wide column, and verified-but-unhealed saves (e.g. KOReader pushes) never refreshed it, leaving it stale behind the anchor. VerifyProgressAnchor now returns the verified book- wide rune offset and SaveProgress refreshes the column on every verified save. Tests: astral currency round trip (offset after an emoji must shift by one unit between currencies, in both heal and exact-verify directions) and book-offset ordering. The cmd/server/tests integration harness failures under docker (library folder 400 during setup) reproduce on the pre-change tree and are unrelated.
This commit is contained in:
+107
-23
@@ -431,7 +431,9 @@ func firstTextDescendant(n *html.Node) *html.Node {
|
||||
// textNodeAtRuneOffset walks text nodes under elem in document order and
|
||||
// returns the node containing the rune offset plus the local offset within
|
||||
// that node. Offsets beyond the end clamp to the last node.
|
||||
func textNodeAtRuneOffset(elem *html.Node, offset int) (*html.Node, int) {
|
||||
// textNodeAtUTF16Offset resolves a crengine text().N offset — UTF-16 code
|
||||
// units — to (text node, rune offset) within elem's text.
|
||||
func textNodeAtUTF16Offset(elem *html.Node, offset int) (*html.Node, int) {
|
||||
if offset < 0 {
|
||||
offset = 0
|
||||
}
|
||||
@@ -441,15 +443,15 @@ func textNodeAtRuneOffset(elem *html.Node, offset int) (*html.Node, int) {
|
||||
var walk func(*html.Node) bool
|
||||
walk = func(node *html.Node) bool {
|
||||
if node.Type == html.TextNode {
|
||||
length := utf8.RuneCountInString(node.Data)
|
||||
length := utf16Len(node.Data)
|
||||
if remaining < length {
|
||||
target = node
|
||||
local = remaining
|
||||
local = utf16ToRuneIndex(node.Data, remaining)
|
||||
return true
|
||||
}
|
||||
remaining -= length
|
||||
target = node
|
||||
local = length
|
||||
local = utf8.RuneCountInString(node.Data)
|
||||
return false
|
||||
}
|
||||
for child := node.FirstChild; child != nil; child = child.NextSibling {
|
||||
@@ -507,7 +509,7 @@ func (c *CFIConverter) convertByStructuralPath(body *html.Node, xp *CREXPointer,
|
||||
var textNode *html.Node
|
||||
var localOffset int
|
||||
if xp.CharOffset > 0 {
|
||||
textNode, localOffset = textNodeAtRuneOffset(elem, xp.CharOffset)
|
||||
textNode, localOffset = textNodeAtUTF16Offset(elem, xp.CharOffset)
|
||||
} else {
|
||||
textNode = firstTextDescendant(elem)
|
||||
localOffset = 0
|
||||
@@ -1058,24 +1060,96 @@ func indexChildNodes(parent *html.Node) []indexedNode {
|
||||
return nodes
|
||||
}
|
||||
|
||||
func findTextChunkIndex(parent *html.Node, textNode *html.Node) (int, int) {
|
||||
// Character-offset currency policy: the EPUB CFI spec and crengine both
|
||||
// count UTF-16 code units (JavaScript `.length` semantics — what foliate,
|
||||
// readium, KOReader and Kobo clients all observe), so every offset that
|
||||
// CROSSES the wire — CFI terminals, CRE text() offsets, the served
|
||||
// char_offset handle — is UTF-16. Internal arithmetic (book-level
|
||||
// character_offset, percentage fractions) stays rune-based, consistent
|
||||
// with TotalCharacters. These helpers convert at the boundaries; for
|
||||
// all-BMP text the two currencies are identical, so ASCII books are
|
||||
// unaffected.
|
||||
|
||||
// utf16Len returns the UTF-16 code-unit length of s.
|
||||
func utf16Len(s string) int {
|
||||
n := 0
|
||||
for _, r := range s {
|
||||
if r >= 0x10000 {
|
||||
n += 2
|
||||
} else {
|
||||
n++
|
||||
}
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
// utf16ToRuneIndex converts a UTF-16 code-unit offset within s to a rune
|
||||
// index (clamped to len(runes)).
|
||||
func utf16ToRuneIndex(s string, u16 int) int {
|
||||
if u16 <= 0 {
|
||||
return 0
|
||||
}
|
||||
units := 0
|
||||
i := 0
|
||||
for _, r := range s {
|
||||
if units >= u16 {
|
||||
return i
|
||||
}
|
||||
if r >= 0x10000 {
|
||||
units += 2
|
||||
} else {
|
||||
units++
|
||||
}
|
||||
i++
|
||||
}
|
||||
return i
|
||||
}
|
||||
|
||||
// runeToUTF16Index converts a rune index within s to a UTF-16 code-unit
|
||||
// offset (clamped to the string's unit length).
|
||||
func runeToUTF16Index(s string, runeIdx int) int {
|
||||
if runeIdx <= 0 {
|
||||
return 0
|
||||
}
|
||||
units := 0
|
||||
i := 0
|
||||
for _, r := range s {
|
||||
if i >= runeIdx {
|
||||
return units
|
||||
}
|
||||
if r >= 0x10000 {
|
||||
units += 2
|
||||
} else {
|
||||
units++
|
||||
}
|
||||
i++
|
||||
}
|
||||
return units
|
||||
}
|
||||
|
||||
// findTextChunk locates the indexed text chunk containing textNode and
|
||||
// returns its chunk index, the rune offset of the node within the chunk,
|
||||
// and the chunk text up to and including the node (for UTF-16 conversion
|
||||
// of chunk-relative offsets).
|
||||
func findTextChunk(parent *html.Node, textNode *html.Node) (int, int, string) {
|
||||
indexed := indexChildNodes(parent)
|
||||
for i, node := range indexed {
|
||||
if node.isTextChunk() {
|
||||
for j, tn := range node.textChunk {
|
||||
var sb strings.Builder
|
||||
chunkOffset := 0
|
||||
for _, tn := range node.textChunk {
|
||||
if tn == textNode {
|
||||
chunkOffset := 0
|
||||
for k := 0; k < j; k++ {
|
||||
chunkOffset += utf8.RuneCountInString(node.textChunk[k].Data)
|
||||
}
|
||||
return i, chunkOffset
|
||||
return i, chunkOffset, sb.String() + textNode.Data
|
||||
}
|
||||
chunkOffset += utf8.RuneCountInString(tn.Data)
|
||||
sb.WriteString(tn.Data)
|
||||
}
|
||||
}
|
||||
}
|
||||
return -1, 0
|
||||
return -1, 0, ""
|
||||
}
|
||||
|
||||
|
||||
func findElementCFIIndex(parent *html.Node, element *html.Node) int {
|
||||
indexed := indexChildNodes(parent)
|
||||
for i, node := range indexed {
|
||||
@@ -1110,15 +1184,17 @@ func buildCFI(spineIndex int, textNode *html.Node, charOffset int) (string, erro
|
||||
return "", fmt.Errorf("text node has no parent")
|
||||
}
|
||||
|
||||
chunkIdx, chunkOffset := findTextChunkIndex(parent, textNode)
|
||||
chunkIdx, chunkOffset, chunkText := findTextChunk(parent, textNode)
|
||||
if chunkIdx == -1 {
|
||||
return "", fmt.Errorf("text node not found in parent's indexed children")
|
||||
}
|
||||
|
||||
// The CFI terminal offset is UTF-16 code units (spec currency); the
|
||||
// internal charOffset is runes. Convert over the chunk text.
|
||||
totalOffset := chunkOffset + charOffset
|
||||
|
||||
var parts []string
|
||||
parts = append(parts, fmt.Sprintf("/%d:%d", chunkIdx, totalOffset))
|
||||
parts = append(parts, fmt.Sprintf("/%d:%d", chunkIdx, runeToUTF16Index(chunkText, totalOffset)))
|
||||
|
||||
current := parent
|
||||
for current != nil {
|
||||
@@ -1519,24 +1595,31 @@ func resolveCFIToNode(doc *html.Node, steps []cfiStep) (*html.Node, int, error)
|
||||
entry := indexed[lastStep.Index]
|
||||
|
||||
if entry.isTextChunk() {
|
||||
textOffset := 0
|
||||
// The CFI terminal offset arrives in UTF-16 code units (spec
|
||||
// currency — foliate/readium/KOReader/Kobo all emit UTF-16).
|
||||
// Walk the chunk in UTF-16 units, then convert the hit position
|
||||
// to the internal rune offset.
|
||||
u16Remaining := 0
|
||||
if lastStep.HasOffset {
|
||||
textOffset = lastStep.Offset
|
||||
u16Remaining = lastStep.Offset
|
||||
}
|
||||
|
||||
var targetNode *html.Node
|
||||
remainingOffset := textOffset
|
||||
runeIntoTarget := 0
|
||||
for _, tn := range entry.textChunk {
|
||||
textLen := utf8.RuneCountInString(tn.Data)
|
||||
if remainingOffset < textLen || (remainingOffset == textLen && targetNode == nil) {
|
||||
units := utf16Len(tn.Data)
|
||||
if u16Remaining < units || (u16Remaining == units && targetNode == nil) {
|
||||
targetNode = tn
|
||||
runeIntoTarget = utf16ToRuneIndex(tn.Data, u16Remaining)
|
||||
break
|
||||
}
|
||||
remainingOffset -= textLen
|
||||
u16Remaining -= units
|
||||
targetNode = tn
|
||||
runeIntoTarget = utf8.RuneCountInString(tn.Data)
|
||||
}
|
||||
if targetNode == nil && len(entry.textChunk) > 0 {
|
||||
targetNode = entry.textChunk[len(entry.textChunk)-1]
|
||||
runeIntoTarget = utf8.RuneCountInString(targetNode.Data)
|
||||
}
|
||||
|
||||
parent := targetNode.Parent
|
||||
@@ -1547,7 +1630,7 @@ func resolveCFIToNode(doc *html.Node, steps []cfiStep) (*html.Node, int, error)
|
||||
}
|
||||
totalOffset += countTextChars(c)
|
||||
}
|
||||
totalOffset += remainingOffset
|
||||
totalOffset += runeIntoTarget
|
||||
|
||||
return targetNode, totalOffset, nil
|
||||
}
|
||||
@@ -1606,7 +1689,8 @@ func buildCREXPointer(spineIndex int, node *html.Node, charOffset int) (string,
|
||||
xpointer := fmt.Sprintf("/body/DocFragment[%d]/body%s", fragIndex, strings.Join(parts, ""))
|
||||
|
||||
if charOffset > 0 || (node.Type == html.TextNode) {
|
||||
xpointer += fmt.Sprintf("/text().%d", charOffset)
|
||||
// crengine counts UTF-16 code units; charOffset is internal runes.
|
||||
xpointer += fmt.Sprintf("/text().%d", runeToUTF16Index(node.Data, charOffset))
|
||||
}
|
||||
|
||||
return xpointer, nil
|
||||
|
||||
Reference in New Issue
Block a user