perf(sync): share parsed EPUBs across conversions, make converters concurrency-safe
ConvertToCanonical/ConvertFromCanonical built a fresh CFIConverter per call, and each annotation converts twice (pos0+pos1) — a book with 200 highlights re-opened and re-parsed the EPUB 400+ times per sync, and again per metadata pull. A bounded 8-entry cache keyed by path now shares converters (the parsing work belongs on the server; clients stay thin). CFIConverter gained a mutex around its lazily built spine/doc caches since instances are now shared between concurrent requests. Adds CFIConverter.SectionPercentage: book-wide percentage for a CRE xpointer from the spine char distribution (midpoint of its document) — the server-side counterpart to dropping per-annotation getPageFromXPointer lookups from the plugin.
This commit is contained in:
@@ -11,6 +11,7 @@ import (
|
||||
"regexp"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
"unicode/utf8"
|
||||
|
||||
"golang.org/x/net/html"
|
||||
@@ -19,6 +20,9 @@ import (
|
||||
type CFIConverter struct {
|
||||
epubPath string
|
||||
cache *spineCache
|
||||
// mu guards the lazily-built spine/doc caches: converter instances are
|
||||
// shared across concurrent requests via the package cache in locators.go.
|
||||
mu sync.Mutex
|
||||
}
|
||||
|
||||
type spineItem struct {
|
||||
@@ -37,6 +41,8 @@ func NewCFIConverter(epubPath string) *CFIConverter {
|
||||
}
|
||||
|
||||
func (c *CFIConverter) loadSpine() (*spineCache, error) {
|
||||
c.mu.Lock()
|
||||
defer c.mu.Unlock()
|
||||
if c.cache != nil {
|
||||
return c.cache, nil
|
||||
}
|
||||
@@ -94,6 +100,8 @@ func (c *CFIConverter) getContentDoc(fragmentIndex int) (*html.Node, string, err
|
||||
item := spine.items[spineIndex]
|
||||
href := item.href
|
||||
|
||||
c.mu.Lock()
|
||||
defer c.mu.Unlock()
|
||||
if cached, ok := spine.docCache[href]; ok {
|
||||
return cached, href, nil
|
||||
}
|
||||
@@ -243,6 +251,46 @@ type ConversionResult struct {
|
||||
Precision string
|
||||
}
|
||||
|
||||
// SectionPercentage derives an approximate book-wide percentage for a CRE
|
||||
// xpointer from the char distribution across the spine: the midpoint of the
|
||||
// document it points into. Precision is per-section, which is what
|
||||
// percentage_start is used for (ordering/filtering) — and it lets thin
|
||||
// clients skip their own per-annotation page lookups entirely.
|
||||
func (c *CFIConverter) SectionPercentage(xpointer string) float64 {
|
||||
xp, err := ParseCREXPointer(xpointer)
|
||||
if err != nil {
|
||||
return 0
|
||||
}
|
||||
spine, err := c.loadSpine()
|
||||
if err != nil {
|
||||
return 0
|
||||
}
|
||||
total := 0
|
||||
charCounts := make([]int, len(spine.items))
|
||||
for i := range spine.items {
|
||||
doc, _, docErr := c.getContentDoc(i + 1)
|
||||
if docErr != nil {
|
||||
continue
|
||||
}
|
||||
if b := findBody(doc); b != nil {
|
||||
charCounts[i] = countTextChars(b)
|
||||
total += charCounts[i]
|
||||
}
|
||||
}
|
||||
if total <= 0 {
|
||||
return 0
|
||||
}
|
||||
idx := xp.FragmentIndex - 1
|
||||
if idx < 0 || idx >= len(spine.items) {
|
||||
return 0
|
||||
}
|
||||
before := 0
|
||||
for i := 0; i < idx; i++ {
|
||||
before += charCounts[i]
|
||||
}
|
||||
return (float64(before) + float64(charCounts[idx])/2) / float64(total)
|
||||
}
|
||||
|
||||
func (c *CFIConverter) ConvertCREToStandard(xpointer string, storedPercentage float64, contextText string) (*ConversionResult, error) {
|
||||
if IsCREFragmentID(xpointer) {
|
||||
return c.convertFragmentID(xpointer, storedPercentage)
|
||||
@@ -884,8 +932,8 @@ func readZipFile(zr *zip.Reader, name string) ([]byte, error) {
|
||||
}
|
||||
|
||||
type opfContainer struct {
|
||||
XMLName xml.Name `xml:"container"`
|
||||
RootFiles []opfRoot `xml:"rootfiles>rootfile"`
|
||||
XMLName xml.Name `xml:"container"`
|
||||
RootFiles []opfRoot `xml:"rootfiles>rootfile"`
|
||||
}
|
||||
|
||||
type opfRoot struct {
|
||||
@@ -906,8 +954,8 @@ func extractOPFPath(data []byte) (string, error) {
|
||||
}
|
||||
|
||||
type xmlPackage struct {
|
||||
XMLName xml.Name `xml:"package"`
|
||||
Spine xmlSpine `xml:"spine"`
|
||||
XMLName xml.Name `xml:"package"`
|
||||
Spine xmlSpine `xml:"spine"`
|
||||
Manifest xmlManifest `xml:"manifest"`
|
||||
}
|
||||
|
||||
@@ -1036,9 +1084,9 @@ func preprocessXHTML(input string) string {
|
||||
}
|
||||
|
||||
type cfiStep struct {
|
||||
Index int
|
||||
ID string
|
||||
Offset int
|
||||
Index int
|
||||
ID string
|
||||
Offset int
|
||||
HasOffset bool
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user