diff --git a/main.lua b/main.lua index e78fda9..7d88fbf 100644 --- a/main.lua +++ b/main.lua @@ -764,23 +764,81 @@ function Bookhoard:getContextText() local xp = self:getLastProgress() if not xp then return "" end - local text = self.ui.document:getTextFromXPointer(xp) - if not text or text == "" then - text = self.ui.document:getTextFromXPointer(self.ui.document:getNormalizedXPointer(xp)) + local function clean(s) + if not s or s == "" then return "" end + if #s > 100 then + s = s:sub(1, 100) + end + return s:gsub("%s+", " "):match("^%s*(.-)%s*$") or "" + end + + local function tryPointer(p) + if not p or p == "" then return "" end + local ok, text = pcall(function() + return self.ui.document:getTextFromXPointer(p) + end) + if not ok then return "" end + return text or "" + end + + local text = tryPointer(xp) + if text == "" then + local ok, nxp = pcall(function() + return self.ui.document:getNormalizedXPointer(xp) + end) + if ok and nxp and nxp ~= "" and nxp ~= xp then + text = tryPointer(nxp) + end + end + if text == "" then return "" end + + local cleaned = clean(text) + -- Single text nodes split by inline markup (e.g. drop-caps like + --
Convergenceā¦
) return just "C" here. That one + -- character then false-positives the server text search to the first + -- "C" in the chapter (doc start). Walk up to the enclosing block + -- element(s) so we capture words, not the node. Kindle-light: at most + -- 3 extra local reads, only on this tiny-text path, zero extra network. + local function runeLen(s) + -- Cheap UTF-8 aware length without dependencies. + local _, n = s:gsub("[^\128-\191]", "") + return n + end + if cleaned ~= "" and runeLen(cleaned) < 15 then + local parent = xp + for _ = 1, 3 do + -- Strip one trailing path segment: "/text().N" or "/span[1]" etc. + local stripped = parent:gsub("/[^/]+$", "") + if not stripped or stripped == "" or stripped == parent then + break + end + parent = stripped + -- Don't walk past the document body; element pointers above the + -- block would return whole-chapter text. + if parent:match("/body%s*$") or parent:match("DocFragment%[%d+%]$") then + break + end + local ptext = tryPointer(parent) + if ptext and ptext ~= "" then + local pcleaned = clean(ptext) + if pcleaned ~= "" and runeLen(pcleaned) >= 15 then + return pcleaned + end + -- Keep the longest thing seen in case no level reaches 15. + if runeLen(pcleaned) > runeLen(cleaned) then + cleaned = pcleaned + end + end + end + return cleaned end - if not text or text == "" then return "" end local char_offset = tonumber(xp:match("text%(%)%.?(%d+)")) or 0 if char_offset > 0 and char_offset < #text then text = text:sub(char_offset + 1) end - if #text > 100 then - text = text:sub(1, 100) - end - - text = text:gsub("%s+", " "):match("^%s*(.-)%s*$") or "" - return text + return clean(text) end function Bookhoard:collectBookData()