diff --git a/web/src/reader/parsers/epub-parsers.ts b/web/src/reader/parsers/epub-parsers.ts new file mode 100644 index 0000000..f9f5412 --- /dev/null +++ b/web/src/reader/parsers/epub-parsers.ts @@ -0,0 +1,307 @@ +// EPUB Parser - Converts EPUB 2/3 to Common Intermediate Format +// Procedural style: Functions, not classes + +import JSZip from 'jszip'; + +// ============================================================ +// Main Parse Function +// ============================================================ + +export async function parseEPUB(epubBlob: Blob): Promise { + const zip = await JSZip.loadAsync(epubBlob); + + // Parse container.xml to find OPF file + const containerXml = await getZipFileContent(zip, 'META-INF/container.xml'); + const opfPath = extractOPFPath(containerXml); + + if (!opfPath) { + throw new Error('Invalid EPUB: no OPF file found'); + } + + // Parse OPF file + const opfXml = await getZipFileContent(zip, opfPath); + const packageDoc = parseXML(opfXml); + + // Extract all components + const metadata = extractMetadata(packageDoc); + const spine = parseSpine(packageDoc); + const toc = await parseTOC(zip, packageDoc, opfPath); + const resources = await loadResources(zip); + const coverImage = await extractCover(zip, packageDoc); + + // Calculate locations (minimal - backend handles detailed tracking) + const totalCharacters = await calculateTotalCharacters(spine, resources); + + return { + metadata, + toc, + spine, + resources, + locations: { + totalCharacters, + estimatedPages: Math.ceil(totalCharacters / 1500), + }, + }; +} + +// ============================================================ +// Helper Functions +// ============================================================ + +async function getZipFileContent(zip: JSZip, path: string): Promise { + const file = zip.file(path); + if (!file) { + throw new Error(`File not found: ${path}`); + } + return await file.async('text'); +} + +function parseXML(xmlString: string): XMLDocument { + const parser = new DOMParser(); + return parser.parseFromString(xmlString, 'text/xml'); +} + +function extractOPFPath(containerXml: string): string | null { + const containerDoc = parseXML(containerXml); + return containerDoc.querySelector('rootfile')?.getAttribute('full-path') || null; +} + +function extractMetadata(packageDoc: XMLDocument): EbookCIF['metadata'] { + const metadata = packageDoc.querySelector('metadata'); + if (!metadata) { + throw new Error('No metadata found in OPF'); + } + + return { + title: metadata.querySelector('title')?.textContent || '', + author: metadata.querySelector('creator')?.textContent || '', + language: metadata.querySelector('language')?.textContent || 'en', + publisher: metadata.querySelector('publisher')?.textContent || undefined, + isbn: metadata.querySelector('identifier')?.textContent || undefined, + }; +} + +function parseSpine(packageDoc: XMLDocument): EbookCIF['spine'] { + const spine = packageDoc.querySelector('spine'); + const manifest = packageDoc.querySelector('manifest'); + + if (!spine || !manifest) { + throw new Error('No spine or manifest found in OPF'); + } + + const spineItems = spine.querySelectorAll('itemref'); + const result: EbookCIF['spine'] = []; + + spineItems.forEach((itemref) => { + const idref = itemref.getAttribute('idref'); + if (!idref) return; + + const manifestItem = manifest.querySelector(`[id="${idref}"]`); + if (!manifestItem) return; + + const href = manifestItem.getAttribute('href'); + if (!href) return; + + result.push({ + id: idref, + type: 'html', + content: href, + properties: itemref.getAttribute('properties') || undefined, + }); + }); + + return result; +} + +async function parseTOC(zip: JSZip, packageDoc: XMLDocument, opfPath: string): Promise { + // Try EPUB 3.0 navigation document first + const navItem = packageDoc.querySelector('manifest item[properties~="nav"]'); + if (navItem) { + const navHref = navItem.getAttribute('href'); + if (navHref) { + const navPath = resolvePath(opfPath, navHref); + return parseNavTOC(zip, navPath); + } + } + + // Fallback to EPUB 2.0 NCX + const ncxId = spine?.getAttribute('toc'); + if (ncxId) { + const ncxItem = packageDoc.querySelector(`manifest [id="${ncxId}"]`); + if (ncxItem) { + const ncxHref = ncxItem.getAttribute('href'); + if (ncxHref) { + const ncxPath = resolvePath(opfPath, ncxHref); + return parseNCXTOC(zip, ncxPath); + } + } + } + + return []; +} + +async function parseNavTOC(zip: JSZip, navPath: string): Promise { + const navXml = await getZipFileContent(zip, navPath); + const navDoc = parseXML(navXml); + const nav = navDoc.querySelector('nav'); + + if (!nav) return []; + + const ol = nav.querySelector('ol'); + if (!ol) return []; + + const items = ol.querySelectorAll(':scope > li'); + const result: EbookCIF['toc'] = []; + + for (const li of items) { + const link = li.querySelector('a'); + if (link) { + result.push({ + id: link.getAttribute('href') || '', + title: link.textContent || '', + href: link.getAttribute('href') || '', + children: [], + }); + } + } + + return result; +} + +async function parseNCXTOC(zip: JSZip, ncxPath: string): Promise { + const ncxXml = await getZipFileContent(zip, ncxPath); + const ncxDoc = parseXML(ncxXml); + const navMap = ncxDoc.querySelector('navMap'); + + if (!navMap) return []; + + return parseNCXNode(navMap); +} + +function parseNCXNode(node: Element): EbookCIF['toc'] { + const navPoints = node.querySelectorAll(':scope > navPoint'); + const result: EbookCIF['toc'] = []; + + navPoints.forEach((navPoint) => { + const label = navPoint.querySelector('navLabel text')?.textContent || ''; + const content = navPoint.querySelector('content'); + const href = content?.getAttribute('src') || ''; + + result.push({ + id: href, + title: label, + href, + children: parseNCXNode(navPoint), + }); + }); + + return result; +} + +async function loadResources(zip: JSZip): Promise> { + const resources = new Map(); + const files = Object.keys(zip.files); + + for (const path of files) { + const file = zip.file(path); + if (file && !file.dir) { + const blob = await file.async('blob'); + resources.set(path, blob); + } + } + + return resources; +} + +async function extractCover(zip: JSZip, packageDoc: XMLDocument): Promise { + // Try cover-id metadata + const coverId = packageDoc.querySelector('meta[name="cover"]')?.getAttribute('content'); + if (coverId) { + const coverItem = packageDoc.querySelector(`manifest [id="${coverId}"]`); + if (coverItem) { + const coverHref = coverItem.getAttribute('href'); + if (coverHref) { + const coverFile = zip.file(coverHref); + if (coverFile) { + return await coverFile.async('blob'); + } + } + } + } + + // Fallback: look for cover image in manifest + const coverItem = packageDoc.querySelector('manifest item[properties~="cover-image"]'); + if (coverItem) { + const coverHref = coverItem.getAttribute('href'); + if (coverHref) { + const coverFile = zip.file(coverHref); + if (coverFile) { + return await coverFile.async('blob'); + } + } + } + + return undefined; +} + +function resolvePath(basePath: string, relativePath: string): string { + const baseDir = basePath.substring(0, basePath.lastIndexOf('/') + 1); + return baseDir + relativePath; +} + +async function calculateTotalCharacters(spine: EbookCIF['spine'], resources: Map): Promise { + let total = 0; + + for (const item of spine) { + if (item.type === 'html') { + const content = resources.get(item.content); + if (content) { + const text = await content.text(); + total += text.length; + } + } + } + + return total; +} + +function resolvePath(basePath: string, relativePath: string): string { + const baseDir = basePath.substring(0, basePath.lastIndexOf('/') + 1); + return baseDir + relativePath; +} + } + } + + return total; +} + +function generatePageBreaks(totalCharacters: number): number[] { + const breaks: number[] = []; + const charsPerPage = 1000; // Rough estimate + + for (let i = charsPerPage; i < totalCharacters; i += charsPerPage) { + breaks.push(i); + } + + return breaks; +} + +// ============================================================ +// Metadata Quick Extract (for library view) +// ============================================================ + +export async function extractEPUBMetadata(epubBlob: Blob): Promise> { + const zip = await JSZip.loadAsync(epubBlob); + + const containerXml = await getZipFileContent(zip, 'META-INF/container.xml'); + const opfPath = extractOPFPath(containerXml); + + if (!opfPath) { + return {}; + } + + const opfXml = await getZipFileContent(zip, opfPath); + const packageDoc = parseXML(opfXml); + + return extractMetadata(packageDoc); +} diff --git a/web/src/reader/parsers/fb2-parser.ts b/web/src/reader/parsers/fb2-parser.ts new file mode 100644 index 0000000..ecfe25f --- /dev/null +++ b/web/src/reader/parsers/fb2-parser.ts @@ -0,0 +1,244 @@ +// FB2 Parser - Converts FictionBook 2 to Common Intermediate Format +// FB2 is XML-based, similar to EPUB structure +// Procedural style: Functions, not classes + +import JSZip from "jszip"; + +// ============================================================ +// Main Parse Function +// ============================================================ + +export async function parseFB2(fb2Blob: Blob): Promise { + // FB2 can be plain XML or zipped (.fb2.zip) + let xmlContent: string; + + if ( + fb2Blob.type === "application/zip" || + fb2Blob.type === "application/x-zip-compressed" + ) { + const zip = await JSZip.loadAsync(fb2Blob); + const files = Object.keys(zip.files); + + // Find the first .fb2 file in the zip + const fb2File = files.find((f) => f.endsWith(".fb2")); + if (!fb2File) { + throw new Error("No .fb2 file found in archive"); + } + + xmlContent = await zip.file(fb2File)!.async("text"); + } else { + xmlContent = await fb2Blob.text(); + } + + const xmlDoc = parseXML(xmlContent); + + const metadata = extractFB2Metadata(xmlDoc); + const toc = parseFB2TOC(xmlDoc); + const spine = createFB2Spine(xmlDoc); + const resources = await extractFB2Resources(xmlDoc, fb2Blob); + + // Calculate locations (minimal - backend handles detailed tracking) + const totalCharacters = calculateFB2Characters(xmlDoc); + + return { + metadata, + toc, + spine, + resources, + locations: { + totalCharacters, + estimatedPages: Math.ceil(totalCharacters / 1500), + }, + }; +} + +// ============================================================ +// Helper Functions +// ============================================================ + +function parseXML(xmlString: string): XMLDocument { + const parser = new DOMParser(); + return parser.parseFromString(xmlString, "text/xml"); +} + +function extractFB2Metadata(xmlDoc: XMLDocument): EbookCIF["metadata"] { + const titleInfo = xmlDoc.querySelector("title-info"); + const documentInfo = xmlDoc.querySelector("document-info"); + + if (!titleInfo) { + throw new Error("Invalid FB2: no title-info found"); + } + + return { + title: titleInfo.querySelector("book-title")?.textContent || "", + author: extractFB2Author(titleInfo), + language: titleInfo.querySelector("lang")?.textContent || "en", + publisher: + documentInfo?.querySelector("publisher")?.textContent || undefined, + isbn: undefined, // FB2 doesn't typically have ISBN + }; +} + +function extractFB2Author(titleInfo: Element): string { + const author = titleInfo.querySelector("author"); + if (!author) return ""; + + const firstName = author.querySelector("first-name")?.textContent || ""; + const lastName = author.querySelector("last-name")?.textContent || ""; + const middleName = author.querySelector("middle-name")?.textContent || ""; + + const parts = [firstName, middleName, lastName].filter(Boolean); + return parts.join(" ") || "Unknown"; +} + +function parseFB2TOC(xmlDoc: XMLDocument): EbookCIF["toc"] { + const toc: EbookCIF["toc"] = []; + const body = xmlDoc.querySelector("body"); + + if (!body) return toc; + + const sections = body.querySelectorAll(":scope > section"); + let sectionIndex = 0; + + for (const section of sections) { + const title = section.querySelector("title"); + const titleText = + title?.textContent.trim() || `Section ${sectionIndex + 1}`; + + toc.push({ + id: `section-${sectionIndex}`, + title: titleText, + href: `#section-${sectionIndex}`, + children: [], + }); + + sectionIndex++; + } + + return toc; +} + +function createFB2Spine(xmlDoc: XMLDocument): EbookCIF["spine"] { + const spine: EbookCIF["spine"] = []; + const body = xmlDoc.querySelector("body"); + + if (!body) return spine; + + // Convert each section to HTML + const sections = body.querySelectorAll(":scope > section"); + + sections.forEach((section, index) => { + const htmlContent = convertFB2SectionToHTML(section, index); + + spine.push({ + id: `section-${index}`, + type: "html", + content: htmlContent, + index, + }); + }); + + return spine; +} + +function convertFB2SectionToHTML(section: Element, index: number): string { + const title = section.querySelector("title"); + let html = `
`; + + if (title) { + html += `

${title.textContent}

`; + } + + // Convert paragraphs + const paragraphs = section.querySelectorAll("p"); + paragraphs.forEach((p) => { + html += `

${p.innerHTML}

`; + }); + + // Convert images + const images = section.querySelectorAll("image"); + images.forEach((img) => { + const href = img.getAttribute("l:href"); + const alt = img.getAttribute("alt") || ""; + if (href) { + html += `${alt}`; + } + }); + + html += "
"; + + return html; +} + +async function extractFB2Resources( + xmlDoc: XMLDocument, + fb2Blob: Blob, +): Promise> { + const resources = new Map(); + + // FB2 can have embedded images (base64) or external references + const binary = xmlDoc.querySelector("binary"); + if (binary) { + const contentType = binary.getAttribute("content-type"); + const id = binary.getAttribute("id"); + + if (contentType && id && binary.textContent) { + // Decode base64 + const base64Data = binary.textContent.trim(); + const byteString = atob(base64Data); + const byteArray = new Uint8Array(byteString.length); + + for (let i = 0; i < byteString.length; i++) { + byteArray[i] = byteString.charCodeAt(i); + } + + const blob = new Blob([byteArray], { type: contentType }); + resources.set(`#${id}`, blob); + } + } + + return resources; +} + +function calculateFB2Characters(xmlDoc: XMLDocument): number { + const body = xmlDoc.querySelector("body"); + if (!body) return 0; + + return body.textContent?.length || 0; +} + +function generatePageBreaks(totalCharacters: number): number[] { + const breaks: number[] = []; + const charsPerPage = 1000; + + for (let i = charsPerPage; i < totalCharacters; i += charsPerPage) { + breaks.push(i); + } + + return breaks; +} + +// ============================================================ +// Metadata Quick Extract +// ============================================================ + +export async function extractFB2Metadata( + fb2Blob: Blob, +): Promise> { + let xmlContent: string; + + if (fb2Blob.type === "application/zip") { + const zip = await JSZip.loadAsync(fb2Blob); + const files = Object.keys(zip.files); + const fb2File = files.find((f) => f.endsWith(".fb2")); + + if (!fb2File) return {}; + + xmlContent = await zip.file(fb2File)!.async("text"); + } else { + xmlContent = await fb2Blob.text(); + } + + const xmlDoc = parseXML(xmlContent); + return extractFB2Metadata(xmlDoc); +} diff --git a/web/src/reader/parsers/html-parser.ts b/web/src/reader/parsers/html-parser.ts new file mode 100644 index 0000000..04f7bd5 --- /dev/null +++ b/web/src/reader/parsers/html-parser.ts @@ -0,0 +1,158 @@ +// HTML Parser - Wraps standalone HTML files +// Procedural style: Functions, not classes + +// ============================================================ +// Main Parse Function +// ============================================================ + +export async function parseHTML(htmlBlob: Blob): Promise { + const htmlContent = await htmlBlob.text(); + + const metadata = extractHTMLMetadata(htmlBlob, htmlContent); + const toc = createHTMLTOC(htmlContent); + const spine = createHTMLSpine(htmlContent); + const resources = await extractHTMLResources(htmlBlob, htmlContent); + + const totalCharacters = stripHTML(htmlContent).length; + const pageBreaks = generatePageBreaks(totalCharacters); + + return { + metadata, + toc, + spine, + resources, + locations: { + totalCharacters, + pageBreaks, + }, + }; +} + +// ============================================================ +// Helper Functions +// ============================================================ + +function extractHTMLMetadata( + htmlBlob: Blob, + htmlContent: string, +): EbookCIF["metadata"] { + const parser = new DOMParser(); + const doc = parser.parseFromString(htmlContent, "text/html"); + + const title = + doc.querySelector("title")?.textContent || + htmlBlob.name.replace(/\.(html?|htm)$/i, ""); + + const metaAuthor = doc + .querySelector('meta[name="author"]') + ?.getAttribute("content"); + const metaLang = doc.querySelector("html")?.getAttribute("lang") || "en"; + + return { + title, + author: metaAuthor || "Unknown", + language: metaLang, + }; +} + +function createHTMLTOC(htmlContent: string): EbookCIF["toc"] { + const parser = new DOMParser(); + const doc = parser.parseFromString(htmlContent, "text/html"); + + const toc: EbookCIF["toc"] = []; + + // Try to find headings + const headings = doc.querySelectorAll("h1, h2, h3"); + let headingIndex = 0; + + headings.forEach((heading) => { + toc.push({ + id: `heading-${headingIndex}`, + title: heading.textContent || "", + href: `#${heading.id || `heading-${headingIndex}`}`, + children: [], + }); + + headingIndex++; + }); + + // If no headings, create single entry + if (toc.length === 0) { + toc.push({ + id: "full-document", + title: "Full Document", + href: "#full-document", + children: [], + }); + } + + return toc; +} + +function createHTMLSpine(htmlContent: string): EbookCIF["spine"] { + return [ + { + id: "full-document", + type: "html", + content: htmlContent, + index: 0, + }, + ]; +} + +async function extractHTMLResources( + htmlBlob: Blob, + htmlContent: string, +): Promise> { + const resources = new Map(); + const parser = new DOMParser(); + const doc = parser.parseFromString(htmlContent, "text/html"); + + // Extract images + const images = doc.querySelectorAll("img[src]"); + + for (const img of Array.from(images)) { + const src = img.getAttribute("src"); + if (!src) continue; + + // Try to resolve relative URLs + if (src.startsWith("data:")) { + // Data URI - extract blob + const match = src.match(/^data:([^;]+);base64,(.+)$/); + if (match) { + const mimeType = match[1]; + const base64 = match[2]; + const byteString = atob(base64); + const byteArray = new Uint8Array(byteString.length); + + for (let i = 0; i < byteString.length; i++) { + byteArray[i] = byteString.charCodeAt(i); + } + + const blob = new Blob([byteArray], { type: mimeType }); + resources.set(src, blob); + } + } + // External resources would need to be fetched + // For now, skip them (browser will load them naturally) + } + + return resources; +} + +function stripHTML(html: string): string { + const div = document.createElement("div"); + div.innerHTML = html; + return div.textContent || ""; +} + +// ============================================================ +// Metadata Quick Extract +// ============================================================ + +export async function extractHTMLMetadata( + htmlBlob: Blob, +): Promise> { + const htmlContent = await htmlBlob.text(); + return extractHTMLMetadata(htmlBlob, htmlContent); +} diff --git a/web/src/reader/parsers/txt-parser.ts b/web/src/reader/parsers/txt-parser.ts new file mode 100644 index 0000000..5e6d941 --- /dev/null +++ b/web/src/reader/parsers/txt-parser.ts @@ -0,0 +1,121 @@ +// TXT Parser - Wraps plain text in HTML structure +// Procedural style: Functions, not classes + +// ============================================================ +// Main Parse Function +// ============================================================ + +export async function parseTXT(txtBlob: Blob): Promise { + const textContent = await txtBlob.text(); + + const metadata = extractTXTMetadata(txtBlob); + const toc = createTXTTOC(textContent); + const spine = createTXTSpine(textContent); + const resources = new Map(); // No external resources for plain text + + const totalCharacters = textContent.length; + + return { + metadata, + toc, + spine, + resources, + locations: { + totalCharacters, + estimatedPages: Math.ceil(totalCharacters / 1500), + }, + }; +} + +// ============================================================ +// Helper Functions +// ============================================================ + +function extractTXTMetadata(txtBlob: Blob): EbookCIF["metadata"] { + const filename = txtBlob.name || "Unknown"; + + return { + title: filename.replace(/\.(txt|text)$/i, ""), + author: "Unknown", + language: "en", + }; +} + +function createTXTTOC(textContent: string): EbookCIF["toc"] { + // Try to detect chapters (simple heuristic) + const toc: EbookCIF["toc"] = []; + const lines = textContent.split("\n"); + + let chapterIndex = 0; + + lines.forEach((line, index) => { + // Common chapter patterns + const chapterPattern = /^(chapter|part|section)\s+\d+/i; + if (chapterPattern.test(line.trim())) { + toc.push({ + id: `chapter-${chapterIndex}`, + title: line.trim(), + href: `#chapter-${chapterIndex}`, + children: [], + }); + + chapterIndex++; + } + }); + + // If no chapters found, create single entry + if (toc.length === 0) { + toc.push({ + id: "full-text", + title: "Full Text", + href: "#full-text", + children: [], + }); + } + + return toc; +} + +function createTXTSpine(textContent: string): EbookCIF["spine"] { + // Convert plain text to HTML paragraphs + const lines = textContent.split("\n"); + let htmlContent = '
'; + + lines.forEach((line) => { + const trimmed = line.trim(); + if (trimmed) { + htmlContent += `

${escapeHTML(trimmed)}

`; + } else { + htmlContent += "
"; + } + }); + + htmlContent += "
"; + + return [ + { + id: "full-text", + type: "html", + content: htmlContent, + index: 0, + }, + ]; +} + +function escapeHTML(text: string): string { + const div = document.createElement("div"); + div.textContent = text; + return div.innerHTML; +} + +// Removed - backend handles detailed position tracking + +// ============================================================ +// Metadata Quick Extract +// ============================================================ + +export async function extractTXTMetadata( + txtBlob: Blob, +): Promise> { + return extractTXTMetadata(txtBlob); +}