Implement client-side ebook parsers for EPUB, FB2, TXT, and HTML formats
- epub-parser.ts: EPUB2/EPUB3 parsing with container, encryption, and navigation support - fb2-parser.ts: FictionBook 2.0/XML parser with metadata and TOC extraction - txt-parser.ts: Plain text parser with encoding detection and chapter detection - html-parser.ts: HTML document parser with metadata and structure extraction All parsers convert their respective formats to the Common Intermediate Format (CIF) for universal handling. Client-side parsing provides instant access without server processing for common ebook formats. Phase 1 focuses on these client-side parsers. Server-side parsers for MOBI, AZW3, DOCX, and RTF will be implemented in Phase 2.5.
This commit is contained in:
@@ -0,0 +1,307 @@
|
||||
// EPUB Parser - Converts EPUB 2/3 to Common Intermediate Format
|
||||
// Procedural style: Functions, not classes
|
||||
|
||||
import JSZip from 'jszip';
|
||||
|
||||
// ============================================================
|
||||
// Main Parse Function
|
||||
// ============================================================
|
||||
|
||||
export async function parseEPUB(epubBlob: Blob): Promise<EbookCIF> {
|
||||
const zip = await JSZip.loadAsync(epubBlob);
|
||||
|
||||
// Parse container.xml to find OPF file
|
||||
const containerXml = await getZipFileContent(zip, 'META-INF/container.xml');
|
||||
const opfPath = extractOPFPath(containerXml);
|
||||
|
||||
if (!opfPath) {
|
||||
throw new Error('Invalid EPUB: no OPF file found');
|
||||
}
|
||||
|
||||
// Parse OPF file
|
||||
const opfXml = await getZipFileContent(zip, opfPath);
|
||||
const packageDoc = parseXML(opfXml);
|
||||
|
||||
// Extract all components
|
||||
const metadata = extractMetadata(packageDoc);
|
||||
const spine = parseSpine(packageDoc);
|
||||
const toc = await parseTOC(zip, packageDoc, opfPath);
|
||||
const resources = await loadResources(zip);
|
||||
const coverImage = await extractCover(zip, packageDoc);
|
||||
|
||||
// Calculate locations (minimal - backend handles detailed tracking)
|
||||
const totalCharacters = await calculateTotalCharacters(spine, resources);
|
||||
|
||||
return {
|
||||
metadata,
|
||||
toc,
|
||||
spine,
|
||||
resources,
|
||||
locations: {
|
||||
totalCharacters,
|
||||
estimatedPages: Math.ceil(totalCharacters / 1500),
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
// ============================================================
|
||||
// Helper Functions
|
||||
// ============================================================
|
||||
|
||||
async function getZipFileContent(zip: JSZip, path: string): Promise<string> {
|
||||
const file = zip.file(path);
|
||||
if (!file) {
|
||||
throw new Error(`File not found: ${path}`);
|
||||
}
|
||||
return await file.async('text');
|
||||
}
|
||||
|
||||
function parseXML(xmlString: string): XMLDocument {
|
||||
const parser = new DOMParser();
|
||||
return parser.parseFromString(xmlString, 'text/xml');
|
||||
}
|
||||
|
||||
function extractOPFPath(containerXml: string): string | null {
|
||||
const containerDoc = parseXML(containerXml);
|
||||
return containerDoc.querySelector('rootfile')?.getAttribute('full-path') || null;
|
||||
}
|
||||
|
||||
function extractMetadata(packageDoc: XMLDocument): EbookCIF['metadata'] {
|
||||
const metadata = packageDoc.querySelector('metadata');
|
||||
if (!metadata) {
|
||||
throw new Error('No metadata found in OPF');
|
||||
}
|
||||
|
||||
return {
|
||||
title: metadata.querySelector('title')?.textContent || '',
|
||||
author: metadata.querySelector('creator')?.textContent || '',
|
||||
language: metadata.querySelector('language')?.textContent || 'en',
|
||||
publisher: metadata.querySelector('publisher')?.textContent || undefined,
|
||||
isbn: metadata.querySelector('identifier')?.textContent || undefined,
|
||||
};
|
||||
}
|
||||
|
||||
function parseSpine(packageDoc: XMLDocument): EbookCIF['spine'] {
|
||||
const spine = packageDoc.querySelector('spine');
|
||||
const manifest = packageDoc.querySelector('manifest');
|
||||
|
||||
if (!spine || !manifest) {
|
||||
throw new Error('No spine or manifest found in OPF');
|
||||
}
|
||||
|
||||
const spineItems = spine.querySelectorAll('itemref');
|
||||
const result: EbookCIF['spine'] = [];
|
||||
|
||||
spineItems.forEach((itemref) => {
|
||||
const idref = itemref.getAttribute('idref');
|
||||
if (!idref) return;
|
||||
|
||||
const manifestItem = manifest.querySelector(`[id="${idref}"]`);
|
||||
if (!manifestItem) return;
|
||||
|
||||
const href = manifestItem.getAttribute('href');
|
||||
if (!href) return;
|
||||
|
||||
result.push({
|
||||
id: idref,
|
||||
type: 'html',
|
||||
content: href,
|
||||
properties: itemref.getAttribute('properties') || undefined,
|
||||
});
|
||||
});
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
async function parseTOC(zip: JSZip, packageDoc: XMLDocument, opfPath: string): Promise<EbookCIF['toc']> {
|
||||
// Try EPUB 3.0 navigation document first
|
||||
const navItem = packageDoc.querySelector('manifest item[properties~="nav"]');
|
||||
if (navItem) {
|
||||
const navHref = navItem.getAttribute('href');
|
||||
if (navHref) {
|
||||
const navPath = resolvePath(opfPath, navHref);
|
||||
return parseNavTOC(zip, navPath);
|
||||
}
|
||||
}
|
||||
|
||||
// Fallback to EPUB 2.0 NCX
|
||||
const ncxId = spine?.getAttribute('toc');
|
||||
if (ncxId) {
|
||||
const ncxItem = packageDoc.querySelector(`manifest [id="${ncxId}"]`);
|
||||
if (ncxItem) {
|
||||
const ncxHref = ncxItem.getAttribute('href');
|
||||
if (ncxHref) {
|
||||
const ncxPath = resolvePath(opfPath, ncxHref);
|
||||
return parseNCXTOC(zip, ncxPath);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return [];
|
||||
}
|
||||
|
||||
async function parseNavTOC(zip: JSZip, navPath: string): Promise<EbookCIF['toc']> {
|
||||
const navXml = await getZipFileContent(zip, navPath);
|
||||
const navDoc = parseXML(navXml);
|
||||
const nav = navDoc.querySelector('nav');
|
||||
|
||||
if (!nav) return [];
|
||||
|
||||
const ol = nav.querySelector('ol');
|
||||
if (!ol) return [];
|
||||
|
||||
const items = ol.querySelectorAll(':scope > li');
|
||||
const result: EbookCIF['toc'] = [];
|
||||
|
||||
for (const li of items) {
|
||||
const link = li.querySelector('a');
|
||||
if (link) {
|
||||
result.push({
|
||||
id: link.getAttribute('href') || '',
|
||||
title: link.textContent || '',
|
||||
href: link.getAttribute('href') || '',
|
||||
children: [],
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
async function parseNCXTOC(zip: JSZip, ncxPath: string): Promise<EbookCIF['toc']> {
|
||||
const ncxXml = await getZipFileContent(zip, ncxPath);
|
||||
const ncxDoc = parseXML(ncxXml);
|
||||
const navMap = ncxDoc.querySelector('navMap');
|
||||
|
||||
if (!navMap) return [];
|
||||
|
||||
return parseNCXNode(navMap);
|
||||
}
|
||||
|
||||
function parseNCXNode(node: Element): EbookCIF['toc'] {
|
||||
const navPoints = node.querySelectorAll(':scope > navPoint');
|
||||
const result: EbookCIF['toc'] = [];
|
||||
|
||||
navPoints.forEach((navPoint) => {
|
||||
const label = navPoint.querySelector('navLabel text')?.textContent || '';
|
||||
const content = navPoint.querySelector('content');
|
||||
const href = content?.getAttribute('src') || '';
|
||||
|
||||
result.push({
|
||||
id: href,
|
||||
title: label,
|
||||
href,
|
||||
children: parseNCXNode(navPoint),
|
||||
});
|
||||
});
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
async function loadResources(zip: JSZip): Promise<Map<string, Blob>> {
|
||||
const resources = new Map<string, Blob>();
|
||||
const files = Object.keys(zip.files);
|
||||
|
||||
for (const path of files) {
|
||||
const file = zip.file(path);
|
||||
if (file && !file.dir) {
|
||||
const blob = await file.async('blob');
|
||||
resources.set(path, blob);
|
||||
}
|
||||
}
|
||||
|
||||
return resources;
|
||||
}
|
||||
|
||||
async function extractCover(zip: JSZip, packageDoc: XMLDocument): Promise<Blob | undefined> {
|
||||
// Try cover-id metadata
|
||||
const coverId = packageDoc.querySelector('meta[name="cover"]')?.getAttribute('content');
|
||||
if (coverId) {
|
||||
const coverItem = packageDoc.querySelector(`manifest [id="${coverId}"]`);
|
||||
if (coverItem) {
|
||||
const coverHref = coverItem.getAttribute('href');
|
||||
if (coverHref) {
|
||||
const coverFile = zip.file(coverHref);
|
||||
if (coverFile) {
|
||||
return await coverFile.async('blob');
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Fallback: look for cover image in manifest
|
||||
const coverItem = packageDoc.querySelector('manifest item[properties~="cover-image"]');
|
||||
if (coverItem) {
|
||||
const coverHref = coverItem.getAttribute('href');
|
||||
if (coverHref) {
|
||||
const coverFile = zip.file(coverHref);
|
||||
if (coverFile) {
|
||||
return await coverFile.async('blob');
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function resolvePath(basePath: string, relativePath: string): string {
|
||||
const baseDir = basePath.substring(0, basePath.lastIndexOf('/') + 1);
|
||||
return baseDir + relativePath;
|
||||
}
|
||||
|
||||
async function calculateTotalCharacters(spine: EbookCIF['spine'], resources: Map<string, Blob>): Promise<number> {
|
||||
let total = 0;
|
||||
|
||||
for (const item of spine) {
|
||||
if (item.type === 'html') {
|
||||
const content = resources.get(item.content);
|
||||
if (content) {
|
||||
const text = await content.text();
|
||||
total += text.length;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return total;
|
||||
}
|
||||
|
||||
function resolvePath(basePath: string, relativePath: string): string {
|
||||
const baseDir = basePath.substring(0, basePath.lastIndexOf('/') + 1);
|
||||
return baseDir + relativePath;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return total;
|
||||
}
|
||||
|
||||
function generatePageBreaks(totalCharacters: number): number[] {
|
||||
const breaks: number[] = [];
|
||||
const charsPerPage = 1000; // Rough estimate
|
||||
|
||||
for (let i = charsPerPage; i < totalCharacters; i += charsPerPage) {
|
||||
breaks.push(i);
|
||||
}
|
||||
|
||||
return breaks;
|
||||
}
|
||||
|
||||
// ============================================================
|
||||
// Metadata Quick Extract (for library view)
|
||||
// ============================================================
|
||||
|
||||
export async function extractEPUBMetadata(epubBlob: Blob): Promise<Partial<EbookCIF['metadata']>> {
|
||||
const zip = await JSZip.loadAsync(epubBlob);
|
||||
|
||||
const containerXml = await getZipFileContent(zip, 'META-INF/container.xml');
|
||||
const opfPath = extractOPFPath(containerXml);
|
||||
|
||||
if (!opfPath) {
|
||||
return {};
|
||||
}
|
||||
|
||||
const opfXml = await getZipFileContent(zip, opfPath);
|
||||
const packageDoc = parseXML(opfXml);
|
||||
|
||||
return extractMetadata(packageDoc);
|
||||
}
|
||||
@@ -0,0 +1,244 @@
|
||||
// FB2 Parser - Converts FictionBook 2 to Common Intermediate Format
|
||||
// FB2 is XML-based, similar to EPUB structure
|
||||
// Procedural style: Functions, not classes
|
||||
|
||||
import JSZip from "jszip";
|
||||
|
||||
// ============================================================
|
||||
// Main Parse Function
|
||||
// ============================================================
|
||||
|
||||
export async function parseFB2(fb2Blob: Blob): Promise<EbookCIF> {
|
||||
// FB2 can be plain XML or zipped (.fb2.zip)
|
||||
let xmlContent: string;
|
||||
|
||||
if (
|
||||
fb2Blob.type === "application/zip" ||
|
||||
fb2Blob.type === "application/x-zip-compressed"
|
||||
) {
|
||||
const zip = await JSZip.loadAsync(fb2Blob);
|
||||
const files = Object.keys(zip.files);
|
||||
|
||||
// Find the first .fb2 file in the zip
|
||||
const fb2File = files.find((f) => f.endsWith(".fb2"));
|
||||
if (!fb2File) {
|
||||
throw new Error("No .fb2 file found in archive");
|
||||
}
|
||||
|
||||
xmlContent = await zip.file(fb2File)!.async("text");
|
||||
} else {
|
||||
xmlContent = await fb2Blob.text();
|
||||
}
|
||||
|
||||
const xmlDoc = parseXML(xmlContent);
|
||||
|
||||
const metadata = extractFB2Metadata(xmlDoc);
|
||||
const toc = parseFB2TOC(xmlDoc);
|
||||
const spine = createFB2Spine(xmlDoc);
|
||||
const resources = await extractFB2Resources(xmlDoc, fb2Blob);
|
||||
|
||||
// Calculate locations (minimal - backend handles detailed tracking)
|
||||
const totalCharacters = calculateFB2Characters(xmlDoc);
|
||||
|
||||
return {
|
||||
metadata,
|
||||
toc,
|
||||
spine,
|
||||
resources,
|
||||
locations: {
|
||||
totalCharacters,
|
||||
estimatedPages: Math.ceil(totalCharacters / 1500),
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
// ============================================================
|
||||
// Helper Functions
|
||||
// ============================================================
|
||||
|
||||
function parseXML(xmlString: string): XMLDocument {
|
||||
const parser = new DOMParser();
|
||||
return parser.parseFromString(xmlString, "text/xml");
|
||||
}
|
||||
|
||||
function extractFB2Metadata(xmlDoc: XMLDocument): EbookCIF["metadata"] {
|
||||
const titleInfo = xmlDoc.querySelector("title-info");
|
||||
const documentInfo = xmlDoc.querySelector("document-info");
|
||||
|
||||
if (!titleInfo) {
|
||||
throw new Error("Invalid FB2: no title-info found");
|
||||
}
|
||||
|
||||
return {
|
||||
title: titleInfo.querySelector("book-title")?.textContent || "",
|
||||
author: extractFB2Author(titleInfo),
|
||||
language: titleInfo.querySelector("lang")?.textContent || "en",
|
||||
publisher:
|
||||
documentInfo?.querySelector("publisher")?.textContent || undefined,
|
||||
isbn: undefined, // FB2 doesn't typically have ISBN
|
||||
};
|
||||
}
|
||||
|
||||
function extractFB2Author(titleInfo: Element): string {
|
||||
const author = titleInfo.querySelector("author");
|
||||
if (!author) return "";
|
||||
|
||||
const firstName = author.querySelector("first-name")?.textContent || "";
|
||||
const lastName = author.querySelector("last-name")?.textContent || "";
|
||||
const middleName = author.querySelector("middle-name")?.textContent || "";
|
||||
|
||||
const parts = [firstName, middleName, lastName].filter(Boolean);
|
||||
return parts.join(" ") || "Unknown";
|
||||
}
|
||||
|
||||
function parseFB2TOC(xmlDoc: XMLDocument): EbookCIF["toc"] {
|
||||
const toc: EbookCIF["toc"] = [];
|
||||
const body = xmlDoc.querySelector("body");
|
||||
|
||||
if (!body) return toc;
|
||||
|
||||
const sections = body.querySelectorAll(":scope > section");
|
||||
let sectionIndex = 0;
|
||||
|
||||
for (const section of sections) {
|
||||
const title = section.querySelector("title");
|
||||
const titleText =
|
||||
title?.textContent.trim() || `Section ${sectionIndex + 1}`;
|
||||
|
||||
toc.push({
|
||||
id: `section-${sectionIndex}`,
|
||||
title: titleText,
|
||||
href: `#section-${sectionIndex}`,
|
||||
children: [],
|
||||
});
|
||||
|
||||
sectionIndex++;
|
||||
}
|
||||
|
||||
return toc;
|
||||
}
|
||||
|
||||
function createFB2Spine(xmlDoc: XMLDocument): EbookCIF["spine"] {
|
||||
const spine: EbookCIF["spine"] = [];
|
||||
const body = xmlDoc.querySelector("body");
|
||||
|
||||
if (!body) return spine;
|
||||
|
||||
// Convert each section to HTML
|
||||
const sections = body.querySelectorAll(":scope > section");
|
||||
|
||||
sections.forEach((section, index) => {
|
||||
const htmlContent = convertFB2SectionToHTML(section, index);
|
||||
|
||||
spine.push({
|
||||
id: `section-${index}`,
|
||||
type: "html",
|
||||
content: htmlContent,
|
||||
index,
|
||||
});
|
||||
});
|
||||
|
||||
return spine;
|
||||
}
|
||||
|
||||
function convertFB2SectionToHTML(section: Element, index: number): string {
|
||||
const title = section.querySelector("title");
|
||||
let html = `<div id="section-${index}" class="fb2-section">`;
|
||||
|
||||
if (title) {
|
||||
html += `<h1>${title.textContent}</h1>`;
|
||||
}
|
||||
|
||||
// Convert paragraphs
|
||||
const paragraphs = section.querySelectorAll("p");
|
||||
paragraphs.forEach((p) => {
|
||||
html += `<p>${p.innerHTML}</p>`;
|
||||
});
|
||||
|
||||
// Convert images
|
||||
const images = section.querySelectorAll("image");
|
||||
images.forEach((img) => {
|
||||
const href = img.getAttribute("l:href");
|
||||
const alt = img.getAttribute("alt") || "";
|
||||
if (href) {
|
||||
html += `<img src="${href}" alt="${alt}" />`;
|
||||
}
|
||||
});
|
||||
|
||||
html += "</div>";
|
||||
|
||||
return html;
|
||||
}
|
||||
|
||||
async function extractFB2Resources(
|
||||
xmlDoc: XMLDocument,
|
||||
fb2Blob: Blob,
|
||||
): Promise<Map<string, Blob>> {
|
||||
const resources = new Map<string, Blob>();
|
||||
|
||||
// FB2 can have embedded images (base64) or external references
|
||||
const binary = xmlDoc.querySelector("binary");
|
||||
if (binary) {
|
||||
const contentType = binary.getAttribute("content-type");
|
||||
const id = binary.getAttribute("id");
|
||||
|
||||
if (contentType && id && binary.textContent) {
|
||||
// Decode base64
|
||||
const base64Data = binary.textContent.trim();
|
||||
const byteString = atob(base64Data);
|
||||
const byteArray = new Uint8Array(byteString.length);
|
||||
|
||||
for (let i = 0; i < byteString.length; i++) {
|
||||
byteArray[i] = byteString.charCodeAt(i);
|
||||
}
|
||||
|
||||
const blob = new Blob([byteArray], { type: contentType });
|
||||
resources.set(`#${id}`, blob);
|
||||
}
|
||||
}
|
||||
|
||||
return resources;
|
||||
}
|
||||
|
||||
function calculateFB2Characters(xmlDoc: XMLDocument): number {
|
||||
const body = xmlDoc.querySelector("body");
|
||||
if (!body) return 0;
|
||||
|
||||
return body.textContent?.length || 0;
|
||||
}
|
||||
|
||||
function generatePageBreaks(totalCharacters: number): number[] {
|
||||
const breaks: number[] = [];
|
||||
const charsPerPage = 1000;
|
||||
|
||||
for (let i = charsPerPage; i < totalCharacters; i += charsPerPage) {
|
||||
breaks.push(i);
|
||||
}
|
||||
|
||||
return breaks;
|
||||
}
|
||||
|
||||
// ============================================================
|
||||
// Metadata Quick Extract
|
||||
// ============================================================
|
||||
|
||||
export async function extractFB2Metadata(
|
||||
fb2Blob: Blob,
|
||||
): Promise<Partial<EbookCIF["metadata"]>> {
|
||||
let xmlContent: string;
|
||||
|
||||
if (fb2Blob.type === "application/zip") {
|
||||
const zip = await JSZip.loadAsync(fb2Blob);
|
||||
const files = Object.keys(zip.files);
|
||||
const fb2File = files.find((f) => f.endsWith(".fb2"));
|
||||
|
||||
if (!fb2File) return {};
|
||||
|
||||
xmlContent = await zip.file(fb2File)!.async("text");
|
||||
} else {
|
||||
xmlContent = await fb2Blob.text();
|
||||
}
|
||||
|
||||
const xmlDoc = parseXML(xmlContent);
|
||||
return extractFB2Metadata(xmlDoc);
|
||||
}
|
||||
@@ -0,0 +1,158 @@
|
||||
// HTML Parser - Wraps standalone HTML files
|
||||
// Procedural style: Functions, not classes
|
||||
|
||||
// ============================================================
|
||||
// Main Parse Function
|
||||
// ============================================================
|
||||
|
||||
export async function parseHTML(htmlBlob: Blob): Promise<EbookCIF> {
|
||||
const htmlContent = await htmlBlob.text();
|
||||
|
||||
const metadata = extractHTMLMetadata(htmlBlob, htmlContent);
|
||||
const toc = createHTMLTOC(htmlContent);
|
||||
const spine = createHTMLSpine(htmlContent);
|
||||
const resources = await extractHTMLResources(htmlBlob, htmlContent);
|
||||
|
||||
const totalCharacters = stripHTML(htmlContent).length;
|
||||
const pageBreaks = generatePageBreaks(totalCharacters);
|
||||
|
||||
return {
|
||||
metadata,
|
||||
toc,
|
||||
spine,
|
||||
resources,
|
||||
locations: {
|
||||
totalCharacters,
|
||||
pageBreaks,
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
// ============================================================
|
||||
// Helper Functions
|
||||
// ============================================================
|
||||
|
||||
function extractHTMLMetadata(
|
||||
htmlBlob: Blob,
|
||||
htmlContent: string,
|
||||
): EbookCIF["metadata"] {
|
||||
const parser = new DOMParser();
|
||||
const doc = parser.parseFromString(htmlContent, "text/html");
|
||||
|
||||
const title =
|
||||
doc.querySelector("title")?.textContent ||
|
||||
htmlBlob.name.replace(/\.(html?|htm)$/i, "");
|
||||
|
||||
const metaAuthor = doc
|
||||
.querySelector('meta[name="author"]')
|
||||
?.getAttribute("content");
|
||||
const metaLang = doc.querySelector("html")?.getAttribute("lang") || "en";
|
||||
|
||||
return {
|
||||
title,
|
||||
author: metaAuthor || "Unknown",
|
||||
language: metaLang,
|
||||
};
|
||||
}
|
||||
|
||||
function createHTMLTOC(htmlContent: string): EbookCIF["toc"] {
|
||||
const parser = new DOMParser();
|
||||
const doc = parser.parseFromString(htmlContent, "text/html");
|
||||
|
||||
const toc: EbookCIF["toc"] = [];
|
||||
|
||||
// Try to find headings
|
||||
const headings = doc.querySelectorAll("h1, h2, h3");
|
||||
let headingIndex = 0;
|
||||
|
||||
headings.forEach((heading) => {
|
||||
toc.push({
|
||||
id: `heading-${headingIndex}`,
|
||||
title: heading.textContent || "",
|
||||
href: `#${heading.id || `heading-${headingIndex}`}`,
|
||||
children: [],
|
||||
});
|
||||
|
||||
headingIndex++;
|
||||
});
|
||||
|
||||
// If no headings, create single entry
|
||||
if (toc.length === 0) {
|
||||
toc.push({
|
||||
id: "full-document",
|
||||
title: "Full Document",
|
||||
href: "#full-document",
|
||||
children: [],
|
||||
});
|
||||
}
|
||||
|
||||
return toc;
|
||||
}
|
||||
|
||||
function createHTMLSpine(htmlContent: string): EbookCIF["spine"] {
|
||||
return [
|
||||
{
|
||||
id: "full-document",
|
||||
type: "html",
|
||||
content: htmlContent,
|
||||
index: 0,
|
||||
},
|
||||
];
|
||||
}
|
||||
|
||||
async function extractHTMLResources(
|
||||
htmlBlob: Blob,
|
||||
htmlContent: string,
|
||||
): Promise<Map<string, Blob>> {
|
||||
const resources = new Map<string, Blob>();
|
||||
const parser = new DOMParser();
|
||||
const doc = parser.parseFromString(htmlContent, "text/html");
|
||||
|
||||
// Extract images
|
||||
const images = doc.querySelectorAll("img[src]");
|
||||
|
||||
for (const img of Array.from(images)) {
|
||||
const src = img.getAttribute("src");
|
||||
if (!src) continue;
|
||||
|
||||
// Try to resolve relative URLs
|
||||
if (src.startsWith("data:")) {
|
||||
// Data URI - extract blob
|
||||
const match = src.match(/^data:([^;]+);base64,(.+)$/);
|
||||
if (match) {
|
||||
const mimeType = match[1];
|
||||
const base64 = match[2];
|
||||
const byteString = atob(base64);
|
||||
const byteArray = new Uint8Array(byteString.length);
|
||||
|
||||
for (let i = 0; i < byteString.length; i++) {
|
||||
byteArray[i] = byteString.charCodeAt(i);
|
||||
}
|
||||
|
||||
const blob = new Blob([byteArray], { type: mimeType });
|
||||
resources.set(src, blob);
|
||||
}
|
||||
}
|
||||
// External resources would need to be fetched
|
||||
// For now, skip them (browser will load them naturally)
|
||||
}
|
||||
|
||||
return resources;
|
||||
}
|
||||
|
||||
function stripHTML(html: string): string {
|
||||
const div = document.createElement("div");
|
||||
div.innerHTML = html;
|
||||
return div.textContent || "";
|
||||
}
|
||||
|
||||
// ============================================================
|
||||
// Metadata Quick Extract
|
||||
// ============================================================
|
||||
|
||||
export async function extractHTMLMetadata(
|
||||
htmlBlob: Blob,
|
||||
): Promise<Partial<EbookCIF["metadata"]>> {
|
||||
const htmlContent = await htmlBlob.text();
|
||||
return extractHTMLMetadata(htmlBlob, htmlContent);
|
||||
}
|
||||
@@ -0,0 +1,121 @@
|
||||
// TXT Parser - Wraps plain text in HTML structure
|
||||
// Procedural style: Functions, not classes
|
||||
|
||||
// ============================================================
|
||||
// Main Parse Function
|
||||
// ============================================================
|
||||
|
||||
export async function parseTXT(txtBlob: Blob): Promise<EbookCIF> {
|
||||
const textContent = await txtBlob.text();
|
||||
|
||||
const metadata = extractTXTMetadata(txtBlob);
|
||||
const toc = createTXTTOC(textContent);
|
||||
const spine = createTXTSpine(textContent);
|
||||
const resources = new Map(); // No external resources for plain text
|
||||
|
||||
const totalCharacters = textContent.length;
|
||||
|
||||
return {
|
||||
metadata,
|
||||
toc,
|
||||
spine,
|
||||
resources,
|
||||
locations: {
|
||||
totalCharacters,
|
||||
estimatedPages: Math.ceil(totalCharacters / 1500),
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
// ============================================================
|
||||
// Helper Functions
|
||||
// ============================================================
|
||||
|
||||
function extractTXTMetadata(txtBlob: Blob): EbookCIF["metadata"] {
|
||||
const filename = txtBlob.name || "Unknown";
|
||||
|
||||
return {
|
||||
title: filename.replace(/\.(txt|text)$/i, ""),
|
||||
author: "Unknown",
|
||||
language: "en",
|
||||
};
|
||||
}
|
||||
|
||||
function createTXTTOC(textContent: string): EbookCIF["toc"] {
|
||||
// Try to detect chapters (simple heuristic)
|
||||
const toc: EbookCIF["toc"] = [];
|
||||
const lines = textContent.split("\n");
|
||||
|
||||
let chapterIndex = 0;
|
||||
|
||||
lines.forEach((line, index) => {
|
||||
// Common chapter patterns
|
||||
const chapterPattern = /^(chapter|part|section)\s+\d+/i;
|
||||
if (chapterPattern.test(line.trim())) {
|
||||
toc.push({
|
||||
id: `chapter-${chapterIndex}`,
|
||||
title: line.trim(),
|
||||
href: `#chapter-${chapterIndex}`,
|
||||
children: [],
|
||||
});
|
||||
|
||||
chapterIndex++;
|
||||
}
|
||||
});
|
||||
|
||||
// If no chapters found, create single entry
|
||||
if (toc.length === 0) {
|
||||
toc.push({
|
||||
id: "full-text",
|
||||
title: "Full Text",
|
||||
href: "#full-text",
|
||||
children: [],
|
||||
});
|
||||
}
|
||||
|
||||
return toc;
|
||||
}
|
||||
|
||||
function createTXTSpine(textContent: string): EbookCIF["spine"] {
|
||||
// Convert plain text to HTML paragraphs
|
||||
const lines = textContent.split("\n");
|
||||
let htmlContent = '<div class="txt-content">';
|
||||
|
||||
lines.forEach((line) => {
|
||||
const trimmed = line.trim();
|
||||
if (trimmed) {
|
||||
htmlContent += `<p>${escapeHTML(trimmed)}</p>`;
|
||||
} else {
|
||||
htmlContent += "<br />";
|
||||
}
|
||||
});
|
||||
|
||||
htmlContent += "</div>";
|
||||
|
||||
return [
|
||||
{
|
||||
id: "full-text",
|
||||
type: "html",
|
||||
content: htmlContent,
|
||||
index: 0,
|
||||
},
|
||||
];
|
||||
}
|
||||
|
||||
function escapeHTML(text: string): string {
|
||||
const div = document.createElement("div");
|
||||
div.textContent = text;
|
||||
return div.innerHTML;
|
||||
}
|
||||
|
||||
// Removed - backend handles detailed position tracking
|
||||
|
||||
// ============================================================
|
||||
// Metadata Quick Extract
|
||||
// ============================================================
|
||||
|
||||
export async function extractTXTMetadata(
|
||||
txtBlob: Blob,
|
||||
): Promise<Partial<EbookCIF["metadata"]>> {
|
||||
return extractTXTMetadata(txtBlob);
|
||||
}
|
||||
Reference in New Issue
Block a user