From 72839921f2d86652eb2e903bcc08b983a2cb021a Mon Sep 17 00:00:00 2001 From: John O'Keefe Date: Sat, 3 Oct 2026 16:11:04 -0400 Subject: [PATCH] drop DOC and LIT from the pipeline (deliberately unsupported) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit .IMAGE — extensions removed from AllowedExtensions (scan gate), bookExtensions, MimeTypes and sync/format.go maps: .doc/.lit files are no longer scanned or indexed. classifyFormatGroup gains the RTF and PDB reflowable arms that were missing when those rungs shipped (rows only reclassify on creation). Scanner docs updated. DOC + LIT have no credible JS tooling (mammoth is docx-only; LIT needs LZX and its DRM variants are dead) and text-extraction-only support would misrepresent what the reader can do. Verified: probe .doc/.lit files watched but never scanned; probe .txt control scanned; TestClassifyFormatGroup extended and passing. Docs change granted in-session by the user (extensions + format docs). --- docs/developer/api/scanner/overview.md | 9 +++++++-- docs/developer/api/scanner/scan_library.md | 2 +- internal/services/library_service.go | 4 +--- internal/services/media_scanner.go | 6 +++--- internal/services/media_scanner_test.go | 2 ++ internal/sync/format.go | 4 ---- 6 files changed, 14 insertions(+), 13 deletions(-) diff --git a/docs/developer/api/scanner/overview.md b/docs/developer/api/scanner/overview.md index 7c0780a..3595715 100644 --- a/docs/developer/api/scanner/overview.md +++ b/docs/developer/api/scanner/overview.md @@ -32,8 +32,13 @@ The Bookhoard scanner provides comprehensive library management for ebooks, comi | PDF | `.pdf` | | Kindle | `.mobi` | | Text | `.txt`, `.rtf` | -| Document | `.doc`, `.docx` | -| Other | `.lit`, `.fb2`, `.pdb` | +| Document | `.docx` | +| Other | `.fb2`, `.pdb` | + +DOC (Word 97-2003 binary) and LIT (Microsoft Reader) are deliberately +unsupported: no credible JavaScript parser exists for either, and +compromising on fidelity is against the project's format goals. Files +with these extensions are not scanned. ### Comics diff --git a/docs/developer/api/scanner/scan_library.md b/docs/developer/api/scanner/scan_library.md index 777eb18..c69eed4 100644 --- a/docs/developer/api/scanner/scan_library.md +++ b/docs/developer/api/scanner/scan_library.md @@ -18,7 +18,7 @@ Initiate a one-time scan of a library for ebooks, manga, or comics. The scanner automatically detects and processes files based on the library type: -**Ebooks:** .epub, .pdf, .mobi, .txt, .rtf, .doc, .docx, .lit, .fb2, .pdb +**Ebooks:** .epub, .pdf, .mobi, .txt, .rtf, .docx, .fb2, .pdb **Comics:** .cbz, .cbr, .cb7, .cbt, .pdf diff --git a/internal/services/library_service.go b/internal/services/library_service.go index 1619606..9746da6 100644 --- a/internal/services/library_service.go +++ b/internal/services/library_service.go @@ -30,7 +30,7 @@ const ( ) var AllowedExtensions = map[string][]string{ - LibraryTypeEbooks: {".epub", ".pdf", ".mobi", ".azw", ".azw3", ".txt", ".rtf", ".doc", ".docx", ".lit", ".fb2", ".pdb"}, + LibraryTypeEbooks: {".epub", ".pdf", ".mobi", ".azw", ".azw3", ".txt", ".rtf", ".docx", ".fb2", ".pdb"}, LibraryTypeComics: {".cbz", ".cbr", ".cb7", ".cbt", ".epub", ".pdf"}, LibraryTypeManga: {".cbz", ".cbr", ".epub", ".pdf", ".png", ".jpg", ".jpeg", ".gif", ".bmp", ".webp", ".avif", ".tiff", ".tif"}, } @@ -59,9 +59,7 @@ var MimeTypes = map[string]string{ ".azw3": "application/vnd.amazon.ebook", ".txt": "text/plain", ".rtf": "application/rtf", - ".doc": "application/msword", ".docx": "application/vnd.openxmlformats-officedocument.wordprocessingml.document", - ".lit": "application/x-msreader", ".fb2": "application/x-fictionbook+xml", ".pdb": "application/vnd.palm", } diff --git a/internal/services/media_scanner.go b/internal/services/media_scanner.go index 8a20567..a962226 100644 --- a/internal/services/media_scanner.go +++ b/internal/services/media_scanner.go @@ -662,8 +662,8 @@ func (s *MediaScanner) extractFolderStructureMetadata(path, rootFolder string) * var bookExtensions = map[string]bool{ ".epub": true, ".pdf": true, ".mobi": true, ".azw": true, ".azw3": true, - ".fb2": true, ".txt": true, ".rtf": true, ".doc": true, ".docx": true, - ".lit": true, ".pdb": true, ".djvu": true, + ".fb2": true, ".txt": true, ".rtf": true, ".docx": true, + ".pdb": true, ".djvu": true, ".cbz": true, ".cbr": true, ".cb7": true, ".cbt": true, } @@ -3873,7 +3873,7 @@ func classifyFormatGroup(ext string, epubIsFixedLayout bool) (formatGroup string return "fixed_layout", false, true } return "reflowable", true, false - case ".mobi", ".azw", ".azw3", ".fb2", ".txt", ".docx": + case ".mobi", ".azw", ".azw3", ".fb2", ".txt", ".docx", ".rtf", ".pdb": return "reflowable", true, false case ".pdf", ".djvu": return "fixed_layout", false, true diff --git a/internal/services/media_scanner_test.go b/internal/services/media_scanner_test.go index 1ea9e47..c97732e 100644 --- a/internal/services/media_scanner_test.go +++ b/internal/services/media_scanner_test.go @@ -135,6 +135,8 @@ func TestClassifyFormatGroup(t *testing.T) { {"FB2", ".fb2", false, "reflowable", true, false}, {"TXT", ".txt", false, "reflowable", true, false}, {"DOCX (was unknown before the fix)", ".docx", false, "reflowable", true, false}, + {"RTF", ".rtf", false, "reflowable", true, false}, + {"PDB (PalmDOC)", ".pdb", false, "reflowable", true, false}, {"PDF", ".pdf", false, "fixed_layout", false, true}, {"DJVU", ".djvu", false, "fixed_layout", false, true}, {"CBZ", ".cbz", false, "comic_archive", false, true}, diff --git a/internal/sync/format.go b/internal/sync/format.go index 7ffd47e..ecd8a83 100644 --- a/internal/sync/format.go +++ b/internal/sync/format.go @@ -30,9 +30,7 @@ var mimeTypes = map[string]string{ ".fb2": "application/x-fictionbook+xml", ".txt": "text/plain", ".rtf": "application/rtf", - ".doc": "application/msword", ".docx": "application/vnd.openxmlformats-officedocument.wordprocessingml.document", - ".lit": "application/x-ms-reader", ".pdb": "application/vnd.palm", ".prc": "application/vnd.palm", } @@ -46,9 +44,7 @@ var ReflowableFormats = map[string]bool{ ".fb2": true, ".txt": true, ".rtf": true, - ".doc": true, ".docx": true, - ".lit": true, ".pdb": true, ".prc": true, }