feat(web): notes render as markdown — markdown-it+DOMPurify pipeline, rich paste

Objective 6 (web half). Storage/sync stay raw markdown (KOReader shows
literal source, accepted). Notes-grade syntax: headings, emphasis/
strikethrough, code, links, lists, blockquotes, GFM tables, reference
links, footnotes (markdown-it-footnote); images/raw HTML/math excluded
(html:false posture + FORBID img/style/form + default safe-scheme URI
check). Display sites: annotations-drawer note rows and highlight note
lines render pre-sanitized HTML (x-html); render-time sanitize only.

Rich paste: both note textareas intercept paste; a text/html clipboard
flavor converts via turndown (gfm tables incl. a headingless-table rule
emitting pipe syntax; images blanked) and inserts at the caret; plain
pastics fall through to the default paste unchanged.

Verified live in Brave: bold/italic/strike/code/lists/quote/table/
footnote-ref/reference-link render; <script> renders inert literal
text with no alert; javascript: hrefs absent from the DOM; paste of
rich HTML lands markdown at the caret; plain pastes untouched.
This commit is contained in:
John O'Keefe
2026-10-04 13:25:17 -04:00
parent b9f797789f
commit 4b5e193d9f
5 changed files with 143 additions and 9 deletions
+102
View File
@@ -0,0 +1,102 @@
// Notes render as markdown: storage and every sync payload carry the raw
// source string (KOReader devices show it literally — accepted), and every
// DISPLAY site renders it through this pipeline. Raw HTML is disabled at the
// parser level (markdown-it's default) and DOMPurify strips whatever the
// enabled syntax still produced (link hrefs are scheme-whitelisted below),
// so a malicious note can render nothing more than styled text.
import MarkdownIt from "markdown-it";
import footnote from "markdown-it-footnote";
import DOMPurify from "dompurify";
import TurndownService from "turndown";
import { gfm, tables } from "turndown-plugin-gfm";
const md = new MarkdownIt({ html: false, linkify: false });
md.use(footnote);
export function renderNoteHtml(source: string): string {
if (!source) return "";
const raw = md.render(source);
// Disallowing img/input/style/form by tag and keeping DOMPurify's default
// safe-URI scheme check (http/https/mailto + same-page anchors only)
// closes the link-injection hole, which is the one markdown pivot into
// script execution.
const clean = DOMPurify.sanitize(raw, {
ALLOWED_ATTR: ["href", "class", "id", "role"],
FORBID_TAGS: ["img", "input", "style", "form"],
});
return clean;
}
// Compact rows may show a plain line instead of rich markup.
export function stripNote(source: string): string {
if (!source) return "";
const holder = document.createElement("div");
holder.innerHTML = renderNoteHtml(source);
return (holder.textContent || "").replace(/\s+/g, " ").trim();
}
// ----- rich paste: HTML flavor -> markdown source -----
const turndown = new TurndownService({
// Keep the converted output inside our supported syntax: headings,
// emphasis, lists, code, quotes, links, tables. Everything else degrades.
headingStyle: "atx",
codeBlockStyle: "fenced",
bulletListMarker: "-",
hr: "---",
});
turndown.use(gfm);
turndown.use(tables);
// Images are explicitly out of scope for notes: blank them at convert time
// (remove() alone can leave the node's alt-text shape behind).
turndown.addRule("noImages", { filter: "img", replacement: () => "" });
// The gfm tables service only converts tables with a heading row and keeps
// headingless ones as raw HTML — which our no-raw-HTML parser would then
// show as literal tags. Convert those to pipe syntax with a blank header.
turndown.addRule("tablesNoHeading", {
filter: (node: Node) => {
const el = node as HTMLElement;
if (el.nodeName !== "TABLE" || !el.querySelector) return false;
return !el.querySelector("th") && el.querySelectorAll("tr").length > 0;
},
replacement: (_content: string, node: Node) => {
const table = node as HTMLTableElement;
const rows = Array.from(table.querySelectorAll("tr")).map((tr) =>
Array.from(tr.querySelectorAll("td, th")).map((cell) =>
(cell.textContent ?? "").replace(/\s+/g, " ").trim().replace(/\|/g, "\\|"),
),
);
const cols = Math.max(1, ...rows.map((r) => r.length));
const pad = (r: string[]) => {
const copy = [...r];
while (copy.length < cols) copy.push("");
return copy.map((c) => c || " ");
};
const head = pad(rows[0] ?? []);
const lines = [
"| " + head.join(" | ") + " |",
"|" + Array.from({ length: cols }, () => " --- ").join("|") + "|",
...rows.slice(1).map((r) => "| " + pad(r).join(" | ") + " |"),
];
return "\n\n" + lines.join("\n") + "\n\n";
},
});
export function htmlToMarkdown(html: string): string {
return turndown.turndown(html);
}
/** Reads the paste event's clipboard and returns markdown converted from the
* rich (text/html) flavor, or null when the clipboard holds no usable rich
* content (plain pastes must fall through to the browser default). */
export function richPasteToMarkdown(event: ClipboardEvent): string | null {
const data = event.clipboardData;
if (!data) return null;
const html = data.getData("text/html");
if (!html || !html.trim()) return null;
// A rich flavor that is just an HTML wrapper of plain text (some apps
// emit <p>...</p> wrappers for everything Chrome copies) still converts
// harmlessly to the same text, so no content sniffing is needed here.
const markdown = htmlToMarkdown(html);
if (!markdown || !markdown.trim()) return null;
return markdown;
}
+20 -1
View File
@@ -5,6 +5,7 @@ import { Alpine } from "../alpine";
import { loadSettings, saveSettings } from "./settings-manager";
import { getToken } from "../storage";
import { showToast } from "../toast";
import { renderNoteHtml, richPasteToMarkdown } from "../markdown";
import {
extractPdfPages,
searchPdfPages,
@@ -435,6 +436,7 @@ document.addEventListener("alpine:init", () => {
id: string;
text: string;
note: string;
noteHtml: string;
color: string;
cfi: string;
cfiEnd: string;
@@ -443,7 +445,7 @@ document.addEventListener("alpine:init", () => {
pdfPage: number;
pdfRects: number[][];
}[],
noteItems: [] as { id: string; content: string; positionLabel: string }[],
noteItems: [] as { id: string; content: string; contentHtml: string; positionLabel: string }[],
annotationsTab: "highlights" as string,
newNoteText: "",
// Mirrors the server-side validator cap on note content
@@ -1556,6 +1558,7 @@ document.addEventListener("alpine:init", () => {
id: r.id,
text: r.selection_text ?? "",
note: r.note_text ?? "",
noteHtml: renderNoteHtml(r.note_text ?? ""),
color: r.color ?? "#ffff00",
cfi,
cfiEnd,
@@ -1642,6 +1645,7 @@ document.addEventListener("alpine:init", () => {
this.noteItems = (rows as any[]).map((r) => ({
id: r.id,
content: r.content ?? "",
contentHtml: renderNoteHtml(r.content ?? ""),
positionLabel: r.position ?? "",
}));
}
@@ -2021,6 +2025,21 @@ document.addEventListener("alpine:init", () => {
noteCountLabel(text: string): string {
return `${text.length.toLocaleString()}/${this.noteMaxLength.toLocaleString()}`;
},
// Rich clipboard (text/html) converts to markdown source at the caret;
// plain-only clipboards fall through to the browser default paste.
richNotePaste(event: ClipboardEvent) {
const markdown = richPasteToMarkdown(event);
if (markdown === null) return;
event.preventDefault();
const el = event.target as HTMLTextAreaElement | null;
if (!el) return;
const start = el.selectionStart ?? el.value.length;
const end = el.selectionEnd ?? start;
el.value = el.value.slice(0, start) + markdown + el.value.slice(end);
const caret = start + markdown.length;
el.setSelectionRange(caret, caret);
el.dispatchEvent(new Event("input", { bubbles: true }));
},
async addNote(content: string) {
const token = getToken();
if (!token || !this.mediaItemId || !content.trim()) return;
+2
View File
@@ -0,0 +1,2 @@
declare module "markdown-it-footnote";
declare module "turndown-plugin-gfm";