feat(web): notes render as markdown — markdown-it+DOMPurify pipeline, rich paste

Objective 6 (web half). Storage/sync stay raw markdown (KOReader shows
literal source, accepted). Notes-grade syntax: headings, emphasis/
strikethrough, code, links, lists, blockquotes, GFM tables, reference
links, footnotes (markdown-it-footnote); images/raw HTML/math excluded
(html:false posture + FORBID img/style/form + default safe-scheme URI
check). Display sites: annotations-drawer note rows and highlight note
lines render pre-sanitized HTML (x-html); render-time sanitize only.

Rich paste: both note textareas intercept paste; a text/html clipboard
flavor converts via turndown (gfm tables incl. a headingless-table rule
emitting pipe syntax; images blanked) and inserts at the caret; plain
pastics fall through to the default paste unchanged.

Verified live in Brave: bold/italic/strike/code/lists/quote/table/
footnote-ref/reference-link render; <script> renders inert literal
text with no alert; javascript: hrefs absent from the DOM; paste of
rich HTML lands markdown at the caret; plain pastes untouched.
This commit is contained in:
John O'Keefe
2026-10-04 13:25:17 -04:00
parent b9f797789f
commit 4b5e193d9f
5 changed files with 143 additions and 9 deletions
+102
View File
@@ -0,0 +1,102 @@
// Notes render as markdown: storage and every sync payload carry the raw
// source string (KOReader devices show it literally — accepted), and every
// DISPLAY site renders it through this pipeline. Raw HTML is disabled at the
// parser level (markdown-it's default) and DOMPurify strips whatever the
// enabled syntax still produced (link hrefs are scheme-whitelisted below),
// so a malicious note can render nothing more than styled text.
import MarkdownIt from "markdown-it";
import footnote from "markdown-it-footnote";
import DOMPurify from "dompurify";
import TurndownService from "turndown";
import { gfm, tables } from "turndown-plugin-gfm";
const md = new MarkdownIt({ html: false, linkify: false });
md.use(footnote);
export function renderNoteHtml(source: string): string {
if (!source) return "";
const raw = md.render(source);
// Disallowing img/input/style/form by tag and keeping DOMPurify's default
// safe-URI scheme check (http/https/mailto + same-page anchors only)
// closes the link-injection hole, which is the one markdown pivot into
// script execution.
const clean = DOMPurify.sanitize(raw, {
ALLOWED_ATTR: ["href", "class", "id", "role"],
FORBID_TAGS: ["img", "input", "style", "form"],
});
return clean;
}
// Compact rows may show a plain line instead of rich markup.
export function stripNote(source: string): string {
if (!source) return "";
const holder = document.createElement("div");
holder.innerHTML = renderNoteHtml(source);
return (holder.textContent || "").replace(/\s+/g, " ").trim();
}
// ----- rich paste: HTML flavor -> markdown source -----
const turndown = new TurndownService({
// Keep the converted output inside our supported syntax: headings,
// emphasis, lists, code, quotes, links, tables. Everything else degrades.
headingStyle: "atx",
codeBlockStyle: "fenced",
bulletListMarker: "-",
hr: "---",
});
turndown.use(gfm);
turndown.use(tables);
// Images are explicitly out of scope for notes: blank them at convert time
// (remove() alone can leave the node's alt-text shape behind).
turndown.addRule("noImages", { filter: "img", replacement: () => "" });
// The gfm tables service only converts tables with a heading row and keeps
// headingless ones as raw HTML — which our no-raw-HTML parser would then
// show as literal tags. Convert those to pipe syntax with a blank header.
turndown.addRule("tablesNoHeading", {
filter: (node: Node) => {
const el = node as HTMLElement;
if (el.nodeName !== "TABLE" || !el.querySelector) return false;
return !el.querySelector("th") && el.querySelectorAll("tr").length > 0;
},
replacement: (_content: string, node: Node) => {
const table = node as HTMLTableElement;
const rows = Array.from(table.querySelectorAll("tr")).map((tr) =>
Array.from(tr.querySelectorAll("td, th")).map((cell) =>
(cell.textContent ?? "").replace(/\s+/g, " ").trim().replace(/\|/g, "\\|"),
),
);
const cols = Math.max(1, ...rows.map((r) => r.length));
const pad = (r: string[]) => {
const copy = [...r];
while (copy.length < cols) copy.push("");
return copy.map((c) => c || " ");
};
const head = pad(rows[0] ?? []);
const lines = [
"| " + head.join(" | ") + " |",
"|" + Array.from({ length: cols }, () => " --- ").join("|") + "|",
...rows.slice(1).map((r) => "| " + pad(r).join(" | ") + " |"),
];
return "\n\n" + lines.join("\n") + "\n\n";
},
});
export function htmlToMarkdown(html: string): string {
return turndown.turndown(html);
}
/** Reads the paste event's clipboard and returns markdown converted from the
* rich (text/html) flavor, or null when the clipboard holds no usable rich
* content (plain pastes must fall through to the browser default). */
export function richPasteToMarkdown(event: ClipboardEvent): string | null {
const data = event.clipboardData;
if (!data) return null;
const html = data.getData("text/html");
if (!html || !html.trim()) return null;
// A rich flavor that is just an HTML wrapper of plain text (some apps
// emit <p>...</p> wrappers for everything Chrome copies) still converts
// harmlessly to the same text, so no content sniffing is needed here.
const markdown = htmlToMarkdown(html);
if (!markdown || !markdown.trim()) return null;
return markdown;
}