Files
mathew/frontend/app/utils/markdown.ts
2026-09-30 01:49:35 +02:00

176 lines
5.3 KiB
TypeScript

/**
* Markdown -> HTML rendering with LaTeX support.
*
* Pipeline: markdown-it (html disabled for safety) + markdown-it-texmath
* (KaTeX for $...$ and $$...$$) + highlight.js for fenced code blocks.
* The result is sanitized with DOMPurify (isomorphic, works during SSR).
*/
import MarkdownIt from 'markdown-it'
import texmath from 'markdown-it-texmath'
import katex from 'katex'
import hljs from 'highlight.js/lib/common'
import DOMPurify from 'isomorphic-dompurify'
import type { StateInline, Token } from 'markdown-it'
export interface TocEntry {
id: string
text: string
level: number
}
export interface RenderedMarkdown {
html: string
toc: TocEntry[]
}
const md = new MarkdownIt({
html: false, // escape raw HTML in articles — safety first
linkify: true,
breaks: false,
highlight(code: string, lang: string) {
if (lang && hljs.getLanguage(lang)) {
try {
return hljs.highlight(code, { language: lang, ignoreIllegals: true }).value
} catch {
/* fall through to plain rendering */
}
}
return ''
},
})
md.use(texmath, {
engine: katex,
delimiters: 'dollars',
katexOptions: { throwOnError: false, strict: false },
macros: {
'\\R': '\\mathbb{R}',
'\\N': '\\mathbb{N}',
'\\Z': '\\mathbb{Z}',
'\\C': '\\mathbb{C}',
'\\Q': '\\mathbb{Q}',
},
})
/** Convert a title to a URL slug — kept in sync with the backend slugify(). */
function slugify(title: string): string {
return title
.toLowerCase()
.trim()
.replace(/[^\w\s-]/g, '')
.replace(/[\s_]+/g, '-')
.replace(/-+/g, '-')
.replace(/^-|-$/g, '')
}
/** Turn heading text into a URL fragment id (kept in sync with slugs). */
function slugifyFragment(text: string): string {
return slugify(text) || 'section'
}
/**
* Wiki links: `[[Article Title]]` or `[[Article Title|display text]]`
* become links to /wiki/<slug>. Registered before the `link` rule so
* code spans, emphasis and TeX ($...$) keep their precedence.
*/
function wikilinkRule(state: StateInline, silent: boolean): boolean {
const src = state.src
const start = state.pos
// must open with [[ but not [[[
if (src.charCodeAt(start) !== 0x5b || src.charCodeAt(start + 1) !== 0x5b) return false
if (src.charCodeAt(start + 2) === 0x5b) return false
// find the closing ]] — bail on a nested [[ or end of the inline content
let end = -1
for (let p = start + 2; p < src.length - 1; p++) {
if (src.charCodeAt(p) === 0x5b && src.charCodeAt(p + 1) === 0x5b) return false
if (src.charCodeAt(p) === 0x5d && src.charCodeAt(p + 1) === 0x5d) {
end = p
break
}
}
if (end === -1) return false
const inner = src.slice(start + 2, end)
if (!inner.trim() || inner.includes('\n')) return false
const bar = inner.indexOf('|')
const target = (bar === -1 ? inner : inner.slice(0, bar)).trim()
const slug = slugify(target)
if (!slug) return false
if (silent) {
// markdown-it requires a matching rule to consume its span, even silently
state.pos = end + 2
return true
}
const label = (bar === -1 ? '' : inner.slice(bar + 1)).trim() || target
const open = state.push('link_open', 'a', 1)
open.attrs = [['href', `/wiki/${slug}`], ['class', 'wikilink']]
const text = state.push('text', '', 0)
text.content = label
state.push('link_close', 'a', -1)
state.pos = end + 2
return true
}
md.inline.ruler.before('link', 'wikilink', wikilinkRule)
/** Best-effort plain text of an inline token (for TOC labels). */
function inlineText(inline: Token): string {
let out = ''
for (const child of inline.children ?? []) {
if (child.type === 'text' || child.type === 'code_inline') out += child.content
else if (child.type === 'math_inline') out += child.content
else if (child.type === 'image') {
const alt = child.attrs?.find((a: string[]) => a[0] === 'alt')
if (alt) out += alt[1]
}
}
return out.replace(/\s+/g, ' ').trim()
}
const sanitizeConfig = {
USE_PROFILES: { html: true, mathMl: true, svg: true },
ADD_ATTR: ['target', 'class', 'style', 'aria-hidden', 'encoding'],
}
/**
* Parse markdown once: assign heading ids (for the TOC), collect the table
* of contents, render to HTML and sanitize it.
*
* `namespace` prefixes the heading ids (the reader passes the article slug),
* so that the articles open side by side never hand the document two
* headings with the same id.
*/
export function renderMarkdown(source: string, namespace?: string): RenderedMarkdown {
const src = source ?? ''
const tokens = md.parse(src, {})
const toc: TocEntry[] = []
const usedIds = new Set<string>()
const prefix = namespace ? `${slugify(namespace)}--` : ''
for (let i = 0; i < tokens.length; i++) {
const token = tokens[i]!
if (token.type !== 'heading_open') continue
const level = Number(token.tag.slice(1))
const inline = tokens[i + 1]
const text = inline?.type === 'inline' ? inlineText(inline) : ''
if (!text) continue
let id = slugifyFragment(text)
let n = 2
while (usedIds.has(id)) id = `${slugifyFragment(text)}-${n++}`
usedIds.add(id)
token.attrSet('id', `${prefix}${id}`)
if (level >= 2 && level <= 3) toc.push({ id: `${prefix}${id}`, text, level })
}
const raw = md.renderer.render(tokens, md.options, {})
const html = DOMPurify.sanitize(raw, sanitizeConfig) as unknown as string
return { html, toc }
}