Files
mathew/frontend/app/utils/markdown.ts
Aran Roig e36fef7290
All checks were successful
Build and Deploy Nuxt / build (push) Successful in 29s
Calc2
2026-10-02 02:14:29 +02:00

249 lines
11 KiB
TypeScript
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
/**
* Markdown -> HTML rendering with LaTeX support.
*
* Pipeline: markdown-it (html disabled for safety) + markdown-it-texmath
* (KaTeX for $...$ and $$...$$) + highlight.js for fenced code blocks.
* The result is sanitized with DOMPurify (isomorphic, works during SSR).
*/
import MarkdownIt from 'markdown-it'
import texmath from 'markdown-it-texmath'
import katex from 'katex'
import hljs from 'highlight.js/lib/common'
import DOMPurify from 'isomorphic-dompurify'
import type { StateInline, Token } from 'markdown-it'
export interface TocEntry {
id: string
text: string
level: number
}
export interface RenderedMarkdown {
html: string
toc: TocEntry[]
}
const md = new MarkdownIt({
html: false, // escape raw HTML in articles — safety first
linkify: true,
breaks: false,
highlight(code: string, lang: string) {
if (lang && hljs.getLanguage(lang)) {
try {
return hljs.highlight(code, { language: lang, ignoreIllegals: true }).value
} catch {
/* fall through to plain rendering */
}
}
return ''
},
})
md.use(texmath, {
engine: katex,
delimiters: 'dollars',
katexOptions: { throwOnError: false, strict: false },
macros: {
'\\R': '\\mathbb{R}',
'\\N': '\\mathbb{N}',
'\\Z': '\\mathbb{Z}',
'\\C': '\\mathbb{C}',
'\\Q': '\\mathbb{Q}',
},
})
/** Convert a title to a URL slug — kept in sync with the backend slugify(). */
function slugify(title: string): string {
return title
.toLowerCase()
.trim()
.replace(/[^\w\s-]/g, '')
.replace(/[\s_]+/g, '-')
.replace(/-+/g, '-')
.replace(/^-|-$/g, '')
}
/** Turn heading text into a URL fragment id (kept in sync with slugs). */
function slugifyFragment(text: string): string {
return slugify(text) || 'section'
}
/**
* Wiki links: `[[Article Title]]` or `[[Article Title|display text]]`
* become links to /wiki/<slug>. Registered before the `link` rule so
* code spans, emphasis and TeX ($...$) keep their precedence.
*/
function wikilinkRule(state: StateInline, silent: boolean): boolean {
const src = state.src
const start = state.pos
// must open with [[ but not [[[
if (src.charCodeAt(start) !== 0x5b || src.charCodeAt(start + 1) !== 0x5b) return false
if (src.charCodeAt(start + 2) === 0x5b) return false
// find the closing ]] — bail on a nested [[ or end of the inline content
let end = -1
for (let p = start + 2; p < src.length - 1; p++) {
if (src.charCodeAt(p) === 0x5b && src.charCodeAt(p + 1) === 0x5b) return false
if (src.charCodeAt(p) === 0x5d && src.charCodeAt(p + 1) === 0x5d) {
end = p
break
}
}
if (end === -1) return false
const inner = src.slice(start + 2, end)
if (!inner.trim() || inner.includes('\n')) return false
const bar = inner.indexOf('|')
const target = (bar === -1 ? inner : inner.slice(0, bar)).trim()
const slug = slugify(target)
if (!slug) return false
if (silent) {
// markdown-it requires a matching rule to consume its span, even silently
state.pos = end + 2
return true
}
const label = (bar === -1 ? '' : inner.slice(bar + 1)).trim() || target
const open = state.push('link_open', 'a', 1)
open.attrs = [['href', `/wiki/${slug}`], ['class', 'wikilink']]
const text = state.push('text', '', 0)
text.content = label
state.push('link_close', 'a', -1)
state.pos = end + 2
return true
}
md.inline.ruler.before('link', 'wikilink', wikilinkRule)
/**
* Math callout blocks: a fenced block tagged with one of these keywords
* (```theorem, ```example, ...) renders as a blockquote-styled box with the
* keyword's colour on its left edge instead of as code. The body is parsed as
* ordinary Markdown; any text after the keyword titles the box
* (```theorem Pythagoras). Colours live in _tokens.scss, shapes in _prose.scss.
* The label names the kind of statement *the article* makes, so it is written
* in the article's language (carried through the render env) — the reader's
* interface language never touches the page's text. Unknown languages keep English.
*/
type CalloutKind = 'example' | 'theorem' | 'corollary' | 'definition' | 'proof' | 'proposition' | 'lemma'
const CALLOUT_EN: Record<CalloutKind, string> = {
example: 'Example',
theorem: 'Theorem',
corollary: 'Corollary',
definition: 'Definition',
proof: 'Proof',
proposition: 'Proposition',
lemma: 'Lemma',
}
const CALLOUT_LABELS: Record<string, Partial<Record<CalloutKind, string>>> = {
fr: { theorem: 'Théorème', corollary: 'Corollaire', definition: 'Définition', proof: 'Démonstration', example: 'Exemple', lemma: 'Lemme' },
de: { theorem: 'Satz', corollary: 'Korollar', proof: 'Beweis', example: 'Beispiel' },
es: { theorem: 'Teorema', corollary: 'Corolario', definition: 'Definición', proof: 'Demostración', proposition: 'Proposición', example: 'Ejemplo', lemma: 'Lema' },
ca: { theorem: 'Teorema', corollary: 'Corol·lari', definition: 'Definició', proof: 'Demostració', proposition: 'Proposició', example: 'Exemple', lemma: 'Lema' },
pt: { theorem: 'Teorema', corollary: 'Corolário', definition: 'Definição', proof: 'Demonstração', proposition: 'Proposição', example: 'Exemplo', lemma: 'Lema' },
it: { theorem: 'Teorema', corollary: 'Corollario', definition: 'Definizione', proof: 'Dimostrazione', proposition: 'Proposizione', example: 'Esempio' },
nl: { theorem: 'Stelling', corollary: 'Gevolg', proof: 'Bewijs', example: 'Voorbeeld', definition: 'Definitie', proposition: 'Propositie' },
pl: { theorem: 'Twierdzenie', corollary: 'Wniosek', definition: 'Definicja', proof: 'Dowód', proposition: 'Propozycja', example: 'Przykład', lemma: 'Lemat' },
ru: { theorem: 'Теорема', corollary: 'Следствие', definition: 'Определение', proof: 'Доказательство', proposition: 'Утверждение', example: 'Пример', lemma: 'Лемма' },
uk: { theorem: 'Теорема', corollary: 'Наслідок', definition: 'Означення', proof: 'Доведення', proposition: 'Твердження', example: 'Приклад', lemma: 'Лема' },
tr: { theorem: 'Teorem', corollary: 'Sonuç', definition: 'Tanım', proof: 'İspat', proposition: 'Önerme', example: 'Örnek' },
ar: { theorem: 'مبرهنة', corollary: 'نتيجة', definition: 'تعريف', proof: 'برهان', proposition: 'قضية', example: 'مثال', lemma: 'ليما' },
fa: { theorem: 'قضیه', corollary: 'نتیجه', definition: 'تعریف', proof: 'اثبات', proposition: 'گزاره', example: 'مثال', lemma: 'لم' },
hi: { theorem: 'प्रमेय', corollary: 'निष्कर्ष', definition: 'परिभाषा', proof: 'प्रमाण', example: 'उदाहरण', proposition: 'कथन', lemma: 'उपप्रमेय' },
zh: { theorem: '定理', corollary: '推论', definition: '定义', proof: '证明', proposition: '命题', lemma: '引理', example: '例' },
ja: { theorem: '定理', corollary: '系', definition: '定義', proof: '証明', proposition: '命題', lemma: '補題', example: '例' },
ko: { theorem: '정리', corollary: '계', definition: '정의', proof: '증명', proposition: '명제', lemma: '보조정리', example: '예' },
}
function isCalloutWord(word: string): word is CalloutKind {
return Object.hasOwn(CALLOUT_EN, word)
}
const defaultFence = md.renderer.rules.fence!.bind(md.renderer)
md.renderer.rules.fence = (tokens, idx, options, env, self) => {
const token = tokens[idx]!
const info = token.info.trim()
const space = info.search(/\s/)
const word = (space === -1 ? info : info.slice(0, space)).toLowerCase()
if (!isCalloutWord(word)) return defaultFence(tokens, idx, options, env, self)
// renderMarkdown hands the article's language through the env; anything
// rendered without one (or in a language not listed above) keeps English
const lang = (env as { articleLanguage?: unknown } | undefined)?.articleLanguage
const labels = (typeof lang === 'string' && CALLOUT_LABELS[lang.trim().toLowerCase()]) || CALLOUT_EN
const kind = labels[word] ?? CALLOUT_EN[word]
const title = space === -1 ? '' : info.slice(space + 1).trim()
const label = title ? `${kind} — ${md.utils.escapeHtml(title)}` : kind
const body = md.render(token.content, env).trim()
// a native <details> makes the label row the toggle: it works on the first
// paint and from the keyboard, and MarkdownView animates the fold; every
// box starts open. The body is one wrapper so the fold has a single thing
// to measure and shrink.
return `<blockquote class="md-box md-box--${word}"><details open><summary class="md-box-title">${label}</summary><div class="md-box-body">${body}</div></details></blockquote>\n`
}
/** Best-effort plain text of an inline token (for TOC labels). */
function inlineText(inline: Token): string {
let out = ''
for (const child of inline.children ?? []) {
if (child.type === 'text' || child.type === 'code_inline') out += child.content
else if (child.type === 'math_inline') out += child.content
else if (child.type === 'image') {
const alt = child.attrs?.find((a: string[]) => a[0] === 'alt')
if (alt) out += alt[1]
}
}
return out.replace(/\s+/g, ' ').trim()
}
const sanitizeConfig = {
USE_PROFILES: { html: true, mathMl: true, svg: true },
ADD_ATTR: ['target', 'class', 'style', 'aria-hidden', 'encoding'],
}
/**
* Parse markdown once: assign heading ids (for the TOC), collect the table
* of contents, render to HTML and sanitize it.
*
* `namespace` prefixes the heading ids (the reader passes the article slug),
* so that the articles open side by side never hand the document two
* headings with the same id.
*
* `language` is the article's own language — callout boxes (```theorem, …)
* write their labels in it, never in the interface language.
*/
export function renderMarkdown(source: string, namespace?: string, language?: string): RenderedMarkdown {
const src = source ?? ''
const tokens = md.parse(src, {})
const toc: TocEntry[] = []
const usedIds = new Set<string>()
const prefix = namespace ? `${slugify(namespace)}--` : ''
for (let i = 0; i < tokens.length; i++) {
const token = tokens[i]!
if (token.type !== 'heading_open') continue
const level = Number(token.tag.slice(1))
const inline = tokens[i + 1]
const text = inline?.type === 'inline' ? inlineText(inline) : ''
if (!text) continue
let id = slugifyFragment(text)
let n = 2
while (usedIds.has(id)) id = `${slugifyFragment(text)}-${n++}`
usedIds.add(id)
token.attrSet('id', `${prefix}${id}`)
if (level >= 2 && level <= 3) toc.push({ id: `${prefix}${id}`, text, level })
}
const raw = md.renderer.render(tokens, md.options, { articleLanguage: language })
const html = DOMPurify.sanitize(raw, sanitizeConfig) as unknown as string
return { html, toc }
}