All checks were successful
Build and Deploy Nuxt / build (push) Successful in 29s
249 lines
11 KiB
TypeScript
249 lines
11 KiB
TypeScript
/**
|
||
* Markdown -> HTML rendering with LaTeX support.
|
||
*
|
||
* Pipeline: markdown-it (html disabled for safety) + markdown-it-texmath
|
||
* (KaTeX for $...$ and $$...$$) + highlight.js for fenced code blocks.
|
||
* The result is sanitized with DOMPurify (isomorphic, works during SSR).
|
||
*/
|
||
import MarkdownIt from 'markdown-it'
|
||
import texmath from 'markdown-it-texmath'
|
||
import katex from 'katex'
|
||
import hljs from 'highlight.js/lib/common'
|
||
import DOMPurify from 'isomorphic-dompurify'
|
||
import type { StateInline, Token } from 'markdown-it'
|
||
|
||
export interface TocEntry {
|
||
id: string
|
||
text: string
|
||
level: number
|
||
}
|
||
|
||
export interface RenderedMarkdown {
|
||
html: string
|
||
toc: TocEntry[]
|
||
}
|
||
|
||
const md = new MarkdownIt({
|
||
html: false, // escape raw HTML in articles — safety first
|
||
linkify: true,
|
||
breaks: false,
|
||
highlight(code: string, lang: string) {
|
||
if (lang && hljs.getLanguage(lang)) {
|
||
try {
|
||
return hljs.highlight(code, { language: lang, ignoreIllegals: true }).value
|
||
} catch {
|
||
/* fall through to plain rendering */
|
||
}
|
||
}
|
||
return ''
|
||
},
|
||
})
|
||
|
||
md.use(texmath, {
|
||
engine: katex,
|
||
delimiters: 'dollars',
|
||
katexOptions: { throwOnError: false, strict: false },
|
||
macros: {
|
||
'\\R': '\\mathbb{R}',
|
||
'\\N': '\\mathbb{N}',
|
||
'\\Z': '\\mathbb{Z}',
|
||
'\\C': '\\mathbb{C}',
|
||
'\\Q': '\\mathbb{Q}',
|
||
},
|
||
})
|
||
|
||
/** Convert a title to a URL slug — kept in sync with the backend slugify(). */
|
||
function slugify(title: string): string {
|
||
return title
|
||
.toLowerCase()
|
||
.trim()
|
||
.replace(/[^\w\s-]/g, '')
|
||
.replace(/[\s_]+/g, '-')
|
||
.replace(/-+/g, '-')
|
||
.replace(/^-|-$/g, '')
|
||
}
|
||
|
||
/** Turn heading text into a URL fragment id (kept in sync with slugs). */
|
||
function slugifyFragment(text: string): string {
|
||
return slugify(text) || 'section'
|
||
}
|
||
|
||
/**
|
||
* Wiki links: `[[Article Title]]` or `[[Article Title|display text]]`
|
||
* become links to /wiki/<slug>. Registered before the `link` rule so
|
||
* code spans, emphasis and TeX ($...$) keep their precedence.
|
||
*/
|
||
function wikilinkRule(state: StateInline, silent: boolean): boolean {
|
||
const src = state.src
|
||
const start = state.pos
|
||
|
||
// must open with [[ but not [[[
|
||
if (src.charCodeAt(start) !== 0x5b || src.charCodeAt(start + 1) !== 0x5b) return false
|
||
if (src.charCodeAt(start + 2) === 0x5b) return false
|
||
|
||
// find the closing ]] — bail on a nested [[ or end of the inline content
|
||
let end = -1
|
||
for (let p = start + 2; p < src.length - 1; p++) {
|
||
if (src.charCodeAt(p) === 0x5b && src.charCodeAt(p + 1) === 0x5b) return false
|
||
if (src.charCodeAt(p) === 0x5d && src.charCodeAt(p + 1) === 0x5d) {
|
||
end = p
|
||
break
|
||
}
|
||
}
|
||
if (end === -1) return false
|
||
|
||
const inner = src.slice(start + 2, end)
|
||
if (!inner.trim() || inner.includes('\n')) return false
|
||
|
||
const bar = inner.indexOf('|')
|
||
const target = (bar === -1 ? inner : inner.slice(0, bar)).trim()
|
||
const slug = slugify(target)
|
||
if (!slug) return false
|
||
if (silent) {
|
||
// markdown-it requires a matching rule to consume its span, even silently
|
||
state.pos = end + 2
|
||
return true
|
||
}
|
||
|
||
const label = (bar === -1 ? '' : inner.slice(bar + 1)).trim() || target
|
||
|
||
const open = state.push('link_open', 'a', 1)
|
||
open.attrs = [['href', `/wiki/${slug}`], ['class', 'wikilink']]
|
||
const text = state.push('text', '', 0)
|
||
text.content = label
|
||
state.push('link_close', 'a', -1)
|
||
state.pos = end + 2
|
||
return true
|
||
}
|
||
|
||
md.inline.ruler.before('link', 'wikilink', wikilinkRule)
|
||
|
||
/**
|
||
* Math callout blocks: a fenced block tagged with one of these keywords
|
||
* (```theorem, ```example, ...) renders as a blockquote-styled box with the
|
||
* keyword's colour on its left edge instead of as code. The body is parsed as
|
||
* ordinary Markdown; any text after the keyword titles the box
|
||
* (```theorem Pythagoras). Colours live in _tokens.scss, shapes in _prose.scss.
|
||
* The label names the kind of statement *the article* makes, so it is written
|
||
* in the article's language (carried through the render env) — the reader's
|
||
* interface language never touches the page's text. Unknown languages keep English.
|
||
*/
|
||
type CalloutKind = 'example' | 'theorem' | 'corollary' | 'definition' | 'proof' | 'proposition' | 'lemma'
|
||
|
||
const CALLOUT_EN: Record<CalloutKind, string> = {
|
||
example: 'Example',
|
||
theorem: 'Theorem',
|
||
corollary: 'Corollary',
|
||
definition: 'Definition',
|
||
proof: 'Proof',
|
||
proposition: 'Proposition',
|
||
lemma: 'Lemma',
|
||
}
|
||
|
||
const CALLOUT_LABELS: Record<string, Partial<Record<CalloutKind, string>>> = {
|
||
fr: { theorem: 'Théorème', corollary: 'Corollaire', definition: 'Définition', proof: 'Démonstration', example: 'Exemple', lemma: 'Lemme' },
|
||
de: { theorem: 'Satz', corollary: 'Korollar', proof: 'Beweis', example: 'Beispiel' },
|
||
es: { theorem: 'Teorema', corollary: 'Corolario', definition: 'Definición', proof: 'Demostración', proposition: 'Proposición', example: 'Ejemplo', lemma: 'Lema' },
|
||
ca: { theorem: 'Teorema', corollary: 'Corol·lari', definition: 'Definició', proof: 'Demostració', proposition: 'Proposició', example: 'Exemple', lemma: 'Lema' },
|
||
pt: { theorem: 'Teorema', corollary: 'Corolário', definition: 'Definição', proof: 'Demonstração', proposition: 'Proposição', example: 'Exemplo', lemma: 'Lema' },
|
||
it: { theorem: 'Teorema', corollary: 'Corollario', definition: 'Definizione', proof: 'Dimostrazione', proposition: 'Proposizione', example: 'Esempio' },
|
||
nl: { theorem: 'Stelling', corollary: 'Gevolg', proof: 'Bewijs', example: 'Voorbeeld', definition: 'Definitie', proposition: 'Propositie' },
|
||
pl: { theorem: 'Twierdzenie', corollary: 'Wniosek', definition: 'Definicja', proof: 'Dowód', proposition: 'Propozycja', example: 'Przykład', lemma: 'Lemat' },
|
||
ru: { theorem: 'Теорема', corollary: 'Следствие', definition: 'Определение', proof: 'Доказательство', proposition: 'Утверждение', example: 'Пример', lemma: 'Лемма' },
|
||
uk: { theorem: 'Теорема', corollary: 'Наслідок', definition: 'Означення', proof: 'Доведення', proposition: 'Твердження', example: 'Приклад', lemma: 'Лема' },
|
||
tr: { theorem: 'Teorem', corollary: 'Sonuç', definition: 'Tanım', proof: 'İspat', proposition: 'Önerme', example: 'Örnek' },
|
||
ar: { theorem: 'مبرهنة', corollary: 'نتيجة', definition: 'تعريف', proof: 'برهان', proposition: 'قضية', example: 'مثال', lemma: 'ليما' },
|
||
fa: { theorem: 'قضیه', corollary: 'نتیجه', definition: 'تعریف', proof: 'اثبات', proposition: 'گزاره', example: 'مثال', lemma: 'لم' },
|
||
hi: { theorem: 'प्रमेय', corollary: 'निष्कर्ष', definition: 'परिभाषा', proof: 'प्रमाण', example: 'उदाहरण', proposition: 'कथन', lemma: 'उपप्रमेय' },
|
||
zh: { theorem: '定理', corollary: '推论', definition: '定义', proof: '证明', proposition: '命题', lemma: '引理', example: '例' },
|
||
ja: { theorem: '定理', corollary: '系', definition: '定義', proof: '証明', proposition: '命題', lemma: '補題', example: '例' },
|
||
ko: { theorem: '정리', corollary: '계', definition: '정의', proof: '증명', proposition: '명제', lemma: '보조정리', example: '예' },
|
||
}
|
||
|
||
function isCalloutWord(word: string): word is CalloutKind {
|
||
return Object.hasOwn(CALLOUT_EN, word)
|
||
}
|
||
|
||
const defaultFence = md.renderer.rules.fence!.bind(md.renderer)
|
||
|
||
md.renderer.rules.fence = (tokens, idx, options, env, self) => {
|
||
const token = tokens[idx]!
|
||
const info = token.info.trim()
|
||
const space = info.search(/\s/)
|
||
const word = (space === -1 ? info : info.slice(0, space)).toLowerCase()
|
||
if (!isCalloutWord(word)) return defaultFence(tokens, idx, options, env, self)
|
||
|
||
// renderMarkdown hands the article's language through the env; anything
|
||
// rendered without one (or in a language not listed above) keeps English
|
||
const lang = (env as { articleLanguage?: unknown } | undefined)?.articleLanguage
|
||
const labels = (typeof lang === 'string' && CALLOUT_LABELS[lang.trim().toLowerCase()]) || CALLOUT_EN
|
||
const kind = labels[word] ?? CALLOUT_EN[word]
|
||
const title = space === -1 ? '' : info.slice(space + 1).trim()
|
||
const label = title ? `${kind} — ${md.utils.escapeHtml(title)}` : kind
|
||
const body = md.render(token.content, env).trim()
|
||
// a native <details> makes the label row the toggle: it works on the first
|
||
// paint and from the keyboard, and MarkdownView animates the fold; every
|
||
// box starts open. The body is one wrapper so the fold has a single thing
|
||
// to measure and shrink.
|
||
return `<blockquote class="md-box md-box--${word}"><details open><summary class="md-box-title">${label}</summary><div class="md-box-body">${body}</div></details></blockquote>\n`
|
||
}
|
||
|
||
/** Best-effort plain text of an inline token (for TOC labels). */
|
||
function inlineText(inline: Token): string {
|
||
let out = ''
|
||
for (const child of inline.children ?? []) {
|
||
if (child.type === 'text' || child.type === 'code_inline') out += child.content
|
||
else if (child.type === 'math_inline') out += child.content
|
||
else if (child.type === 'image') {
|
||
const alt = child.attrs?.find((a: string[]) => a[0] === 'alt')
|
||
if (alt) out += alt[1]
|
||
}
|
||
}
|
||
return out.replace(/\s+/g, ' ').trim()
|
||
}
|
||
|
||
const sanitizeConfig = {
|
||
USE_PROFILES: { html: true, mathMl: true, svg: true },
|
||
ADD_ATTR: ['target', 'class', 'style', 'aria-hidden', 'encoding'],
|
||
}
|
||
|
||
/**
|
||
* Parse markdown once: assign heading ids (for the TOC), collect the table
|
||
* of contents, render to HTML and sanitize it.
|
||
*
|
||
* `namespace` prefixes the heading ids (the reader passes the article slug),
|
||
* so that the articles open side by side never hand the document two
|
||
* headings with the same id.
|
||
*
|
||
* `language` is the article's own language — callout boxes (```theorem, …)
|
||
* write their labels in it, never in the interface language.
|
||
*/
|
||
export function renderMarkdown(source: string, namespace?: string, language?: string): RenderedMarkdown {
|
||
const src = source ?? ''
|
||
const tokens = md.parse(src, {})
|
||
const toc: TocEntry[] = []
|
||
const usedIds = new Set<string>()
|
||
const prefix = namespace ? `${slugify(namespace)}--` : ''
|
||
|
||
for (let i = 0; i < tokens.length; i++) {
|
||
const token = tokens[i]!
|
||
if (token.type !== 'heading_open') continue
|
||
const level = Number(token.tag.slice(1))
|
||
const inline = tokens[i + 1]
|
||
const text = inline?.type === 'inline' ? inlineText(inline) : ''
|
||
if (!text) continue
|
||
|
||
let id = slugifyFragment(text)
|
||
let n = 2
|
||
while (usedIds.has(id)) id = `${slugifyFragment(text)}-${n++}`
|
||
usedIds.add(id)
|
||
token.attrSet('id', `${prefix}${id}`)
|
||
|
||
if (level >= 2 && level <= 3) toc.push({ id: `${prefix}${id}`, text, level })
|
||
}
|
||
|
||
const raw = md.renderer.render(tokens, md.options, { articleLanguage: language })
|
||
const html = DOMPurify.sanitize(raw, sanitizeConfig) as unknown as string
|
||
return { html, toc }
|
||
}
|