fix(ui): prevent Shiki template-call OOM on backtick JS files (#2618)

fix(ui): prevent Shiki template-call OOM on backtick JS files
This commit is contained in:
Bohdan Triapitsyn
2026-08-28 23:48:37 +03:00
committed by GitHub
6 changed files with 201 additions and 11 deletions
@@ -1,6 +1,10 @@
/// <reference lib="webworker" />
import { bundledLanguages, createHighlighter, type BundledLanguage, type ThemedToken } from 'shiki';
import { bundledLanguages, createHighlighter, type BundledLanguage, type LanguageRegistration, type ThemedToken } from 'shiki';
import {
isTemplateCallLanguageId,
sanitizeTemplateCallGrammar,
} from '../../../lib/shiki/sanitizeTemplateCallGrammar';
import { MARKDOWN_SHIKI_THEME, MARKDOWN_SHIKI_THEME_DEFINITION } from './markdownShikiThemeDefinition';
import type { MarkdownWorkerRequest, MarkdownWorkerResponse } from './markdown-worker-protocol';
@@ -60,11 +64,30 @@ self.onmessage = (event: MessageEvent<MarkdownWorkerRequest>) => {
type Instance = Awaited<ReturnType<typeof createHighlighter>>;
type BundledLanguageModule = { default: LanguageRegistration[] };
/**
* Load a language, neutralizing the catastrophic JS/TS `template-call` rule
* before it reaches the Oniguruma scanner (see sanitizeTemplateCallGrammar).
*/
const loadLanguageSafe = async (instance: Instance, lang: BundledLanguage): Promise<void> => {
if (!isTemplateCallLanguageId(lang)) {
await instance.loadLanguage(bundledLanguages[lang]);
return;
}
// SAFETY: every Shiki bundled-language module default-exports its grammar
// array; `lang` is narrowed to a bundled id above.
const mod = (await bundledLanguages[lang]()) as BundledLanguageModule;
const grammars = mod.default.map((grammar) => sanitizeTemplateCallGrammar(grammar));
await instance.loadLanguage(...grammars);
};
const resolveLanguage = async (instance: Instance, requested: string): Promise<string> => {
let lang = requested in bundledLanguages ? requested : 'text';
if (lang !== 'text' && !instance.getLoadedLanguages().includes(lang)) {
try {
await instance.loadLanguage(bundledLanguages[lang as BundledLanguage]);
await loadLanguageSafe(instance, lang as BundledLanguage);
} catch {
lang = 'text';
}
@@ -0,0 +1,6 @@
/**
* Safety-net budget for a single Shiki worker tokenize request.
* Healthy files finish well under this; catastrophic Oniguruma backtracking
* must not run unbounded (openchamber/openchamber#2587).
*/
export const HIGHLIGHT_REQUEST_TIMEOUT_MS = 5_000;
@@ -0,0 +1,12 @@
import { describe, expect, test } from 'bun:test';
import { HIGHLIGHT_REQUEST_TIMEOUT_MS } from './markdown-worker-timeout';
describe('markdown-worker hang safety', () => {
test('exposes a finite highlight timeout budget', () => {
// Catastrophic Oniguruma backtracking must not run unbounded; the main
// thread terminates the worker after this budget (openchamber/openchamber#2587).
expect(HIGHLIGHT_REQUEST_TIMEOUT_MS).toBeGreaterThan(0);
expect(HIGHLIGHT_REQUEST_TIMEOUT_MS).toBeLessThan(15_001);
});
});
@@ -7,12 +7,20 @@ import {
utf16Bytes,
} from './highlightResultCache';
import type { MarkdownTokenRun, MarkdownWorkerRequest, MarkdownWorkerResponse } from './markdown-worker-protocol';
import { HIGHLIGHT_REQUEST_TIMEOUT_MS } from './markdown-worker-timeout';
// Main-thread client for the markdown Shiki worker. Moves syntax tokenization
// Main-thread client for the markdown Shiki Web Worker. Moves syntax tokenization
// off the UI thread: a closed code block is shipped to the worker, which returns
// ready-to-splice Shiki HTML. On any failure (no worker support, worker crash,
// tokenization error) the promise resolves to `null` and the caller keeps the
// escaped plain-text code — highlighting never falls back onto the main thread.
// tokenization error, or hang timeout) the promise resolves to `null` and the
// caller keeps the escaped plain-text code — highlighting never falls back onto
// the main thread.
//
// The per-request timeout exists because TextMate grammars can enter catastrophic
// backtracking on the Oniguruma WASM engine (openchamber/openchamber#2587).
// Matching is synchronous inside the worker, so the only way to reclaim its heap
// is to terminate it from this thread once a request exceeds the budget. A timed
// out request resolves `null` like any other failure, so nothing is memoized.
//
// Results are memoized by content fingerprint (+ lang / theme). Unchanged
// content must not re-enter the worker — that was the sustained ~40 msg/s
@@ -31,6 +39,11 @@ import type { MarkdownTokenRun, MarkdownWorkerRequest, MarkdownWorkerResponse }
type PendingResolver = (response: MarkdownWorkerResponse | null) => void;
type PendingEntry = {
resolve: PendingResolver;
timer: ReturnType<typeof setTimeout>;
};
type CachedHighlight =
| { type: 'highlight'; html: string }
| { type: 'highlightLines'; lines: string[] }
@@ -50,11 +63,15 @@ let worker: Worker | undefined;
let workerCreation: Promise<Worker | undefined> | undefined;
let workerObjectUrl: string | undefined;
let nextId = 0;
const pending = new Map<number, PendingResolver>();
const pending = new Map<number, PendingEntry>();
// Theme names whose full definition we've already shipped to the live worker, so
// repeat tokenization sends only the name (not the whole theme object) again.
const sentThemes = new Set<string>();
const clearPendingTimers = (): void => {
pending.forEach((entry) => clearTimeout(entry.timer));
};
const entryBytes = (key: string, value: CachedHighlight): number => {
const keyBytes = utf16Bytes(key);
if (value.type === 'highlight') return keyBytes + utf16Bytes(value.html);
@@ -67,7 +84,8 @@ const entryBytes = (key: string, value: CachedHighlight): number => {
};
const failAll = (): void => {
pending.forEach((resolve) => resolve(null));
clearPendingTimers();
pending.forEach((entry) => entry.resolve(null));
pending.clear();
sentThemes.clear();
// Drop in-flight waiters; cached results remain valid (pure fn of inputs).
@@ -95,10 +113,11 @@ const createWorker = async (): Promise<Worker | undefined> => {
const instance = new Worker(workerUrl, { type: 'module' });
worker = instance;
instance.onmessage = (event: MessageEvent<MarkdownWorkerResponse>) => {
const resolve = pending.get(event.data.id);
if (!resolve) return;
const entry = pending.get(event.data.id);
if (!entry) return;
clearTimeout(entry.timer);
pending.delete(event.data.id);
resolve(event.data);
entry.resolve(event.data);
};
instance.onerror = failAll;
instance.onmessageerror = failAll;
@@ -127,7 +146,14 @@ const request = async (payload: (id: number) => MarkdownWorkerRequest): Promise<
if (!instance) return Promise.resolve(null);
const id = ++nextId;
return new Promise<MarkdownWorkerResponse | null>((resolve) => {
pending.set(id, resolve);
const timer = setTimeout(() => {
if (!pending.has(id)) return;
// Hung tokenize (e.g. catastrophic backtracking): kill the worker so the
// WASM heap is freed instead of growing until the renderer OOMs.
console.warn(`Shiki worker highlight timed out after ${HIGHLIGHT_REQUEST_TIMEOUT_MS}ms; terminating worker`);
failAll();
}, HIGHLIGHT_REQUEST_TIMEOUT_MS);
pending.set(id, { resolve, timer });
instance.postMessage(payload(id));
});
};