From a1bd478be552a6abd0a2cc7edb8095f5252aff84 Mon Sep 17 00:00:00 2001 From: Tanner Linsley Date: Thu, 1 Oct 2026 08:33:48 -0600 Subject: [PATCH] docs: clarify Highlight registration and annotations --- README.md | 14 ++++++------- docs/architecture.md | 4 +++- docs/comparison.md | 10 +++++----- docs/guides/annotations.md | 6 +++--- docs/guides/language-registration.md | 4 ++-- docs/guides/markdown-pipelines.md | 2 +- docs/guides/performance.md | 20 ++++++++++--------- docs/guides/ssr-and-client.md | 4 ++-- docs/language-inventory.md | 4 ++-- docs/overview.md | 9 +++------ docs/reference/core.md | 10 ++++++---- docs/test-strategy.md | 2 +- .../configure-selective-highlighting/SKILL.md | 4 ++-- skills/theme-and-annotate-code/SKILL.md | 4 ++-- 14 files changed, 50 insertions(+), 47 deletions(-) diff --git a/README.md b/README.md index d75ba3a..c3b4711 100644 --- a/README.md +++ b/README.md @@ -76,7 +76,7 @@ import { highlight } from '@tanstack/highlight' const result = highlight(`const value = 'docs'`, { lang: 'ts' }) ``` -Unknown languages fall back to escaped plaintext. +Unknown languages use the configured fallback, which defaults to escaped plaintext. ## SSR And Client @@ -210,17 +210,17 @@ Each language is available from `@tanstack/highlight/languages/`. The aggr ## Size And Speed -Local browser bundles, minified with esbuild and compressed independently. KB uses 1,000 bytes: +Local browser bundles for 1.0.0, measured on 2026-10-01 with Node 26.3.1 and esbuild 0.28.1, minified and compressed independently. KB uses 1,000 bytes: | Registration | Minified | Gzip | Brotli | | --- | ---: | ---: | ---: | | Core, no languages | 3.84 KB | 1.82 KB | 1.66 KB | -| Core + TSX | 10.11 KB | 4.26 KB | 3.90 KB | -| Octane MDX + TypeScript | 13.86 KB | 5.59 KB | 5.17 KB | -| Nine-language docs set | 16.04 KB | 6.20 KB | 5.66 KB | -| All 30 languages | 30.72 KB | 10.77 KB | 9.79 KB | +| Core + TSX | 10.16 KB | 4.29 KB | 3.93 KB | +| Octane MDX + TypeScript | 13.91 KB | 5.62 KB | 5.21 KB | +| Nine-language docs set | 16.09 KB | 6.22 KB | 5.68 KB | +| All 30 languages | 30.77 KB | 10.79 KB | 9.79 KB | -The following comparison was measured before the 1.0 property-context correction. Re-run the comparison commands below for current timings. +The comparisons below were measured before the 1.0 property-context correction. Re-run the comparison commands below for current timings and output sizes. On 80 real JavaScript/TypeScript/JSX/TSX TanStack docs fixtures repeated across 5,040 blocks, using the median of three runs after warmup: diff --git a/docs/architecture.md b/docs/architecture.md index 90c37fc..76a913a 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -12,11 +12,13 @@ A language returns ordered, non-overlapping character ranges with semantic token ## Context Instead Of A Grammar Runtime -Most languages use priority-ordered regular expressions. Three cases use small stateful scanners because regex-only matching loses necessary context: +Most languages use priority-ordered regular expressions. Context-sensitive cases use focused scanners, including: - JavaScript and TypeScript strings, comments, regular expressions, JSX tags, and recursive template interpolation - Markup tags, attributes, and embedded script/style regions - Shell heredocs and parameter expansions +- CMake bracket strings, nested variables, and generator expressions +- PHP tags, strings, and heredoc/nowdoc bodies This is the package's complexity boundary. It deliberately does not implement TextMate repositories, captures, scopes, backreferences, or a general recursive grammar DSL. diff --git a/docs/comparison.md b/docs/comparison.md index 81000d5..5643a8f 100644 --- a/docs/comparison.md +++ b/docs/comparison.md @@ -11,8 +11,8 @@ Syntax highlighters solve different problems. The smallest correct choice is the | Choose | Best fit | Main tradeoff | | --- | --- | --- | | **TanStack Highlight** | Blogs and docs that know their languages, need SSR/client parity, and value small HTML and selective imports | Focused heuristics instead of editor-grade grammars | -| **Sugar High** | The smallest practical JavaScript/TypeScript/JSX experience | Narrower language set and much more generated markup in our fixture comparison | -| **Shiki** | VS Code-like accuracy, TextMate grammars, and broad language/theme compatibility | Much larger runtime and asynchronous setup | +| **Sugar High** | Compact JavaScript/TypeScript/JSX highlighting | Narrower language set and much more generated markup in our fixture comparison | +| **Shiki** | VS Code-like accuracy, TextMate grammars, and broad language/theme compatibility | Grammar and engine setup; costs depend on the selected bundle and engine | | **highlight.js** | Broad language coverage and optional automatic detection | Larger modular core and detection/parser work not needed by known-language docs | | **Prism** | Mature grammar ecosystem and plugin integrations | Grammar composition and plugin architecture add more surface than this use case needs | | **speed-highlight** | Small, modular, class-based highlighting across browser and terminal use cases | Different grammar/output model and fewer docs-specific adapters | @@ -40,6 +40,8 @@ Syntax highlighters solve different problems. The smallest correct choice is the ## Measured overlap with Sugar High +The recorded comparisons below precede the 1.0 property-context correction. Run `pnpm run compare:sugar-high` and `pnpm run compare:shiki` for results from your checkout; see [Bundle Size and Performance](guides/performance) for current size profiles. + The repository compares its installed Sugar High version against the overlapping JavaScript, TypeScript, JSX, and TSX fixture set. Timings are the median of three samples after two warmup passes. Bundle KB uses 1,000 bytes; HTML sizes count UTF-8 bytes. | | Bundle gzip | 5,040-block runtime | Generated HTML | @@ -47,7 +49,7 @@ The repository compares its installed Sugar High version against the overlapping | TanStack Highlight | 4.11 KB | 78 ms | 6.7 MiB | | Sugar High 1.2.1 | 3.28 KB | 377 ms | 44.8 MiB | -Sugar High wins raw JavaScript bundle size. TanStack Highlight spends about 830 additional gzip bytes on its registry, class-based output, decorations, and broader context handling, then produces substantially less HTML in this corpus. +In this recorded profile, Sugar High's gzip bundle is about 830 bytes smaller, while TanStack Highlight produces less HTML. The profile includes Highlight's registry, class-based output, decorations, and context handling; it does not isolate the cost of each feature. ## Measured overlap with Shiki @@ -73,5 +75,3 @@ Choose another tool when you need: - Semantic tokens from a language service Run `pnpm run compare:sugar-high` and `pnpm run compare:shiki` to reproduce the repository's comparisons. - -These recorded comparison timings precede the 1.0 property-context correction. Run the comparison scripts for current measurements; see the performance guide for current bundle sizes. diff --git a/docs/guides/annotations.md b/docs/guides/annotations.md index 0bfcc9c..0178544 100644 --- a/docs/guides/annotations.md +++ b/docs/guides/annotations.md @@ -4,7 +4,7 @@ title: Annotations # Annotations -Annotations add application classes and data to lines or exact source ranges without changing the token stream. +Annotations add application classes and data to lines or exact source ranges without changing the token stream. Your application supplies the selections and messages; Highlight does not analyze code for errors or infer changes. ## Line decorations @@ -59,7 +59,7 @@ This adds `th-code--line-numbers`, line wrappers, and `data-line`. Base theme CS ## Try annotations -The rendered block combines line numbers, a focused line, and an exact character-range diagnostic. +The rendered block combines line numbers, a focused line, and an error annotation on a character range. The example supplies the identifier range and message directly. ```ts group=highlight-annotations file=/src/main.ts entry env=client import { createHighlighter } from '@tanstack/highlight/core' @@ -107,7 +107,7 @@ code { font-family: ui-monospace, SFMono-Regular, Consolas, monospace; font-size .hint { margin: 12px 4px 0; color: color-mix(in srgb, currentColor 70%, transparent); font-size: 13px; }` document.head.append(style) - output.innerHTML = `${result.html}

Hover the underlined identifier to read its diagnostic.

` + output.innerHTML = `${result.html}

Hover the underlined identifier to read the supplied error message.

` const diagnostic = output.querySelector('.is-error') if (diagnostic) diagnostic.title = diagnostic.dataset.message ?? '' diff --git a/docs/guides/language-registration.md b/docs/guides/language-registration.md index ea8afd2..29ef046 100644 --- a/docs/guides/language-registration.md +++ b/docs/guides/language-registration.md @@ -4,7 +4,7 @@ title: Language Registration # Language Registration -Language registration is the main bundle-size control in TanStack Highlight. +Imports determine which language modules can reach your bundle; registration determines which of those languages the highlighter can use. Import definitions from direct subpaths and register the languages your content needs. Removing a definition from the registry does not guarantee that a bundler removes its imported module. ## Create a selective highlighter @@ -92,7 +92,7 @@ export const highlighter = createHighlighter({ }) ``` -There is no benefit to creating one per block. The registry is immutable after construction and safe to reuse across server requests and client renders. +Reuse the highlighter across blocks and server requests to avoid rebuilding its name and alias maps. There is no API for adding registrations after construction; create another highlighter when the language set changes. Language definitions are retained by reference, so keep their tokenizers unchanged and free of request-specific state. ## Aggregate imports diff --git a/docs/guides/markdown-pipelines.md b/docs/guides/markdown-pipelines.md index 2f2a558..2a60bb2 100644 --- a/docs/guides/markdown-pipelines.md +++ b/docs/guides/markdown-pipelines.md @@ -33,7 +33,7 @@ export function Article({ source }: { source: string }) { } ``` -The adapter maps parsed highlighted lines to `th-line--highlighted`, preserves line-number wrappers, escapes source text, and degrades unknown languages to escaped plaintext. It does not import TanStack Markdown or any languages. +The adapter maps parsed highlighted lines to `th-line--highlighted`, preserves line-number wrappers, escapes source text, and uses the configured fallback for unknown languages. The default fallback is plaintext. It does not import TanStack Markdown or any languages. Generate theme CSS against Markdown's wrapper classes: diff --git a/docs/guides/performance.md b/docs/guides/performance.md index 11df7d7..3fef7a9 100644 --- a/docs/guides/performance.md +++ b/docs/guides/performance.md @@ -10,15 +10,17 @@ CI measures selective browser bundles and highlighting performance on real docum `pnpm run size` builds seventeen browser profiles with esbuild and measures minified, gzip, and Brotli bytes independently. It also checks that helper, adapter, and selective language imports retain only the requested modules. -| Profile | Registered languages | Current gzip | CI budget | +The gzip results below are a local 1.0.0 measurement from 2026-10-01 using Node 26.3.1 and esbuild 0.28.1. Compression results can vary with the toolchain. + +| Profile | Registered languages | Measured gzip | CI budget | | --- | --- | ---: | ---: | | Core | None | 1.82 KB | 2.0 KB | -| TSX | TSX | 4.26 KB | 4.35 KB | -| Octane | TypeScript plus Octane MDX adapter | 5.59 KB | 5.7 KB | -| Docs | CSS, HTML, JS, JSON, JSX, Markdown, Shell, TS, TSX | 6.20 KB | 6.3 KB | -| All | All 30 definitions | 10.77 KB | 10.9 KB | +| TSX | TSX | 4.29 KB | 4.35 KB | +| Octane | TypeScript plus Octane MDX adapter | 5.62 KB | 5.7 KB | +| Docs | CSS, HTML, JS, JSON, JSX, Markdown, Shell, TS, TSX | 6.22 KB | 6.3 KB | +| All | All 30 definitions | 10.79 KB | 10.9 KB | -KB uses 1,000 bytes. Core helpers imported from the root tree-shake to the same engine size. The standalone theme helper is 695 gzip bytes. +KB uses 1,000 bytes. Core helpers imported from the root tree-shake to the same engine size. The standalone theme helper measured 691 gzip bytes in the same run. Selective profiles are the primary metric. The all-language profile exists to prevent convenience-entry growth from becoming invisible. @@ -28,7 +30,7 @@ The committed corpus contains 334 real code fences sampled from TanStack documen `pnpm run bench` measures tokenization, HTML, Markdown, HAST, line numbers, long decorated blocks, and dedicated C++, CMake, and PHP samples. Each profile reports the median of three samples after two warmup passes, with a 1.2 second CI budget. The main highlighting profile processes at least 10,000 blocks. -A local before-and-after review used the same minified bundle settings, fixtures, and benchmark harness on macOS arm64 with Node 24.15.0: +A historical before-and-after review of the 0.0.11 optimization used the same minified bundle settings, fixtures, and benchmark harness on macOS arm64 with Node 24.15.0: | Workload | Blocks | Before | After | | --- | ---: | ---: | ---: | @@ -38,7 +40,7 @@ A local before-and-after review used the same minified bundle settings, fixtures | 1,000-line numbered blocks | 50 | 358 ms | 68 ms | | 1,000-line decorated blocks | 50 | 426 ms | 150 ms | -Generated HTML byte totals were unchanged. The core avoids rescanning earlier tokens for each line, and HAST adapters skip HTML serialization. Highlighter bundles grew by 41 to 173 gzip bytes across the five profiles, while root helper imports and theme CSS generation became smaller. The Octane gzip budget increased from 5.2 KB to 5.5 KB to accommodate correct fence metadata and attribute preservation. +Generated HTML byte totals were unchanged. The core avoids rescanning earlier tokens for each line, and HAST adapters skip HTML serialization. Highlighter bundles grew by 41 to 173 gzip bytes across the five profiles, while root helper imports and theme CSS generation became smaller. At that time, the Octane gzip budget increased from 5.2 KB to 5.5 KB to accommodate correct fence metadata and attribute preservation. The current budget is in the bundle profile table above. ## Comparison scripts @@ -76,4 +78,4 @@ Context-aware fixes are welcome when they solve common docs code. A change shoul The correct response to a crossed budget is to inspect the behavior and architecture. Budgets can move when a measured quality improvement justifies the bytes, but the tradeoff must be explicit. -The 1.0 property-context correction adds roughly 230 gzip bytes to the TSX profile without changing core. Local Node 26 gzip results differ slightly from CI compression: the CI docs profile is 6,217 bytes and all languages is 10,838 bytes. Their budgets are 6,300 and 10,900 bytes respectively, retaining a small explicit margin. +The 1.0 property-context correction increased the measured TSX bundle without changing core. Use `pnpm run size` to check the current profile sizes and their remaining budget margins rather than relying on an earlier run. diff --git a/docs/guides/ssr-and-client.md b/docs/guides/ssr-and-client.md index 0ccb4bd..7f00b78 100644 --- a/docs/guides/ssr-and-client.md +++ b/docs/guides/ssr-and-client.md @@ -75,10 +75,10 @@ Themes do not affect highlighted markup. ## Caching -The highlighter is fast enough for normal docs pages without a cache. For large static builds, cache by package version, language, source, and decoration options if build time becomes material. +Measure your build before adding a cache. If highlighting takes a material share of build time, cache by package version, language registrations, language, source, decorations, and line-number options. Do not cache theme variants separately. They share the same HTML. ## Why no worker? -The docs-sized path is synchronous and usually much cheaper than worker startup and message serialization. A worker is reasonable only for unusually large interactive inputs, which is outside the primary use case. +The synchronous API does not require a worker. For large interactive inputs, measure how long highlighting blocks the UI thread and consider an input limit or a worker. Worker startup and message serialization add costs of their own. diff --git a/docs/language-inventory.md b/docs/language-inventory.md index ac0cd65..a55e7d5 100644 --- a/docs/language-inventory.md +++ b/docs/language-inventory.md @@ -4,14 +4,14 @@ title: Language Inventory # TanStack Docs Language Inventory -Generated from local markdown and MDX files in: +This recorded inventory was generated from local Markdown and MDX files in: - Sibling repositories under `../*/docs` - The sibling `../tanstack.com` repository The scan parses fenced code blocks statefully, so closing fences are not counted as plaintext blocks. Hidden directories and this `highlight` package are ignored. -Scanned files: `2940` +Scanned files in this snapshot: `2940`. Counts depend on the sibling checkouts and are not a current inventory of every TanStack repository. Run `pnpm run scan:languages` from a checkout with the source repositories available to produce a new inventory. Generated real-code fixtures: diff --git a/docs/overview.md b/docs/overview.md index f07aad8..ec3188a 100644 --- a/docs/overview.md +++ b/docs/overview.md @@ -16,16 +16,13 @@ It occupies the space between minimal JavaScript-only scanners and full grammar ## Why another highlighter? -Most syntax highlighters optimize for one of two different products: +A documentation site may need TypeScript, markup, shell commands, and configuration files in the same page. TanStack Highlight lets you register that mix of languages, render code during SSR, and highlight new samples in the browser with the same synchronous API. -1. **Editors and exact grammar compatibility.** These tools support TextMate grammars, hundreds of languages, editor scopes, and VS Code themes. That capability has a real initialization and bundle cost. -2. **A single tiny language scanner.** These tools can be extremely small, but may not cover a documentation site's mix of markup, shell, data, and framework files. - -TanStack Highlight optimizes for a third product: valid code samples rendered repeatedly in documentation. It gives up automatic language detection, editor state, semantic language-service tokens, and TextMate compatibility in exchange for a small synchronous path that can run on both server and client. +Its tokenizers target common, valid documentation samples. They do not provide automatic language detection, editor state, semantic language-service tokens, or TextMate compatibility. See [Comparison](comparison) if you need those capabilities. ## Core principles -### Selective by default +### Selective imports The `@tanstack/highlight/core` entry contains no shipped languages. Every language is an isolated `LanguageDefinition` under `@tanstack/highlight/languages/*`. diff --git a/docs/reference/core.md b/docs/reference/core.md index 057489a..7c06887 100644 --- a/docs/reference/core.md +++ b/docs/reference/core.md @@ -44,7 +44,7 @@ type TokenizerContext = { } ``` -Passed to language tokenizers for opt-in embedded-language delegation. Recursive calls are capped at 24 levels. +Passed to language tokenizers for opt-in embedded-language delegation. The initial call has depth zero; delegated calls beyond depth 24 return no token ranges. ### `defineLanguage` @@ -65,7 +65,9 @@ function createHighlighter(options: { }): Highlighter ``` -Builds an immutable highlighter interface from the supplied registrations. The default fallback name is `plaintext`. An unregistered language, including an unregistered fallback, is rendered as escaped plain text. +Builds a highlighter with name and alias maps from the supplied registrations. There is no method for adding registrations later. The returned object and language definitions are not frozen; keep shared definitions unchanged. + +The default fallback name is `plaintext`. Unknown language names resolve to the configured fallback and use its tokenizer if registered. An unregistered fallback produces escaped plain text under that fallback name. Duplicate canonical names or aliases use the last registration encountered. @@ -124,7 +126,7 @@ type RenderedCodeBlockData = { } ``` -`renderCodeBlockData` trims trailing whitespace before creating every field. +`renderCodeBlockData` applies `code.trimEnd()` before generating `copyText`, `htmlMarkup`, and `tokens`. It preserves the supplied `title` and normalizes `lang`. ## Tokens @@ -141,7 +143,7 @@ type HighlightToken = { } ``` -Untyped source segments omit `className`. Concatenating every `value` always reconstructs the input exactly. +Untyped source segments omit `className`. Concatenating every `value` reconstructs the string passed to tokenization. For block-data and fence helpers, that string has already had trailing whitespace removed. ## Decorations diff --git a/docs/test-strategy.md b/docs/test-strategy.md index 06d4e1d..dd41d0c 100644 --- a/docs/test-strategy.md +++ b/docs/test-strategy.md @@ -13,7 +13,7 @@ The suite protects the package's actual product boundary: valid code commonly pu - Every token stream reconstructs its source byte for byte. - Focused regressions cover context-sensitive failures such as TSX generics, nested template interpolation, regular expressions, Python triple strings, shell heredocs, YAML fragments and block scalars, and markup embeddings. - HTML uses one escaped `
` tree with no inline style attributes.
-- Unknown languages fall back to plaintext.
+- Unknown languages use the configured fallback, which defaults to plaintext.
 - Remark and rehype produce structured nodes and do not require raw HTML.
 - Every public ESM subpath imports directly from the packed package shape.
 
diff --git a/skills/configure-selective-highlighting/SKILL.md b/skills/configure-selective-highlighting/SKILL.md
index 026968e..c7da2a9 100644
--- a/skills/configure-selective-highlighting/SKILL.md
+++ b/skills/configure-selective-highlighting/SKILL.md
@@ -127,7 +127,7 @@ export const html = highlighter.highlightToHtml('const answer = 42', {
 })
 ```
 
-The root entry constructs and retains the all-language registry.
+The root bound helper used here retains the all-language registry. Core-only imports from the root can tree-shake in a compatible bundler; direct `/core` imports make the separation explicit.
 
 Source: `docs/guides/language-registration.md`
 
@@ -159,7 +159,7 @@ export function render(code: string) {
 }
 ```
 
-The immutable registry can be shared across blocks and requests.
+Reuse the highlighter across blocks and requests to avoid rebuilding its name and alias maps. Registrations cannot be added through the API after construction. Keep retained language definitions unchanged and their tokenizers free of request-specific state.
 
 Source: `docs/guides/language-registration.md`
 
diff --git a/skills/theme-and-annotate-code/SKILL.md b/skills/theme-and-annotate-code/SKILL.md
index 22a8952..d0d3323 100644
--- a/skills/theme-and-annotate-code/SKILL.md
+++ b/skills/theme-and-annotate-code/SKILL.md
@@ -98,7 +98,7 @@ export const result = highlighter.highlight(
 
 Line coordinates are one-based and inclusive; decorations activate `th-line` wrappers even without line numbers.
 
-### Attach exact source diagnostics
+### Render application-supplied diagnostics
 
 ```ts
 import { highlighter } from './highlight'
@@ -121,7 +121,7 @@ export const result = highlighter.highlight(code, {
 })
 ```
 
-Character ranges use zero-based, end-exclusive UTF-16 offsets.
+Character ranges use zero-based, end-exclusive UTF-16 offsets. The application supplies the range and diagnostic message; Highlight does not perform diagnostic analysis.
 
 ### Own font styles and annotation presentation in CSS