diff --git a/docs/content/5.reference/1.parse.md b/docs/content/5.reference/1.parse.md index c1b4bb6e..2f96eb7c 100644 --- a/docs/content/5.reference/1.parse.md +++ b/docs/content/5.reference/1.parse.md @@ -364,7 +364,7 @@ Both `parseMarkdown()` and `createMarkdownParser()` accept the same `ParserOptio | `unwrap` | `boolean \| string \| string[]` | `false` | Remove wrapper tags from the tree, hoisting their children (MDC `unwrap` behaviour). `true` unwraps `p`; a comma/whitespace-separated string or array unwraps the listed tags; `'*'` matches any tag. Tags apply sequentially (each descends one level), and adjacent text is merged into a single string. | | `html` | `boolean` | `true` | **Deprecated** (warns). Prefer `registerDefaultPlugins: false` and register `html()` explicitly. `html: false` still skips the default html plugin. | | `linkify` | `boolean` | `true` | Auto-convert URL-like text into links. Set `false` to disable | -| `headingIds` | `boolean` | `true` | Auto-generate `id` attributes for `h1`–`h6` headings. Set `false` to disable | +| `headingIds` | `boolean` | `true` | Auto-generate `id` attributes for `h1`–`h6` headings. Unicode letters, marks, and numbers are kept (`## Café` → `café`); punctuation and symbols are dropped, and a heading that slugifies to nothing gets no id. Set `false` to disable. | | `registerDefaultPlugins` | `boolean` | `true` | Register the built-in default plugins (`frontmatter`, `html`, `alert`, `task-list`, `components`, `attributes`). Set `false` to disable them. | | `plugins` | `ComarkPlugin[]` | `[]` | Ordered plugins to run after the defaults. A same-name plugin replaces its default; duplicate explicit names keep the first instance. See [Default plugins](/plugins#default-plugins). | | `tracer` | `ComarkTracer` | `undefined` | Timing recorder for the parse pipeline — see [Timing the parse](#timing-the-parse) | diff --git a/docs/content/5.reference/3.reference.md b/docs/content/5.reference/3.reference.md index 7132dcae..8002f23b 100644 --- a/docs/content/5.reference/3.reference.md +++ b/docs/content/5.reference/3.reference.md @@ -301,7 +301,7 @@ interface ParserOptions { /** @deprecated Prefer registerDefaultPlugins: false */ html?: boolean // default: true linkify?: boolean // default: true - headingIds?: boolean // default: true + headingIds?: boolean // default: true; keeps Unicode letters/numbers, omits symbol-only ids registerDefaultPlugins?: boolean // default: true plugins?: ComarkPlugin[] tracer?: ComarkTracer // OpenTelemetry-style tracer for parse spans diff --git a/docs/skills/comark/references/markdown-syntax.md b/docs/skills/comark/references/markdown-syntax.md index 6bbcf57e..c834be62 100644 --- a/docs/skills/comark/references/markdown-syntax.md +++ b/docs/skills/comark/references/markdown-syntax.md @@ -29,7 +29,7 @@ Comark supports all standard CommonMark and GitHub Flavored Markdown (GFM) featu ###### Heading 6 ``` -**Note:** All headings automatically get ID attributes generated from their content for linking (e.g., `# Hello World` becomes `

`). Set `headingIds: false` in parse options to disable auto-generated ids. +**Note:** All headings automatically get ID attributes generated from their content for linking (e.g., `# Hello World` becomes `

`). Unicode letters, marks, and numbers are kept (`## Café` → `café`); punctuation and symbols are stripped, and a heading that slugifies to nothing gets no id. Set `headingIds: false` in parse options to disable auto-generated ids. ### Text Formatting diff --git a/packages/comark/SPEC/common-mark/headings-id-unicode.md b/packages/comark/SPEC/common-mark/headings-id-unicode.md new file mode 100644 index 00000000..0d50e4ff --- /dev/null +++ b/packages/comark/SPEC/common-mark/headings-id-unicode.md @@ -0,0 +1,69 @@ +## Input + +```md +## Café + +## Привет мир + +## 日本語の見出し + +## 🚀 +``` + +## AST + +```json +{ + "frontmatter": {}, + "meta": {}, + "nodes": [ + [ + "h2", + { + "id": "café" + }, + "Café" + ], + [ + "h2", + { + "id": "привет-мир" + }, + "Привет мир" + ], + [ + "h2", + { + "id": "日本語の見出し" + }, + "日本語の見出し" + ], + [ + "h2", + {}, + "🚀" + ] + ] +} +``` + +## HTML + +```html +

Café

+

Привет мир

+

日本語の見出し

+

🚀

+``` + +## Markdown + +```md +## Café + +## Привет мир + +## 日本語の見出し + +## 🚀 +``` diff --git a/packages/comark/src/internal/parse/token-processor.ts b/packages/comark/src/internal/parse/token-processor.ts index e8c3e001..0c7790a7 100644 --- a/packages/comark/src/internal/parse/token-processor.ts +++ b/packages/comark/src/internal/parse/token-processor.ts @@ -415,8 +415,14 @@ function processBlockToken( if (state?.headingIds) { const text = children.nodes.map((n) => textContent(n)).join('') const headingId = uniqueSlug(slugify(text), level, state) - // Merge user-supplied attrs with the auto-generated id; user `id` wins. - attrs = { id: headingId, ...userAttrs } + // An empty slug is recorded but never a usable anchor, and each further duplicate + // of it comes back as "-1", "-2", … — none of which is either. + if (headingId && !/^-\d+$/.test(headingId)) { + // Merge user-supplied attrs with the auto-generated id; user `id` wins. + attrs = { id: headingId, ...userAttrs } + } else { + attrs = userAttrs + } } else { attrs = userAttrs } @@ -623,20 +629,33 @@ function mergeAdjacentTextNodes(nodes: Node[]): Node[] { } /** - * Convert text to a slug for heading IDs + * Convert text to a slug for heading IDs. * Example: "Hello World" -> "hello-world" * Example: "1. Introduction" -> "_1-introduction" + * Example: "Café" -> "café" + * + * Keeps Unicode letters, marks, decimal digits, and letter numbers. A combining mark is + * dropped wherever it would lead, so the result is a valid HTML5 id (NameStartChar is a + * letter or `_`, never a mark). */ function slugify(text: string): string { let slug = text + .normalize('NFC') .toLowerCase() .trim() .replace(/\s+/g, '-') // Replace spaces with hyphens - .replace(/[^\w-]+/g, '') // Remove non-word chars (except hyphens) + // Keep Unicode letters, marks, decimal digits and letter numbers; drop the rest. + // Other numbers (No: ①, ½, ²) are not valid HTML5 id or CSS ident characters. + .replace(/[^\p{L}\p{M}\p{Nd}\p{Nl}_-]+/gu, '') .replace(/-{2,}/g, '-') // Replace multiple hyphens with single hyphen - .replace(/^-+|-+$/g, '') // Remove leading/trailing hyphens - - // Prefix with underscore if starts with a digit (HTML IDs can't start with numbers) + // Drop a leading run of marks and hyphens, plus trailing hyphens. Marks and + // hyphens only ever expose each other (`\u0301-1` would otherwise survive as + // "-1", which the dedup guard discards), so one alternation covers the run. + .replace(/^(?:\p{M}+|-+)+|-+$/gu, '') + + // Prefix an ASCII leading digit. `#123` is not a valid CSS ident; a non-ASCII + // digit (U+0660 ARABIC-INDIC DIGIT ZERO and friends) is, so it is left as-is. + // `\d` stays without the `u` flag on purpose — with `u` it would match every Nd. if (/^\d/.test(slug)) { slug = '_' + slug } @@ -654,10 +673,12 @@ function uniqueSlug(slug: string, level: number, state?: ProcessState): string { while (state.headingStack.length > 0 && state.headingStack[state.headingStack.length - 1].level >= level) { state.headingStack.pop() } - // Use parent's full ID as prefix (h1 doesn't prefix children) + // Use parent's full ID as prefix (h1 doesn't prefix children). Skip the + // composition when either side is empty, so a symbol-only parent or child + // never yields a degenerate id like "-café" or "setup-". if (state.headingStack.length > 0) { const parent = state.headingStack[state.headingStack.length - 1] - if (parent.level >= 2) { + if (parent.level >= 2 && parent.id && slug) { slug = parent.id + '-' + slug } } diff --git a/packages/comark/src/types.ts b/packages/comark/src/types.ts index 50ceeae9..97c65b93 100644 --- a/packages/comark/src/types.ts +++ b/packages/comark/src/types.ts @@ -488,10 +488,16 @@ export interface ParserOptions[ * Whether to auto-generate `id` attributes for `h1`–`h6` headings from their text content. * Set `false` to skip auto-generated ids; user-supplied `id` attributes are still preserved. * + * Generated ids keep Unicode letters, marks, decimal digits, and letter numbers + * (`## Café` → `café`, `## 日本語` → `日本語`). Punctuation and symbols are dropped, + * and a heading that slugifies to nothing (emoji, punctuation only) gets no id. + * A leading ASCII digit is prefixed with `_`. + * * @default true * @example * // With headingIds: true (default) * // # Hello World → ['h1', { id: 'hello-world' }, 'Hello World'] + * // # Café → ['h1', { id: 'café' }, 'Café'] * * // With headingIds: false * // # Hello World → ['h1', {}, 'Hello World'] diff --git a/packages/comark/test/heading-ids.test.ts b/packages/comark/test/heading-ids.test.ts index b4730bab..964b9f6c 100644 --- a/packages/comark/test/heading-ids.test.ts +++ b/packages/comark/test/heading-ids.test.ts @@ -61,4 +61,133 @@ describe('headingIds option', () => { expect((tree.nodes[0] as any)[1].id).toBe('star-here') }) }) + + describe('non-ASCII headings', () => { + it('keeps accented Latin letters', async () => { + const tree = await parseMarkdown('## Café') + + expect((tree.nodes[0] as any)[1].id).toBe('café') + }) + + it('keeps accented Latin letters across words', async () => { + const tree = await parseMarkdown('## Ünïcödé Tëxt') + + expect((tree.nodes[0] as any)[1].id).toBe('ünïcödé-tëxt') + }) + + it('keeps Cyrillic letters', async () => { + const tree = await parseMarkdown('## Привет мир') + + expect((tree.nodes[0] as any)[1].id).toBe('привет-мир') + }) + + it('keeps CJK characters', async () => { + const tree = await parseMarkdown('## 日本語') + + expect((tree.nodes[0] as any)[1].id).toBe('日本語') + }) + + it('omits the id for a symbol-only heading', async () => { + const tree = await parseMarkdown('## 🚀') + + expect((tree.nodes[0] as any)[1].id).toBeUndefined() + }) + + it('omits the id for every duplicate symbol-only heading', async () => { + const tree = await parseMarkdown('## 🚀\n\n## 🚀\n\n## 🚀') + + // The third slugifies to "-2", not just "-1" — none of them is an anchor. + const ids = tree.nodes.map((n: any) => n[1].id) + expect(ids).toEqual([undefined, undefined, undefined]) + }) + + it('keeps the parent prefix for a non-ASCII child heading', async () => { + const tree = await parseMarkdown('## Setup\n\n### Café') + + const ids = tree.nodes.map((n: any) => n[1].id) + expect(ids).toEqual(['setup', 'setup-café']) + }) + + it('does not prefix a child with an empty parent id', async () => { + const tree = await parseMarkdown('## 🚀\n\n### Café') + + const ids = tree.nodes.map((n: any) => n[1].id) + expect(ids).toEqual([undefined, 'café']) + }) + + it('omits the id for a symbol-only child of a prefixed parent', async () => { + const tree = await parseMarkdown('## Setup\n\n### 🚀') + + const ids = tree.nodes.map((n: any) => n[1].id) + expect(ids).toEqual(['setup', undefined]) + }) + + it('omits the id for nested symbol-only headings', async () => { + const tree = await parseMarkdown('## 🚀\n\n### 🚀') + + const ids = tree.nodes.map((n: any) => n[1].id) + expect(ids).toEqual([undefined, undefined]) + }) + + it('keeps the underscore prefix for a leading digit', async () => { + const tree = await parseMarkdown('## 2024 résumé') + + expect((tree.nodes[0] as any)[1].id).toBe('_2024-résumé') + }) + + it('does not prefix a leading non-ASCII digit', async () => { + // U+0661 is a CSS ident start (non-ASCII); only an ASCII digit needs `_`. + const tree = await parseMarkdown('## ١٢٣ المقدمة') + + expect((tree.nodes[0] as any)[1].id).toBe('١٢٣-المقدمة') + }) + + it('composes a decomposed accent into the same id as the precomposed letter', async () => { + const precomposed = await parseMarkdown('## Caf\u00e9') + const decomposed = await parseMarkdown('## Cafe\u0301') + + expect((decomposed.nodes[0] as any)[1].id).toBe('caf\u00e9') + expect((decomposed.nodes[0] as any)[1].id).toBe((precomposed.nodes[0] as any)[1].id) + }) + + it('drops a leading combining mark so the id stays a valid HTML5 name', async () => { + const tree = await parseMarkdown('## \u0301accent') + + expect((tree.nodes[0] as any)[1].id).toBe('accent') + }) + + it('drops a combining mark that only becomes leading once hyphens are stripped', async () => { + const tree = await parseMarkdown('## -\u0301accent') + + expect((tree.nodes[0] as any)[1].id).toBe('accent') + }) + + it('still prefixes a digit when a leading mark hides the hyphen in front of it', async () => { + // `\u0301-1` strips to `-1` if the mark goes first, which the dedup guard discards. + const tree = await parseMarkdown('## \u0301-1') + + expect((tree.nodes[0] as any)[1].id).toBe('_1') + }) + + it('drops a trailing hyphen', async () => { + const tree = await parseMarkdown('## Setup -') + + expect((tree.nodes[0] as any)[1].id).toBe('setup') + }) + + it('slugifies a heading whose text is "-1" to a real id, not the empty-slug suffix', async () => { + // Leading hyphens are stripped before the digit prefix, so this is `_1`, + // not the `-1` artifact that only an empty slug's dedup counter produces. + const tree = await parseMarkdown('## -1') + + expect((tree.nodes[0] as any)[1].id).toBe('_1') + }) + + it('does not let symbol-only headings leak a dedup suffix onto the next real heading', async () => { + const tree = await parseMarkdown('## 🚀\n\n## ✨\n\n## Café') + + const ids = tree.nodes.map((n: any) => n[1].id) + expect(ids).toEqual([undefined, undefined, 'café']) + }) + }) })