From 32969e2f397a799996284660c7d99ec58face723 Mon Sep 17 00:00:00 2001 From: Adam DeHaven Date: Mon, 28 Sep 2026 21:53:25 -0400 Subject: [PATCH 1/5] fix(parse): keep non-ASCII letters in heading ids Keep Unicode letters, marks, and numbers when generating heading ids, so headings in any language get a usable anchor. Skip the id when the composed slug is empty or a bare deduplication suffix, which only happens for symbol-only headings. Fixes #464 --- .../SPEC/common-mark/headings-id-unicode.md | 69 +++++++++++++++++++ .../src/internal/parse/token-processor.ts | 12 +++- packages/comark/test/heading-ids.test.ts | 52 ++++++++++++++ 3 files changed, 130 insertions(+), 3 deletions(-) create mode 100644 packages/comark/SPEC/common-mark/headings-id-unicode.md diff --git a/packages/comark/SPEC/common-mark/headings-id-unicode.md b/packages/comark/SPEC/common-mark/headings-id-unicode.md new file mode 100644 index 00000000..0d50e4ff --- /dev/null +++ b/packages/comark/SPEC/common-mark/headings-id-unicode.md @@ -0,0 +1,69 @@ +## Input + +```md +## Café + +## Привет мир + +## 日本語の見出し + +## 🚀 +``` + +## AST + +```json +{ + "frontmatter": {}, + "meta": {}, + "nodes": [ + [ + "h2", + { + "id": "café" + }, + "Café" + ], + [ + "h2", + { + "id": "привет-мир" + }, + "Привет мир" + ], + [ + "h2", + { + "id": "日本語の見出し" + }, + "日本語の見出し" + ], + [ + "h2", + {}, + "🚀" + ] + ] +} +``` + +## HTML + +```html +

Café

+

Привет мир

+

日本語の見出し

+

🚀

+``` + +## Markdown + +```md +## Café + +## Привет мир + +## 日本語の見出し + +## 🚀 +``` diff --git a/packages/comark/src/internal/parse/token-processor.ts b/packages/comark/src/internal/parse/token-processor.ts index e8c3e001..66435d25 100644 --- a/packages/comark/src/internal/parse/token-processor.ts +++ b/packages/comark/src/internal/parse/token-processor.ts @@ -415,8 +415,14 @@ function processBlockToken( if (state?.headingIds) { const text = children.nodes.map((n) => textContent(n)).join('') const headingId = uniqueSlug(slugify(text), level, state) - // Merge user-supplied attrs with the auto-generated id; user `id` wins. - attrs = { id: headingId, ...userAttrs } + // An empty id, or a bare dedup suffix like "-1", is not a usable anchor. The + // stack push and counter in uniqueSlug still run, so dedup stays intact. + if (headingId && !/^-\d/.test(headingId)) { + // Merge user-supplied attrs with the auto-generated id; user `id` wins. + attrs = { id: headingId, ...userAttrs } + } else { + attrs = userAttrs + } } else { attrs = userAttrs } @@ -632,7 +638,7 @@ function slugify(text: string): string { .toLowerCase() .trim() .replace(/\s+/g, '-') // Replace spaces with hyphens - .replace(/[^\w-]+/g, '') // Remove non-word chars (except hyphens) + .replace(/[^\p{L}\p{M}\p{Nd}\p{Nl}_-]+/gu, '') // Keep Unicode letters, marks and numbers; drop the rest .replace(/-{2,}/g, '-') // Replace multiple hyphens with single hyphen .replace(/^-+|-+$/g, '') // Remove leading/trailing hyphens diff --git a/packages/comark/test/heading-ids.test.ts b/packages/comark/test/heading-ids.test.ts index b4730bab..1a610ef9 100644 --- a/packages/comark/test/heading-ids.test.ts +++ b/packages/comark/test/heading-ids.test.ts @@ -61,4 +61,56 @@ describe('headingIds option', () => { expect((tree.nodes[0] as any)[1].id).toBe('star-here') }) }) + + describe('non-ASCII headings', () => { + it('keeps accented Latin letters', async () => { + const tree = await parseMarkdown('## Café') + + expect((tree.nodes[0] as any)[1].id).toBe('café') + }) + + it('keeps accented Latin letters across words', async () => { + const tree = await parseMarkdown('## Ünïcödé Tëxt') + + expect((tree.nodes[0] as any)[1].id).toBe('ünïcödé-tëxt') + }) + + it('keeps Cyrillic letters', async () => { + const tree = await parseMarkdown('## Привет мир') + + expect((tree.nodes[0] as any)[1].id).toBe('привет-мир') + }) + + it('keeps CJK characters', async () => { + const tree = await parseMarkdown('## 日本語') + + expect((tree.nodes[0] as any)[1].id).toBe('日本語') + }) + + it('omits the id for a symbol-only heading', async () => { + const tree = await parseMarkdown('## 🚀') + + expect((tree.nodes[0] as any)[1].id).toBeUndefined() + }) + + it('omits the id for every duplicate symbol-only heading', async () => { + const tree = await parseMarkdown('## 🚀\n\n## 🚀') + + const ids = tree.nodes.map((n: any) => n[1].id) + expect(ids).toEqual([undefined, undefined]) + }) + + it('keeps the parent prefix for a non-ASCII child heading', async () => { + const tree = await parseMarkdown('## Setup\n\n### Café') + + const ids = tree.nodes.map((n: any) => n[1].id) + expect(ids).toEqual(['setup', 'setup-café']) + }) + + it('keeps the underscore prefix for a leading digit', async () => { + const tree = await parseMarkdown('## 2024 résumé') + + expect((tree.nodes[0] as any)[1].id).toBe('_2024-résumé') + }) + }) }) From e09a6b50cc757a24da0d68f8481548d44913fdde Mon Sep 17 00:00:00 2001 From: Adam DeHaven Date: Tue, 29 Sep 2026 09:09:32 -0400 Subject: [PATCH 2/5] fix: do not use empty parent ID as heading prefix --- .../src/internal/parse/token-processor.ts | 6 ++++-- packages/comark/test/heading-ids.test.ts | 21 +++++++++++++++++++ 2 files changed, 25 insertions(+), 2 deletions(-) diff --git a/packages/comark/src/internal/parse/token-processor.ts b/packages/comark/src/internal/parse/token-processor.ts index 66435d25..517924a0 100644 --- a/packages/comark/src/internal/parse/token-processor.ts +++ b/packages/comark/src/internal/parse/token-processor.ts @@ -660,10 +660,12 @@ function uniqueSlug(slug: string, level: number, state?: ProcessState): string { while (state.headingStack.length > 0 && state.headingStack[state.headingStack.length - 1].level >= level) { state.headingStack.pop() } - // Use parent's full ID as prefix (h1 doesn't prefix children) + // Use parent's full ID as prefix (h1 doesn't prefix children). Skip the + // composition when either side is empty, so a symbol-only parent or child + // never yields a degenerate id like "-café" or "setup-". if (state.headingStack.length > 0) { const parent = state.headingStack[state.headingStack.length - 1] - if (parent.level >= 2) { + if (parent.level >= 2 && parent.id && slug) { slug = parent.id + '-' + slug } } diff --git a/packages/comark/test/heading-ids.test.ts b/packages/comark/test/heading-ids.test.ts index 1a610ef9..c9fffde0 100644 --- a/packages/comark/test/heading-ids.test.ts +++ b/packages/comark/test/heading-ids.test.ts @@ -107,6 +107,27 @@ describe('headingIds option', () => { expect(ids).toEqual(['setup', 'setup-café']) }) + it('does not prefix a child with an empty parent id', async () => { + const tree = await parseMarkdown('## 🚀\n\n### Café') + + const ids = tree.nodes.map((n: any) => n[1].id) + expect(ids).toEqual([undefined, 'café']) + }) + + it('omits the id for a symbol-only child of a prefixed parent', async () => { + const tree = await parseMarkdown('## Setup\n\n### 🚀') + + const ids = tree.nodes.map((n: any) => n[1].id) + expect(ids).toEqual(['setup', undefined]) + }) + + it('omits the id for nested symbol-only headings', async () => { + const tree = await parseMarkdown('## 🚀\n\n### 🚀') + + const ids = tree.nodes.map((n: any) => n[1].id) + expect(ids).toEqual([undefined, undefined]) + }) + it('keeps the underscore prefix for a leading digit', async () => { const tree = await parseMarkdown('## 2024 résumé') From fd910c9bd5f6ab1bcf231fd2ad9e7c8d86d35839 Mon Sep 17 00:00:00 2001 From: Farnabaz Date: Fri, 2 Oct 2026 15:07:59 +0200 Subject: [PATCH 3/5] fix(parse): drop marks and dedup suffixes from heading ids --- docs/content/5.reference/1.parse.md | 2 +- docs/content/5.reference/3.reference.md | 2 +- .../comark/references/markdown-syntax.md | 2 +- .../src/internal/parse/token-processor.ts | 23 ++++++--- packages/comark/src/types.ts | 6 +++ packages/comark/test/heading-ids.test.ts | 47 ++++++++++++++++++- 6 files changed, 71 insertions(+), 11 deletions(-) diff --git a/docs/content/5.reference/1.parse.md b/docs/content/5.reference/1.parse.md index c1b4bb6e..2f96eb7c 100644 --- a/docs/content/5.reference/1.parse.md +++ b/docs/content/5.reference/1.parse.md @@ -364,7 +364,7 @@ Both `parseMarkdown()` and `createMarkdownParser()` accept the same `ParserOptio | `unwrap` | `boolean \| string \| string[]` | `false` | Remove wrapper tags from the tree, hoisting their children (MDC `unwrap` behaviour). `true` unwraps `p`; a comma/whitespace-separated string or array unwraps the listed tags; `'*'` matches any tag. Tags apply sequentially (each descends one level), and adjacent text is merged into a single string. | | `html` | `boolean` | `true` | **Deprecated** (warns). Prefer `registerDefaultPlugins: false` and register `html()` explicitly. `html: false` still skips the default html plugin. | | `linkify` | `boolean` | `true` | Auto-convert URL-like text into links. Set `false` to disable | -| `headingIds` | `boolean` | `true` | Auto-generate `id` attributes for `h1`–`h6` headings. Set `false` to disable | +| `headingIds` | `boolean` | `true` | Auto-generate `id` attributes for `h1`–`h6` headings. Unicode letters, marks, and numbers are kept (`## Café` → `café`); punctuation and symbols are dropped, and a heading that slugifies to nothing gets no id. Set `false` to disable. | | `registerDefaultPlugins` | `boolean` | `true` | Register the built-in default plugins (`frontmatter`, `html`, `alert`, `task-list`, `components`, `attributes`). Set `false` to disable them. | | `plugins` | `ComarkPlugin[]` | `[]` | Ordered plugins to run after the defaults. A same-name plugin replaces its default; duplicate explicit names keep the first instance. See [Default plugins](/plugins#default-plugins). | | `tracer` | `ComarkTracer` | `undefined` | Timing recorder for the parse pipeline — see [Timing the parse](#timing-the-parse) | diff --git a/docs/content/5.reference/3.reference.md b/docs/content/5.reference/3.reference.md index 7132dcae..8002f23b 100644 --- a/docs/content/5.reference/3.reference.md +++ b/docs/content/5.reference/3.reference.md @@ -301,7 +301,7 @@ interface ParserOptions { /** @deprecated Prefer registerDefaultPlugins: false */ html?: boolean // default: true linkify?: boolean // default: true - headingIds?: boolean // default: true + headingIds?: boolean // default: true; keeps Unicode letters/numbers, omits symbol-only ids registerDefaultPlugins?: boolean // default: true plugins?: ComarkPlugin[] tracer?: ComarkTracer // OpenTelemetry-style tracer for parse spans diff --git a/docs/skills/comark/references/markdown-syntax.md b/docs/skills/comark/references/markdown-syntax.md index 6bbcf57e..c834be62 100644 --- a/docs/skills/comark/references/markdown-syntax.md +++ b/docs/skills/comark/references/markdown-syntax.md @@ -29,7 +29,7 @@ Comark supports all standard CommonMark and GitHub Flavored Markdown (GFM) featu ###### Heading 6 ``` -**Note:** All headings automatically get ID attributes generated from their content for linking (e.g., `# Hello World` becomes `

`). Set `headingIds: false` in parse options to disable auto-generated ids. +**Note:** All headings automatically get ID attributes generated from their content for linking (e.g., `# Hello World` becomes `

`). Unicode letters, marks, and numbers are kept (`## Café` → `café`); punctuation and symbols are stripped, and a heading that slugifies to nothing gets no id. Set `headingIds: false` in parse options to disable auto-generated ids. ### Text Formatting diff --git a/packages/comark/src/internal/parse/token-processor.ts b/packages/comark/src/internal/parse/token-processor.ts index 517924a0..9b2549fe 100644 --- a/packages/comark/src/internal/parse/token-processor.ts +++ b/packages/comark/src/internal/parse/token-processor.ts @@ -415,9 +415,9 @@ function processBlockToken( if (state?.headingIds) { const text = children.nodes.map((n) => textContent(n)).join('') const headingId = uniqueSlug(slugify(text), level, state) - // An empty id, or a bare dedup suffix like "-1", is not a usable anchor. The - // stack push and counter in uniqueSlug still run, so dedup stays intact. - if (headingId && !/^-\d/.test(headingId)) { + // An empty slug is recorded but never a usable anchor, and each further duplicate + // of it comes back as "-1", "-2", … — none of which is either. + if (headingId && !/^-\d+$/.test(headingId)) { // Merge user-supplied attrs with the auto-generated id; user `id` wins. attrs = { id: headingId, ...userAttrs } } else { @@ -629,20 +629,31 @@ function mergeAdjacentTextNodes(nodes: Node[]): Node[] { } /** - * Convert text to a slug for heading IDs + * Convert text to a slug for heading IDs. * Example: "Hello World" -> "hello-world" * Example: "1. Introduction" -> "_1-introduction" + * Example: "Café" -> "café" + * + * Keeps Unicode letters, marks, and numbers. A leading combining mark is dropped so the + * result is a valid HTML5 id (NameStartChar is a letter or `_`, never a mark). */ function slugify(text: string): string { let slug = text + .normalize('NFC') .toLowerCase() .trim() .replace(/\s+/g, '-') // Replace spaces with hyphens - .replace(/[^\p{L}\p{M}\p{Nd}\p{Nl}_-]+/gu, '') // Keep Unicode letters, marks and numbers; drop the rest + // Keep Unicode letters, marks and numbers; drop everything else. + .replace(/[^\p{L}\p{M}\p{Nd}\p{Nl}_-]+/gu, '') .replace(/-{2,}/g, '-') // Replace multiple hyphens with single hyphen .replace(/^-+|-+$/g, '') // Remove leading/trailing hyphens + // A mark cannot start an HTML5 id, and one may only become leading once the + // hyphens in front of it are gone (`-\u0301cafe`). + .replace(/^\p{M}+/u, '') - // Prefix with underscore if starts with a digit (HTML IDs can't start with numbers) + // Prefix an ASCII leading digit. `#123` is not a valid CSS ident; a non-ASCII + // digit (U+0660 ARABIC-INDIC DIGIT ZERO and friends) is, so it is left as-is. + // `\d` stays without the `u` flag on purpose — with `u` it would match every Nd. if (/^\d/.test(slug)) { slug = '_' + slug } diff --git a/packages/comark/src/types.ts b/packages/comark/src/types.ts index 50ceeae9..d779289b 100644 --- a/packages/comark/src/types.ts +++ b/packages/comark/src/types.ts @@ -488,10 +488,16 @@ export interface ParserOptions[ * Whether to auto-generate `id` attributes for `h1`–`h6` headings from their text content. * Set `false` to skip auto-generated ids; user-supplied `id` attributes are still preserved. * + * Generated ids keep Unicode letters, marks, and numbers (`## Café` → `café`, + * `## 日本語` → `日本語`). Punctuation and symbols are dropped, and a heading that + * slugifies to nothing (emoji, punctuation only) gets no id. A leading digit is + * prefixed with `_`. + * * @default true * @example * // With headingIds: true (default) * // # Hello World → ['h1', { id: 'hello-world' }, 'Hello World'] + * // # Café → ['h1', { id: 'café' }, 'Café'] * * // With headingIds: false * // # Hello World → ['h1', {}, 'Hello World'] diff --git a/packages/comark/test/heading-ids.test.ts b/packages/comark/test/heading-ids.test.ts index c9fffde0..45961d32 100644 --- a/packages/comark/test/heading-ids.test.ts +++ b/packages/comark/test/heading-ids.test.ts @@ -94,10 +94,11 @@ describe('headingIds option', () => { }) it('omits the id for every duplicate symbol-only heading', async () => { - const tree = await parseMarkdown('## 🚀\n\n## 🚀') + const tree = await parseMarkdown('## 🚀\n\n## 🚀\n\n## 🚀') + // The third slugifies to "-2", not just "-1" — none of them is an anchor. const ids = tree.nodes.map((n: any) => n[1].id) - expect(ids).toEqual([undefined, undefined]) + expect(ids).toEqual([undefined, undefined, undefined]) }) it('keeps the parent prefix for a non-ASCII child heading', async () => { @@ -133,5 +134,47 @@ describe('headingIds option', () => { expect((tree.nodes[0] as any)[1].id).toBe('_2024-résumé') }) + + it('does not prefix a leading non-ASCII digit', async () => { + // U+0661 is a CSS ident start (non-ASCII); only an ASCII digit needs `_`. + const tree = await parseMarkdown('## ١٢٣ المقدمة') + + expect((tree.nodes[0] as any)[1].id).toBe('١٢٣-المقدمة') + }) + + it('composes a decomposed accent into the same id as the precomposed letter', async () => { + const precomposed = await parseMarkdown('## Caf\u00e9') + const decomposed = await parseMarkdown('## Cafe\u0301') + + expect((decomposed.nodes[0] as any)[1].id).toBe('caf\u00e9') + expect((decomposed.nodes[0] as any)[1].id).toBe((precomposed.nodes[0] as any)[1].id) + }) + + it('drops a leading combining mark so the id stays a valid HTML5 name', async () => { + const tree = await parseMarkdown('## \u0301accent') + + expect((tree.nodes[0] as any)[1].id).toBe('accent') + }) + + it('drops a combining mark that only becomes leading once hyphens are stripped', async () => { + const tree = await parseMarkdown('## -\u0301accent') + + expect((tree.nodes[0] as any)[1].id).toBe('accent') + }) + + it('slugifies a heading whose text is "-1" to a real id, not the empty-slug suffix', async () => { + // Leading hyphens are stripped before the digit prefix, so this is `_1`, + // not the `-1` artifact that only an empty slug's dedup counter produces. + const tree = await parseMarkdown('## -1') + + expect((tree.nodes[0] as any)[1].id).toBe('_1') + }) + + it('does not let symbol-only headings leak a dedup suffix onto the next real heading', async () => { + const tree = await parseMarkdown('## 🚀\n\n## ✨\n\n## Café') + + const ids = tree.nodes.map((n: any) => n[1].id) + expect(ids).toEqual([undefined, undefined, 'café']) + }) }) }) From 87afca6e12f1613903308eab3737d7b42db014fd Mon Sep 17 00:00:00 2001 From: Farnabaz Date: Fri, 2 Oct 2026 16:07:25 +0200 Subject: [PATCH 4/5] fix(parse): strip leading marks and hyphens from heading ids in one pass --- .../comark/src/internal/parse/token-processor.ts | 16 +++++++++------- packages/comark/src/types.ts | 8 ++++---- packages/comark/test/heading-ids.test.ts | 7 +++++++ 3 files changed, 20 insertions(+), 11 deletions(-) diff --git a/packages/comark/src/internal/parse/token-processor.ts b/packages/comark/src/internal/parse/token-processor.ts index 9b2549fe..b5881e28 100644 --- a/packages/comark/src/internal/parse/token-processor.ts +++ b/packages/comark/src/internal/parse/token-processor.ts @@ -634,8 +634,9 @@ function mergeAdjacentTextNodes(nodes: Node[]): Node[] { * Example: "1. Introduction" -> "_1-introduction" * Example: "Café" -> "café" * - * Keeps Unicode letters, marks, and numbers. A leading combining mark is dropped so the - * result is a valid HTML5 id (NameStartChar is a letter or `_`, never a mark). + * Keeps Unicode letters, marks, decimal digits, and letter numbers. A combining mark is + * dropped wherever it would lead, so the result is a valid HTML5 id (NameStartChar is a + * letter or `_`, never a mark). */ function slugify(text: string): string { let slug = text @@ -643,13 +644,14 @@ function slugify(text: string): string { .toLowerCase() .trim() .replace(/\s+/g, '-') // Replace spaces with hyphens - // Keep Unicode letters, marks and numbers; drop everything else. + // Keep Unicode letters, marks, decimal digits and letter numbers; drop the rest. + // Other numbers (No: ①, ½, ²) are not valid HTML5 id or CSS ident characters. .replace(/[^\p{L}\p{M}\p{Nd}\p{Nl}_-]+/gu, '') .replace(/-{2,}/g, '-') // Replace multiple hyphens with single hyphen - .replace(/^-+|-+$/g, '') // Remove leading/trailing hyphens - // A mark cannot start an HTML5 id, and one may only become leading once the - // hyphens in front of it are gone (`-\u0301cafe`). - .replace(/^\p{M}+/u, '') + // Drop a leading run of marks and hyphens. They only ever expose each other + // (`\u0301-1` would otherwise survive as "-1", which the dedup guard discards), + // so one alternation covers the whole run. + .replace(/^(?:\p{M}+|-+)+/u, '') // Prefix an ASCII leading digit. `#123` is not a valid CSS ident; a non-ASCII // digit (U+0660 ARABIC-INDIC DIGIT ZERO and friends) is, so it is left as-is. diff --git a/packages/comark/src/types.ts b/packages/comark/src/types.ts index d779289b..97c65b93 100644 --- a/packages/comark/src/types.ts +++ b/packages/comark/src/types.ts @@ -488,10 +488,10 @@ export interface ParserOptions[ * Whether to auto-generate `id` attributes for `h1`–`h6` headings from their text content. * Set `false` to skip auto-generated ids; user-supplied `id` attributes are still preserved. * - * Generated ids keep Unicode letters, marks, and numbers (`## Café` → `café`, - * `## 日本語` → `日本語`). Punctuation and symbols are dropped, and a heading that - * slugifies to nothing (emoji, punctuation only) gets no id. A leading digit is - * prefixed with `_`. + * Generated ids keep Unicode letters, marks, decimal digits, and letter numbers + * (`## Café` → `café`, `## 日本語` → `日本語`). Punctuation and symbols are dropped, + * and a heading that slugifies to nothing (emoji, punctuation only) gets no id. + * A leading ASCII digit is prefixed with `_`. * * @default true * @example diff --git a/packages/comark/test/heading-ids.test.ts b/packages/comark/test/heading-ids.test.ts index 45961d32..a53e86fd 100644 --- a/packages/comark/test/heading-ids.test.ts +++ b/packages/comark/test/heading-ids.test.ts @@ -162,6 +162,13 @@ describe('headingIds option', () => { expect((tree.nodes[0] as any)[1].id).toBe('accent') }) + it('still prefixes a digit when a leading mark hides the hyphen in front of it', async () => { + // `\u0301-1` strips to `-1` if the mark goes first, which the dedup guard discards. + const tree = await parseMarkdown('## \u0301-1') + + expect((tree.nodes[0] as any)[1].id).toBe('_1') + }) + it('slugifies a heading whose text is "-1" to a real id, not the empty-slug suffix', async () => { // Leading hyphens are stripped before the digit prefix, so this is `_1`, // not the `-1` artifact that only an empty slug's dedup counter produces. From b1a662359678373ddc93196128b3453d6e42a2cb Mon Sep 17 00:00:00 2001 From: Farnabaz Date: Fri, 2 Oct 2026 16:22:27 +0200 Subject: [PATCH 5/5] fix(parse): trim trailing hyphens from heading ids --- packages/comark/src/internal/parse/token-processor.ts | 8 ++++---- packages/comark/test/heading-ids.test.ts | 6 ++++++ 2 files changed, 10 insertions(+), 4 deletions(-) diff --git a/packages/comark/src/internal/parse/token-processor.ts b/packages/comark/src/internal/parse/token-processor.ts index b5881e28..0c7790a7 100644 --- a/packages/comark/src/internal/parse/token-processor.ts +++ b/packages/comark/src/internal/parse/token-processor.ts @@ -648,10 +648,10 @@ function slugify(text: string): string { // Other numbers (No: ①, ½, ²) are not valid HTML5 id or CSS ident characters. .replace(/[^\p{L}\p{M}\p{Nd}\p{Nl}_-]+/gu, '') .replace(/-{2,}/g, '-') // Replace multiple hyphens with single hyphen - // Drop a leading run of marks and hyphens. They only ever expose each other - // (`\u0301-1` would otherwise survive as "-1", which the dedup guard discards), - // so one alternation covers the whole run. - .replace(/^(?:\p{M}+|-+)+/u, '') + // Drop a leading run of marks and hyphens, plus trailing hyphens. Marks and + // hyphens only ever expose each other (`\u0301-1` would otherwise survive as + // "-1", which the dedup guard discards), so one alternation covers the run. + .replace(/^(?:\p{M}+|-+)+|-+$/gu, '') // Prefix an ASCII leading digit. `#123` is not a valid CSS ident; a non-ASCII // digit (U+0660 ARABIC-INDIC DIGIT ZERO and friends) is, so it is left as-is. diff --git a/packages/comark/test/heading-ids.test.ts b/packages/comark/test/heading-ids.test.ts index a53e86fd..964b9f6c 100644 --- a/packages/comark/test/heading-ids.test.ts +++ b/packages/comark/test/heading-ids.test.ts @@ -169,6 +169,12 @@ describe('headingIds option', () => { expect((tree.nodes[0] as any)[1].id).toBe('_1') }) + it('drops a trailing hyphen', async () => { + const tree = await parseMarkdown('## Setup -') + + expect((tree.nodes[0] as any)[1].id).toBe('setup') + }) + it('slugifies a heading whose text is "-1" to a real id, not the empty-slug suffix', async () => { // Leading hyphens are stripped before the digit prefix, so this is `_1`, // not the `-1` artifact that only an empty slug's dedup counter produces.