diff --git a/docs/content/5.reference/1.parse.md b/docs/content/5.reference/1.parse.md
index c1b4bb6e..2f96eb7c 100644
--- a/docs/content/5.reference/1.parse.md
+++ b/docs/content/5.reference/1.parse.md
@@ -364,7 +364,7 @@ Both `parseMarkdown()` and `createMarkdownParser()` accept the same `ParserOptio
| `unwrap` | `boolean \| string \| string[]` | `false` | Remove wrapper tags from the tree, hoisting their children (MDC `unwrap` behaviour). `true` unwraps `p`; a comma/whitespace-separated string or array unwraps the listed tags; `'*'` matches any tag. Tags apply sequentially (each descends one level), and adjacent text is merged into a single string. |
| `html` | `boolean` | `true` | **Deprecated** (warns). Prefer `registerDefaultPlugins: false` and register `html()` explicitly. `html: false` still skips the default html plugin. |
| `linkify` | `boolean` | `true` | Auto-convert URL-like text into links. Set `false` to disable |
-| `headingIds` | `boolean` | `true` | Auto-generate `id` attributes for `h1`–`h6` headings. Set `false` to disable |
+| `headingIds` | `boolean` | `true` | Auto-generate `id` attributes for `h1`–`h6` headings. Unicode letters, marks, and numbers are kept (`## Café` → `café`); punctuation and symbols are dropped, and a heading that slugifies to nothing gets no id. Set `false` to disable. |
| `registerDefaultPlugins` | `boolean` | `true` | Register the built-in default plugins (`frontmatter`, `html`, `alert`, `task-list`, `components`, `attributes`). Set `false` to disable them. |
| `plugins` | `ComarkPlugin[]` | `[]` | Ordered plugins to run after the defaults. A same-name plugin replaces its default; duplicate explicit names keep the first instance. See [Default plugins](/plugins#default-plugins). |
| `tracer` | `ComarkTracer` | `undefined` | Timing recorder for the parse pipeline — see [Timing the parse](#timing-the-parse) |
diff --git a/docs/content/5.reference/3.reference.md b/docs/content/5.reference/3.reference.md
index 7132dcae..8002f23b 100644
--- a/docs/content/5.reference/3.reference.md
+++ b/docs/content/5.reference/3.reference.md
@@ -301,7 +301,7 @@ interface ParserOptions {
/** @deprecated Prefer registerDefaultPlugins: false */
html?: boolean // default: true
linkify?: boolean // default: true
- headingIds?: boolean // default: true
+ headingIds?: boolean // default: true; keeps Unicode letters/numbers, omits symbol-only ids
registerDefaultPlugins?: boolean // default: true
plugins?: ComarkPlugin[]
tracer?: ComarkTracer // OpenTelemetry-style tracer for parse spans
diff --git a/docs/skills/comark/references/markdown-syntax.md b/docs/skills/comark/references/markdown-syntax.md
index 6bbcf57e..c834be62 100644
--- a/docs/skills/comark/references/markdown-syntax.md
+++ b/docs/skills/comark/references/markdown-syntax.md
@@ -29,7 +29,7 @@ Comark supports all standard CommonMark and GitHub Flavored Markdown (GFM) featu
###### Heading 6
```
-**Note:** All headings automatically get ID attributes generated from their content for linking (e.g., `# Hello World` becomes `
`). Set `headingIds: false` in parse options to disable auto-generated ids.
+**Note:** All headings automatically get ID attributes generated from their content for linking (e.g., `# Hello World` becomes ``). Unicode letters, marks, and numbers are kept (`## Café` → `café`); punctuation and symbols are stripped, and a heading that slugifies to nothing gets no id. Set `headingIds: false` in parse options to disable auto-generated ids.
### Text Formatting
diff --git a/packages/comark/SPEC/common-mark/headings-id-unicode.md b/packages/comark/SPEC/common-mark/headings-id-unicode.md
new file mode 100644
index 00000000..0d50e4ff
--- /dev/null
+++ b/packages/comark/SPEC/common-mark/headings-id-unicode.md
@@ -0,0 +1,69 @@
+## Input
+
+```md
+## Café
+
+## Привет мир
+
+## 日本語の見出し
+
+## 🚀
+```
+
+## AST
+
+```json
+{
+ "frontmatter": {},
+ "meta": {},
+ "nodes": [
+ [
+ "h2",
+ {
+ "id": "café"
+ },
+ "Café"
+ ],
+ [
+ "h2",
+ {
+ "id": "привет-мир"
+ },
+ "Привет мир"
+ ],
+ [
+ "h2",
+ {
+ "id": "日本語の見出し"
+ },
+ "日本語の見出し"
+ ],
+ [
+ "h2",
+ {},
+ "🚀"
+ ]
+ ]
+}
+```
+
+## HTML
+
+```html
+Café
+Привет мир
+日本語の見出し
+🚀
+```
+
+## Markdown
+
+```md
+## Café
+
+## Привет мир
+
+## 日本語の見出し
+
+## 🚀
+```
diff --git a/packages/comark/src/internal/parse/token-processor.ts b/packages/comark/src/internal/parse/token-processor.ts
index e8c3e001..0c7790a7 100644
--- a/packages/comark/src/internal/parse/token-processor.ts
+++ b/packages/comark/src/internal/parse/token-processor.ts
@@ -415,8 +415,14 @@ function processBlockToken(
if (state?.headingIds) {
const text = children.nodes.map((n) => textContent(n)).join('')
const headingId = uniqueSlug(slugify(text), level, state)
- // Merge user-supplied attrs with the auto-generated id; user `id` wins.
- attrs = { id: headingId, ...userAttrs }
+ // An empty slug is recorded but never a usable anchor, and each further duplicate
+ // of it comes back as "-1", "-2", … — none of which is either.
+ if (headingId && !/^-\d+$/.test(headingId)) {
+ // Merge user-supplied attrs with the auto-generated id; user `id` wins.
+ attrs = { id: headingId, ...userAttrs }
+ } else {
+ attrs = userAttrs
+ }
} else {
attrs = userAttrs
}
@@ -623,20 +629,33 @@ function mergeAdjacentTextNodes(nodes: Node[]): Node[] {
}
/**
- * Convert text to a slug for heading IDs
+ * Convert text to a slug for heading IDs.
* Example: "Hello World" -> "hello-world"
* Example: "1. Introduction" -> "_1-introduction"
+ * Example: "Café" -> "café"
+ *
+ * Keeps Unicode letters, marks, decimal digits, and letter numbers. A combining mark is
+ * dropped wherever it would lead, so the result is a valid HTML5 id (NameStartChar is a
+ * letter or `_`, never a mark).
*/
function slugify(text: string): string {
let slug = text
+ .normalize('NFC')
.toLowerCase()
.trim()
.replace(/\s+/g, '-') // Replace spaces with hyphens
- .replace(/[^\w-]+/g, '') // Remove non-word chars (except hyphens)
+ // Keep Unicode letters, marks, decimal digits and letter numbers; drop the rest.
+ // Other numbers (No: ①, ½, ²) are not valid HTML5 id or CSS ident characters.
+ .replace(/[^\p{L}\p{M}\p{Nd}\p{Nl}_-]+/gu, '')
.replace(/-{2,}/g, '-') // Replace multiple hyphens with single hyphen
- .replace(/^-+|-+$/g, '') // Remove leading/trailing hyphens
-
- // Prefix with underscore if starts with a digit (HTML IDs can't start with numbers)
+ // Drop a leading run of marks and hyphens, plus trailing hyphens. Marks and
+ // hyphens only ever expose each other (`\u0301-1` would otherwise survive as
+ // "-1", which the dedup guard discards), so one alternation covers the run.
+ .replace(/^(?:\p{M}+|-+)+|-+$/gu, '')
+
+ // Prefix an ASCII leading digit. `#123` is not a valid CSS ident; a non-ASCII
+ // digit (U+0660 ARABIC-INDIC DIGIT ZERO and friends) is, so it is left as-is.
+ // `\d` stays without the `u` flag on purpose — with `u` it would match every Nd.
if (/^\d/.test(slug)) {
slug = '_' + slug
}
@@ -654,10 +673,12 @@ function uniqueSlug(slug: string, level: number, state?: ProcessState): string {
while (state.headingStack.length > 0 && state.headingStack[state.headingStack.length - 1].level >= level) {
state.headingStack.pop()
}
- // Use parent's full ID as prefix (h1 doesn't prefix children)
+ // Use parent's full ID as prefix (h1 doesn't prefix children). Skip the
+ // composition when either side is empty, so a symbol-only parent or child
+ // never yields a degenerate id like "-café" or "setup-".
if (state.headingStack.length > 0) {
const parent = state.headingStack[state.headingStack.length - 1]
- if (parent.level >= 2) {
+ if (parent.level >= 2 && parent.id && slug) {
slug = parent.id + '-' + slug
}
}
diff --git a/packages/comark/src/types.ts b/packages/comark/src/types.ts
index 50ceeae9..97c65b93 100644
--- a/packages/comark/src/types.ts
+++ b/packages/comark/src/types.ts
@@ -488,10 +488,16 @@ export interface ParserOptions[
* Whether to auto-generate `id` attributes for `h1`–`h6` headings from their text content.
* Set `false` to skip auto-generated ids; user-supplied `id` attributes are still preserved.
*
+ * Generated ids keep Unicode letters, marks, decimal digits, and letter numbers
+ * (`## Café` → `café`, `## 日本語` → `日本語`). Punctuation and symbols are dropped,
+ * and a heading that slugifies to nothing (emoji, punctuation only) gets no id.
+ * A leading ASCII digit is prefixed with `_`.
+ *
* @default true
* @example
* // With headingIds: true (default)
* // # Hello World → ['h1', { id: 'hello-world' }, 'Hello World']
+ * // # Café → ['h1', { id: 'café' }, 'Café']
*
* // With headingIds: false
* // # Hello World → ['h1', {}, 'Hello World']
diff --git a/packages/comark/test/heading-ids.test.ts b/packages/comark/test/heading-ids.test.ts
index b4730bab..964b9f6c 100644
--- a/packages/comark/test/heading-ids.test.ts
+++ b/packages/comark/test/heading-ids.test.ts
@@ -61,4 +61,133 @@ describe('headingIds option', () => {
expect((tree.nodes[0] as any)[1].id).toBe('star-here')
})
})
+
+ describe('non-ASCII headings', () => {
+ it('keeps accented Latin letters', async () => {
+ const tree = await parseMarkdown('## Café')
+
+ expect((tree.nodes[0] as any)[1].id).toBe('café')
+ })
+
+ it('keeps accented Latin letters across words', async () => {
+ const tree = await parseMarkdown('## Ünïcödé Tëxt')
+
+ expect((tree.nodes[0] as any)[1].id).toBe('ünïcödé-tëxt')
+ })
+
+ it('keeps Cyrillic letters', async () => {
+ const tree = await parseMarkdown('## Привет мир')
+
+ expect((tree.nodes[0] as any)[1].id).toBe('привет-мир')
+ })
+
+ it('keeps CJK characters', async () => {
+ const tree = await parseMarkdown('## 日本語')
+
+ expect((tree.nodes[0] as any)[1].id).toBe('日本語')
+ })
+
+ it('omits the id for a symbol-only heading', async () => {
+ const tree = await parseMarkdown('## 🚀')
+
+ expect((tree.nodes[0] as any)[1].id).toBeUndefined()
+ })
+
+ it('omits the id for every duplicate symbol-only heading', async () => {
+ const tree = await parseMarkdown('## 🚀\n\n## 🚀\n\n## 🚀')
+
+ // The third slugifies to "-2", not just "-1" — none of them is an anchor.
+ const ids = tree.nodes.map((n: any) => n[1].id)
+ expect(ids).toEqual([undefined, undefined, undefined])
+ })
+
+ it('keeps the parent prefix for a non-ASCII child heading', async () => {
+ const tree = await parseMarkdown('## Setup\n\n### Café')
+
+ const ids = tree.nodes.map((n: any) => n[1].id)
+ expect(ids).toEqual(['setup', 'setup-café'])
+ })
+
+ it('does not prefix a child with an empty parent id', async () => {
+ const tree = await parseMarkdown('## 🚀\n\n### Café')
+
+ const ids = tree.nodes.map((n: any) => n[1].id)
+ expect(ids).toEqual([undefined, 'café'])
+ })
+
+ it('omits the id for a symbol-only child of a prefixed parent', async () => {
+ const tree = await parseMarkdown('## Setup\n\n### 🚀')
+
+ const ids = tree.nodes.map((n: any) => n[1].id)
+ expect(ids).toEqual(['setup', undefined])
+ })
+
+ it('omits the id for nested symbol-only headings', async () => {
+ const tree = await parseMarkdown('## 🚀\n\n### 🚀')
+
+ const ids = tree.nodes.map((n: any) => n[1].id)
+ expect(ids).toEqual([undefined, undefined])
+ })
+
+ it('keeps the underscore prefix for a leading digit', async () => {
+ const tree = await parseMarkdown('## 2024 résumé')
+
+ expect((tree.nodes[0] as any)[1].id).toBe('_2024-résumé')
+ })
+
+ it('does not prefix a leading non-ASCII digit', async () => {
+ // U+0661 is a CSS ident start (non-ASCII); only an ASCII digit needs `_`.
+ const tree = await parseMarkdown('## ١٢٣ المقدمة')
+
+ expect((tree.nodes[0] as any)[1].id).toBe('١٢٣-المقدمة')
+ })
+
+ it('composes a decomposed accent into the same id as the precomposed letter', async () => {
+ const precomposed = await parseMarkdown('## Caf\u00e9')
+ const decomposed = await parseMarkdown('## Cafe\u0301')
+
+ expect((decomposed.nodes[0] as any)[1].id).toBe('caf\u00e9')
+ expect((decomposed.nodes[0] as any)[1].id).toBe((precomposed.nodes[0] as any)[1].id)
+ })
+
+ it('drops a leading combining mark so the id stays a valid HTML5 name', async () => {
+ const tree = await parseMarkdown('## \u0301accent')
+
+ expect((tree.nodes[0] as any)[1].id).toBe('accent')
+ })
+
+ it('drops a combining mark that only becomes leading once hyphens are stripped', async () => {
+ const tree = await parseMarkdown('## -\u0301accent')
+
+ expect((tree.nodes[0] as any)[1].id).toBe('accent')
+ })
+
+ it('still prefixes a digit when a leading mark hides the hyphen in front of it', async () => {
+ // `\u0301-1` strips to `-1` if the mark goes first, which the dedup guard discards.
+ const tree = await parseMarkdown('## \u0301-1')
+
+ expect((tree.nodes[0] as any)[1].id).toBe('_1')
+ })
+
+ it('drops a trailing hyphen', async () => {
+ const tree = await parseMarkdown('## Setup -')
+
+ expect((tree.nodes[0] as any)[1].id).toBe('setup')
+ })
+
+ it('slugifies a heading whose text is "-1" to a real id, not the empty-slug suffix', async () => {
+ // Leading hyphens are stripped before the digit prefix, so this is `_1`,
+ // not the `-1` artifact that only an empty slug's dedup counter produces.
+ const tree = await parseMarkdown('## -1')
+
+ expect((tree.nodes[0] as any)[1].id).toBe('_1')
+ })
+
+ it('does not let symbol-only headings leak a dedup suffix onto the next real heading', async () => {
+ const tree = await parseMarkdown('## 🚀\n\n## ✨\n\n## Café')
+
+ const ids = tree.nodes.map((n: any) => n[1].id)
+ expect(ids).toEqual([undefined, undefined, 'café'])
+ })
+ })
})