diff --git a/scripts/bench-parse.mjs b/scripts/bench-parse.mjs index d358f3a..06c1285 100644 --- a/scripts/bench-parse.mjs +++ b/scripts/bench-parse.mjs @@ -117,7 +117,6 @@ function makeState(source) { emptyCodeOffsets: new WeakMap(), urlSegments: new WeakMap(), emptyUrlOffsets: new WeakMap(), - urlSourceSpans: new WeakMap(), }; } diff --git a/scripts/bench-source-map.mjs b/scripts/bench-source-map.mjs index 2e1d062..a93ca65 100644 --- a/scripts/bench-source-map.mjs +++ b/scripts/bench-source-map.mjs @@ -1,5 +1,4 @@ -// Benchmark source-map query performance (issue #51) and URL-field mapping -// construction (issue #74). +// Benchmark source-map query performance and URL destination source mapping. // // The pathological shape is a single text node made of many alternating, // non-mergeable atomic segments (`&` character reference + `\(` escape), @@ -23,9 +22,8 @@ const SMOKE = process.argv.includes('--smoke'); const SIZES_KIB = SMOKE ? [256] : [1, 16, 64, 256]; const SMOKE_QUERY_BUDGET_MS = 1000; const URL_BUILD_SIZES_KIB = [16, 32, 64, 128, 256, 512]; -// A 4× URL grows close to 4× on the bounded scan. The old unbounded `&` -// search grows beyond 6× from 128 KiB to 512 KiB; this larger interval makes -// the quadratic term dominate while leaving headroom for CI scheduling noise. +// A 4× URL should stay close to 4× on the production parse path. +// The larger interval leaves room for CI scheduling noise. const SMOKE_URL_BUILD_RATIO_MAX = 6; /** Build an input of roughly `kib` kibibytes made of repeated UNIT. */ @@ -171,7 +169,7 @@ for (const kib of SIZES_KIB) { } } -console.log('\n=== URL field construction ==='); +console.log('\n=== URL destination source mapping ==='); const urlBuildResults = []; for (const kib of URL_BUILD_SIZES_KIB) { const md = makeUrlInput(kib); @@ -189,7 +187,7 @@ if (SMOKE) { console.log(` ${'128 → 512 KiB growth'.padEnd(28)} ${ratio.toFixed(2)}x`); if (ratio > SMOKE_URL_BUILD_RATIO_MAX) { throw new Error( - `benchmark smoke failed: URL construction grew ${ratio.toFixed(2)}x ` + `benchmark smoke failed: URL source mapping grew ${ratio.toFixed(2)}x ` + `from 128 KiB to 512 KiB (budget ${SMOKE_URL_BUILD_RATIO_MAX}x)`, ); } diff --git a/src/source-map/build-source-map.ts b/src/source-map/build-source-map.ts index fa522cb..c133418 100644 --- a/src/source-map/build-source-map.ts +++ b/src/source-map/build-source-map.ts @@ -1,6 +1,4 @@ import { fromMarkdown } from 'mdast-util-from-markdown'; -import { decodeNumericCharacterReference } from 'micromark-util-decode-numeric-character-reference'; -import { decodeNamedCharacterReference } from 'decode-named-character-reference'; import type { Root } from 'mdast'; import type { MarkdownCodeNode, @@ -24,7 +22,6 @@ import type { MarkdownSourceMapSegment, MarkdownValueSourceIndex, ParsedMarkdownDocument, - SourceSpan, } from './types'; // Use the same parser extensions as `parseMd`. @@ -49,8 +46,6 @@ interface RecordingState { urlSegments: WeakMap /** link / definition node -> source point for an empty URL. */ emptyUrlOffsets: WeakMap - /** link / definition node -> parser-confirmed destination content span. */ - urlSourceSpans: WeakMap } interface TraversableNode { @@ -68,12 +63,6 @@ function hasStringField( return field in node && typeof node[field] === 'string'; } -// micromark limits named character references to 31 code units; numeric -// references are shorter. Include the leading `&` and trailing `;` so URL -// mapping only considers parser-valid candidates and never scans an entire -// destination for every literal ampersand. -const MAX_CHARACTER_REFERENCE_SOURCE_LENGTH = 33; - /** * Offsets (UTF-16 code units) of the first code unit of every line in `md`. * Handles LF, CR, and CRLF line endings the same way micromark does. @@ -127,102 +116,6 @@ function pointAtOffset(lineStarts: number[], md: string, offset: number): Parsed }; } -function isEscapableUrlCharacter(char: number): boolean { - return (char >= 33 && char <= 47) - || (char >= 58 && char <= 64) - || (char >= 91 && char <= 96) - || (char >= 123 && char <= 126); -} - -function characterReferenceEnd( - md: string, - start: number, - end: number, -): number | undefined { - const limit = Math.min(end, start + MAX_CHARACTER_REFERENCE_SOURCE_LENGTH); - for (let offset = start + 1; offset < limit; offset++) { - if (md.charCodeAt(offset) === 59) - return offset; - } - return undefined; -} - -interface UrlSegments { - segments: MarkdownSourceMapSegment[] - emptyOffset?: number -} - -function buildUrlSegments( - md: string, - node: { url: string }, - bounds: SourceSpan, -): UrlSegments | undefined { - if (bounds.start === bounds.end) { - return node.url === '' - ? { segments: [], emptyOffset: bounds.start } - : undefined; - } - const segments: MarkdownSourceMapSegment[] = []; - let valueOffset = 0; - let valueMatch = true; - const add = (sourceStart: number, sourceEnd: number, output: string, kind: MarkdownSourceMapSegment['kind']) => { - segments.push({ - valueStart: valueOffset, - valueEnd: valueOffset + output.length, - sourceStart, - sourceEnd, - kind, - }); - if (valueMatch && !node.url.startsWith(output, valueOffset)) - valueMatch = false; - valueOffset += output.length; - }; - let literalStart = bounds.start; - const flushLiteral = (end: number): void => { - if (literalStart < end) - add(literalStart, end, md.slice(literalStart, end), 'literal'); - }; - - for (let offset = bounds.start; offset < bounds.end;) { - const char = md.charCodeAt(offset); - if (char === 92 && offset + 1 < bounds.end && isEscapableUrlCharacter(md.charCodeAt(offset + 1))) { - flushLiteral(offset); - add(offset, offset + 2, md[offset + 1], 'escape'); - offset += 2; - literalStart = offset; - continue; - } - if (char === 38) { - const semi = characterReferenceEnd(md, offset, bounds.end); - if (semi !== undefined) { - const body = md.slice(offset + 1, semi); - let decoded: string | false; - if (body.startsWith('#')) { - const numeric = body.slice(1); - const radix = numeric.startsWith('x') || numeric.startsWith('X') ? 16 : 10; - decoded = decodeNumericCharacterReference( - radix === 16 ? numeric.slice(1) : numeric, - radix, - ); - } - else { - decoded = decodeNamedCharacterReference(body); - } - if (decoded !== false) { - flushLiteral(offset); - add(offset, semi + 1, decoded, 'character-reference'); - offset = semi + 1; - literalStart = offset; - continue; - } - } - } - offset++; - } - flushLiteral(bounds.end); - return valueMatch && valueOffset === node.url.length ? { segments } : undefined; -} - /** * Find the segment covering `valueIndex`, or undefined. * @@ -445,7 +338,6 @@ export const parseMdWithSourceMap = (md: string): ParsedMarkdownDocument => { emptyCodeOffsets: new WeakMap(), urlSegments: new WeakMap(), emptyUrlOffsets: new WeakMap(), - urlSourceSpans: new WeakMap(), }; const tree = fromMarkdown(md, { @@ -460,7 +352,6 @@ export const parseMdWithSourceMap = (md: string): ParsedMarkdownDocument => { let lineStarts: number[] | undefined; // The index records ownership and parse-time state for all nodes. - // It also builds mappings that the parser extension cannot produce. const owned = new WeakSet(); const originalValues = new WeakMap(); const originalUrls = new WeakMap(); @@ -470,19 +361,6 @@ export const parseMdWithSourceMap = (md: string): ParsedMarkdownDocument => { const sourceGapPrefixes = new WeakMap(); function indexNode(node: TraversableNode): void { - if ( - (node.type === 'link' || node.type === 'definition') - && hasStringField(node, 'url') - ) { - const bounds = state.urlSourceSpans.get(node); - const segments = bounds ? buildUrlSegments(md, node, bounds) : undefined; - if (segments) { - state.urlSegments.set(node, segments.segments); - if (segments.emptyOffset !== undefined) - state.emptyUrlOffsets.set(node, segments.emptyOffset); - } - } - owned.add(node); const mappedSegments = state.segments.get(node) || state.inlineCodeSegments.get(node) diff --git a/src/source-map/recording-extension.ts b/src/source-map/recording-extension.ts index 4df25b8..45ec389 100644 --- a/src/source-map/recording-extension.ts +++ b/src/source-map/recording-extension.ts @@ -1,7 +1,7 @@ import { decodeNamedCharacterReference } from 'decode-named-character-reference'; import { decodeNumericCharacterReference } from 'micromark-util-decode-numeric-character-reference'; import type { ParsedPoint } from '../types'; -import type { MarkdownSourceMapSegment, SourceSpan } from './types'; +import type { MarkdownSourceMapSegment } from './types'; interface RecordingState { source: string @@ -9,7 +9,8 @@ interface RecordingState { inlineCodeSegments: WeakMap codeSegments: WeakMap emptyCodeOffsets: WeakMap - urlSourceSpans: WeakMap + urlSegments: WeakMap + emptyUrlOffsets: WeakMap } interface SegmentMetadata { @@ -28,6 +29,11 @@ interface CodeValueRecording { valueLength: number } +interface UrlRecording { + segments: MarkdownSourceMapSegment[] + valueLength: number +} + interface FencedCodeRecording extends CodeValueRecording { emptyOffset: number sawOpeningLineEnding: boolean @@ -138,7 +144,8 @@ interface CompileContext { /** * Build a `mdast` extension that records source mappings during compilation. * Its handlers record mappings for `text` and `inlineCode` values. - * They also record fenced `code` values. + * They also record block `code` values. + * They record link and definition URL destinations. * * ⚠️ This couples to `mdast-util-from-markdown` / micromark INTERNALS, not the * public remark API. Upgrading any parser-sensitive dependency is a parser @@ -169,6 +176,7 @@ export function recordingExtension(state: RecordingState) { let inlineCodeSegments: MarkdownSourceMapSegment[] | undefined; let fencedCodeRecording: FencedCodeRecording | undefined; let indentedCodeRecording: CodeValueRecording | undefined; + let urlRecording: UrlRecording | undefined; let pendingEmptyFencedCode: object | undefined; let lineIndentStart = 0; @@ -214,7 +222,7 @@ export function recordingExtension(state: RecordingState) { this: CompileContext, token: any, metadata: SegmentMetadata, - ) { + ): string { const tail = this.stack.pop(); const slice = this.sliceSerialize(token); const valueStart = tail.value.length; @@ -228,6 +236,33 @@ export function recordingExtension(state: RecordingState) { ...metadata, }); } + return slice; + }; + + const recordUrlSegment = ( + metadata: SegmentMetadata, + valueLength: number, + ): void => { + if (!urlRecording) + return; + const previous = urlRecording.segments[urlRecording.segments.length - 1]; + if ( + metadata.kind === 'literal' + && previous?.kind === 'literal' + && previous.sourceEnd === metadata.sourceStart + && previous.valueEnd === urlRecording.valueLength + ) { + previous.valueEnd += valueLength; + previous.sourceEnd = metadata.sourceEnd; + } + else { + urlRecording.segments.push({ + valueStart: urlRecording.valueLength, + valueEnd: urlRecording.valueLength + valueLength, + ...metadata, + }); + } + urlRecording.valueLength += valueLength; }; const onexitcharacterreferencevalue = function (this: CompileContext, token: any) { @@ -264,6 +299,7 @@ export function recordingExtension(state: RecordingState) { kind, }); } + recordUrlSegment({ ...construct, kind }, value.length); }; const inlineCodeValueLength = (): number => { @@ -571,18 +607,31 @@ export function recordingExtension(state: RecordingState) { }; const onenterUrlDestination = function (this: CompileContext) { + if (urlRecording) { + throw new Error('A URL source map is already being recorded'); + } this.buffer(); + urlRecording = { + segments: [], + valueLength: 0, + }; }; const onexitUrlDestination = function (this: CompileContext, token: any) { const url = this.resume(); const node = this.stack[this.stack.length - 1]; node.url = url; + const recording = urlRecording; + urlRecording = undefined; + if (!recording) { + throw new Error('Missing URL source-map recording'); + } if (node.type === 'link' || node.type === 'definition') { - state.urlSourceSpans.set(node, { - start: token.start.offset, - end: token.end.offset, - }); + if (recording.valueLength === url.length) { + state.urlSegments.set(node, recording.segments); + if (url.length === 0) + state.emptyUrlOffsets.set(node, token.start.offset); + } } }; @@ -591,12 +640,10 @@ export function recordingExtension(state: RecordingState) { if ( (node.type === 'link' || node.type === 'definition') && node.url === '' - && !state.urlSourceSpans.has(node) + && !state.urlSegments.has(node) ) { - state.urlSourceSpans.set(node, { - start: token.start.offset + 1, - end: token.end.offset - 1, - }); + state.urlSegments.set(node, []); + state.emptyUrlOffsets.set(node, token.start.offset + 1); } }; @@ -609,13 +656,11 @@ export function recordingExtension(state: RecordingState) { if ( node.type === 'link' && node.url === '' - && !state.urlSourceSpans.has(node) + && !state.urlSegments.has(node) ) { const emptyOffset = token.end.offset - 1; - state.urlSourceSpans.set(node, { - start: emptyOffset, - end: emptyOffset, - }); + state.urlSegments.set(node, []); + state.emptyUrlOffsets.set(node, emptyOffset); } }; @@ -644,18 +689,22 @@ export function recordingExtension(state: RecordingState) { codeText: onexitcodetext, codeTextData: onexitcodetextdata, data(this: CompileContext, token: any) { - onexitdata.call(this, token, { + const metadata: SegmentMetadata = { sourceStart: token.start.offset, sourceEnd: token.end.offset, kind: 'literal', - }); + }; + const value = onexitdata.call(this, token, metadata); + recordUrlSegment(metadata, value.length); }, characterEscapeValue(this: CompileContext, token: any) { const construct = takePendingConstruct(); - onexitdata.call(this, token, { + const metadata: SegmentMetadata = { ...construct, kind: 'escape', - }); + }; + const value = onexitdata.call(this, token, metadata); + recordUrlSegment(metadata, value.length); }, characterReferenceValue: onexitcharacterreferencevalue, blockQuotePrefix(this: CompileContext, token: any) { diff --git a/src/source-map/types.ts b/src/source-map/types.ts index f046c1b..df7bca8 100644 --- a/src/source-map/types.ts +++ b/src/source-map/types.ts @@ -8,16 +8,6 @@ import type { ParsedPosition, } from '../types'; -/** - * An absolute half-open interval in the Markdown source. - * - * @internal - */ -export interface SourceSpan { - start: number - end: number -} - /** * The kind of transformation the parser applied to turn a slice of the raw * Markdown source into the corresponding slice of a node's normalized value.