Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,10 @@

## Unreleased

### Added

- 新增 `MarkdownSourceMap.getCodeSourceInfo()`。它返回 block `code` 节点的源码结构:fenced(含 `openingFence` 与 `infoInsertPoint`)或 indented。parser 在解析阶段记录该结构,consumer 不再重新扫描 Markdown(#141)。

## 0.3.2

### Refactored
Expand Down
1 change: 1 addition & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -92,6 +92,7 @@ sourceMap.getRaw(textNode); // 'A&B'

- `getSourceRange(node, valueStart, valueEnd)` 的索引与 JavaScript 字符串下标一致,范围均为半开区间 `[start, end)`;当前支持 `text.value`、`inlineCode.value` 与 block `code.value`。`getFieldSourceRange(node, 'url', valueStart, valueEnd)` 当前支持 inline resource link 与 definition 的 destination;autolink 和 GFM autolink literal 暂不包含。
- `getValueSourceIndex(node).sourceOffsetAt(valueIndex)` 返回原始 Markdown 的绝对 offset。该接口为顺序边界查询保留游标。反向和随机查询使用二分查找。
- `getCodeSourceInfo(node)` 返回 block `code` 节点的源码结构。fenced code 返回 `{ kind: 'fenced', openingFence, infoInsertPoint }`,indented code 返回 `{ kind: 'indented' }`。`openingFence` 是围栏序列(反引号或波浪线)的范围;`infoInsertPoint` 是 opening fence 行末、行结束符之前的位置。当 `node.lang` 缺失时,在该 offset 插入 info string 即可补上语言;该位置不会落在 CRLF 的行结束符中间。parser 在解析阶段记录这些位置,不重新扫描 Markdown。
- 映射覆盖受支持节点的整个 `value`,segment 之间无空洞、无重叠。
- 当 value 对应的原始源码连续时,`getSourceRange(node, 0, node.value.length)` 覆盖该节点 value 的完整原始来源范围。当 blockquote marker、list indentation 等容器语法将来源分隔开时,单个连续的 `ParsedPosition` 无法准确表达该范围,`getSourceRange()` 会抛出 `RangeError`。
- 错误分为三条路径,专属错误均继承 `RangeError`(现有 `catch (RangeError)` 不受影响),并带稳定的 `code` 字段;当跨边界传递时(如跨 CJS/ESM 实例、重复安装、worker 边界),只要错误被显式序列化且 `code` 字段被保留,即可用 `code` 而非 `instanceof` 判断(类本身无法保证任意序列化机制一定保留自定义属性):
Expand Down
259 changes: 259 additions & 0 deletions __tests__/source-map/code-source-info.spec.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,259 @@
import {
SourceMapUnavailableError,
parseMdWithSourceMap,
} from '../helpers';

/** Collect matching nodes in document order. */
function nodesOfType(root: any, type: string): any[] {
const out: any[] = [];
(function walk(n: any) {
if (n.type === type) out.push(n);
for (const c of n.children || []) walk(c);
})(root);
return out;
}

function parse(md: string): { ast: any; sourceMap: any; node: any } {
const { ast, sourceMap } = parseMdWithSourceMap(md);
return { ast, sourceMap, node: nodesOfType(ast, 'code')[0] };
}

function insertAt(md: string, offset: number, text: string): string {
return md.slice(0, offset) + text + md.slice(offset);
}

/** Reparse `md` and return the first code node's lang and value. */
function reparseCode(md: string): { lang: string | null; value: string } {
const { ast } = parseMdWithSourceMap(md);
const node = nodesOfType(ast, 'code')[0];
return { lang: node.lang ?? null, value: node.value };
}

/**
* Apply the `no-empty-code-lang` style fix through the public query. Insert
* `plain` only when the node has no lang and the block is fenced.
*/
function fixEmptyCodeLang(md: string): string {
const { sourceMap, node } = parse(md);
const info = sourceMap.getCodeSourceInfo(node);
if (node.lang || info.kind !== 'fenced')
return md;
return insertAt(md, info.infoInsertPoint.offset, 'plain');
}

describe('sourceMap.getCodeSourceInfo', () => {
describe('fenced code', () => {
test.each([
['```\ncode\n```', 3, '```'],
['~~~\ncode\n~~~', 3, '~~~'],
['````\ncode\n````', 4, '````'],
['~~~~\ncode\n~~~~', 4, '~~~~'],
['`````\ncode\n`````', 5, '`````'],
])(
'reports the opening fence and info insert point for %p',
(md, fenceLength, fence) => {
const { sourceMap, node } = parse(md);
const info = sourceMap.getCodeSourceInfo(node);

expect(info.kind).toBe('fenced');
expect(info.openingFence.start.offset).toBe(node.position.start.offset);
expect(info.openingFence.end.offset).toBe(fenceLength);
expect(
md.slice(info.openingFence.start.offset, info.openingFence.end.offset),
).toBe(fence);
expect(info.infoInsertPoint.offset).toBe(fenceLength);
},
);

test('places infoInsertPoint at the end of the line, not after the fence', () => {
const md = '```js\ncode\n```';
const { sourceMap, node } = parse(md);
const info = sourceMap.getCodeSourceInfo(node);

expect(info.kind).toBe('fenced');
expect(info.openingFence.start.offset).toBe(0);
expect(info.openingFence.end.offset).toBe(3);
// The line is "```js", so the insert point is after "js".
expect(info.infoInsertPoint).toEqual({ line: 1, column: 6, offset: 5 });
});

test('keeps infoInsertPoint before trailing whitespace line ending', () => {
const md = '``` \ncode\n```';
const { sourceMap, node } = parse(md);
const info = sourceMap.getCodeSourceInfo(node);

expect(info.kind).toBe('fenced');
expect(info.infoInsertPoint).toEqual({ line: 1, column: 7, offset: 6 });
});

test.each([
['```\ncode\n```', 3],
['```\r\ncode\r\n```', 3],
['```\rcode\r```', 3],
['``` \ncode\n```', 6],
['``` \r\ncode\r\n```', 6],
['``` \rcode\r```', 6],
])(
'positions infoInsertPoint before the line ending for %p',
(md, insertOffset) => {
const { sourceMap, node } = parse(md);
const info = sourceMap.getCodeSourceInfo(node);

expect(info.kind).toBe('fenced');
expect(info.infoInsertPoint.offset).toBe(insertOffset);
// Never inside a CRLF pair: the point is a line ending or end of input.
const at = md[insertOffset];
expect(at === undefined || at === '\n' || at === '\r').toBe(true);
},
);

test('handles a fenced block in a blockquote', () => {
const md = '> ```\n> code\n> ```';
const { sourceMap, node } = parse(md);
const info = sourceMap.getCodeSourceInfo(node);

expect(info.kind).toBe('fenced');
expect(md.slice(info.openingFence.start.offset, info.openingFence.end.offset))
.toBe('```');
expect(info.infoInsertPoint.offset).toBe(md.indexOf('```') + 3);

const fixed = fixEmptyCodeLang(md);
expect(reparseCode(fixed)).toEqual({ lang: 'plain', value: 'code' });
});

test('handles a fenced block in a blockquote with a tab prefix', () => {
const md = '> \t```\n> \tcode\n> \t```';
const { sourceMap, node } = parse(md);
const info = sourceMap.getCodeSourceInfo(node);

expect(info.kind).toBe('fenced');
expect(info.infoInsertPoint.offset).toBe(md.indexOf('```') + 3);
const fixed = fixEmptyCodeLang(md);
expect(reparseCode(fixed)).toEqual({ lang: 'plain', value: 'code' });
});

test('handles a fenced block nested in a list', () => {
const md = '- item\n ```\n value\n ```';
const { sourceMap, node } = parse(md);
const info = sourceMap.getCodeSourceInfo(node);

expect(info.kind).toBe('fenced');
expect(md.slice(info.openingFence.start.offset, info.openingFence.end.offset))
.toBe('```');
expect(info.infoInsertPoint.offset).toBe(md.indexOf('```') + 3);

const fixed = fixEmptyCodeLang(md);
expect(reparseCode(fixed)).toEqual({ lang: 'plain', value: 'value' });
});

test('handles a fenced block in a list with tab indentation', () => {
const md = '- item\n\t```\n code\n\t```';
const { sourceMap, node } = parse(md);
const info = sourceMap.getCodeSourceInfo(node);

expect(info.kind).toBe('fenced');
expect(info.infoInsertPoint.offset).toBe(md.indexOf('```') + 3);
const fixed = fixEmptyCodeLang(md);
expect(reparseCode(fixed)).toEqual({ lang: 'plain', value: 'code' });
});

test.each([
['```\n```', 3],
['```\n\n```', 3],
['```\r\n```', 3],
])('handles an empty fenced block: %p', (md, insertOffset) => {
const { sourceMap, node } = parse(md);
const info = sourceMap.getCodeSourceInfo(node);

expect(node.value).toBe('');
expect(info.kind).toBe('fenced');
expect(info.infoInsertPoint.offset).toBe(insertOffset);

const fixed = fixEmptyCodeLang(md);
expect(reparseCode(fixed)).toEqual({ lang: 'plain', value: '' });
});

test.each([
['```\ncode', 3],
['```', 3],
['``` ', 6],
['```\r\ncode', 3],
['> ```', 5],
['> ```\n', 5],
['> ```\r', 5],
['> ```\r\n', 5],
])('handles an unclosed fenced block: %p', (md, insertOffset) => {
const { sourceMap, node } = parse(md);
const info = sourceMap.getCodeSourceInfo(node);

expect(info.kind).toBe('fenced');
expect(info.infoInsertPoint.offset).toBe(insertOffset);
if (insertOffset >= md.length)
expect(md[insertOffset]).toBeUndefined();
else
expect(['\n', '\r']).toContain(md[insertOffset]);

const fixed = fixEmptyCodeLang(md);
expect(reparseCode(fixed).lang).toBe('plain');
});

test('is stable when applied repeatedly (fixer convergence)', () => {
const md = '````\ncode\n````';
let current = md;
for (let i = 0; i < 5; i++)
current = fixEmptyCodeLang(current);

expect(reparseCode(current)).toEqual({ lang: 'plain', value: 'code' });
expect(current).toBe('````plain\ncode\n````');
});
});

describe('indented code', () => {
test.each([
' code\n',
'\tcode\n',
' a\n\n b\n',
'> indented\n> code',
'- Foo\n\n bar\n baz',
])('reports indented structure for %p', (md) => {
const { sourceMap, node } = parse(md);
expect(sourceMap.getCodeSourceInfo(node)).toEqual({ kind: 'indented' });
});

test('does not apply an info-string fix to indented code', () => {
const md = ' const a = 1;\n';
expect(fixEmptyCodeLang(md)).toBe(md);
});
});

describe('errors', () => {
test('rejects a node from another document', () => {
const a = parse('```\ncode\n```');
const b = parse('```\ncode\n```');
expect(() => a.sourceMap.getCodeSourceInfo(b.node)).toThrow(
SourceMapUnavailableError,
);
});

test('rejects a node that was added after parsing', () => {
const { ast, sourceMap } = parseMdWithSourceMap('```\ncode\n```');
const generated = { type: 'code', lang: null, meta: null, value: '' };
ast.children.push(generated);
expect(() => sourceMap.getCodeSourceInfo(generated as any)).toThrow(
SourceMapUnavailableError,
);
});

test('rejects an owned non-code node', () => {
const { ast, sourceMap } = parseMdWithSourceMap('paragraph');
const paragraph = ast.children[0];
expect(paragraph.type).toBe('paragraph');
// The node is owned (it came from the parse), but no code structure was
// recorded for it. This must reach the missing-info branch, not the
// ownership branch.
expect(() => sourceMap.getCodeSourceInfo(paragraph as any)).toThrow(
/no source structure is available/,
);
});
});
});
22 changes: 22 additions & 0 deletions __tests__/types/package-exports.cts
Original file line number Diff line number Diff line change
Expand Up @@ -33,6 +33,26 @@ const urlRange = doc.sourceMap.getFieldSourceRange(
0,
1,
);
const codeNode = doc.ast.children[0] as parser.MarkdownCodeNode;
const codeSourceInfo: parser.CodeSourceInfo =
doc.sourceMap.getCodeSourceInfo(codeNode);

// The three new public types must stay exported.
type ExportedCodeSourceInfo = parser.CodeSourceInfo;
type ExportedFencedCodeSourceInfo = parser.FencedCodeSourceInfo;
type ExportedIndentedCodeSourceInfo = parser.IndentedCodeSourceInfo;

// A consumer narrows the result with `kind`.
if (codeSourceInfo.kind === 'fenced') {
const fenceStart: number = codeSourceInfo.openingFence.start.offset;
const infoOffset: number = codeSourceInfo.infoInsertPoint.offset;
void fenceStart;
void infoOffset;
}
else {
const indentedKind: 'indented' = codeSourceInfo.kind;
void indentedKind;
}
const consistency = new parser.SourceMapConsistencyError();
const asRangeError: RangeError = new parser.SourceMapUnavailableError();
const isSourceMapError: boolean = consistency instanceof parser.SourceMapError;
Expand All @@ -45,6 +65,8 @@ void codeRange;
void sourceIndex;
void sourceOffset;
void urlRange;
void codeNode;
void codeSourceInfo;
void codeRaw;
void consistency;
void asRangeError;
Expand Down
25 changes: 25 additions & 0 deletions __tests__/types/package-exports.mts
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,10 @@ import {
SourceMapConsistencyError,
SourceMapUnavailableError,
type SourceMapErrorCode,
type CodeSourceInfo,
type FencedCodeSourceInfo,
type IndentedCodeSourceInfo,
type MarkdownCodeNode,
type ParsedMarkdownDocument,
type MarkdownLinkNode,
type MarkdownTextNode,
Expand Down Expand Up @@ -37,6 +41,25 @@ const urlRange = doc.sourceMap.getFieldSourceRange(
0,
1,
);
const codeNode = doc.ast.children[0] as MarkdownCodeNode;
const codeSourceInfo: CodeSourceInfo = doc.sourceMap.getCodeSourceInfo(codeNode);

// The three new public types must stay exported.
type ExportedCodeSourceInfo = CodeSourceInfo;
type ExportedFencedCodeSourceInfo = FencedCodeSourceInfo;
type ExportedIndentedCodeSourceInfo = IndentedCodeSourceInfo;

// A consumer narrows the result with `kind`.
if (codeSourceInfo.kind === 'fenced') {
const fenceStart: number = codeSourceInfo.openingFence.start.offset;
const infoOffset: number = codeSourceInfo.infoInsertPoint.offset;
void fenceStart;
void infoOffset;
}
else {
const indentedKind: 'indented' = codeSourceInfo.kind;
void indentedKind;
}

// @ts-expect-error segment implementation details are intentionally internal.
type HiddenSegment = import('@lint-md/parser').MarkdownSourceMapSegment;
Expand All @@ -59,6 +82,8 @@ void doc;
void sourceIndex;
void sourceOffset;
void urlRange;
void codeNode;
void codeSourceInfo;
void consistency;
void unavailable;
void asRangeError;
Expand Down
16 changes: 16 additions & 0 deletions etc/parser.api.md
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,21 @@ import { Text as Text_2 } from 'mdast';

export { Code }

// @public
export type CodeSourceInfo = FencedCodeSourceInfo | IndentedCodeSourceInfo;

// @public
export interface FencedCodeSourceInfo {
infoInsertPoint: ParsedPoint;
kind: 'fenced';
openingFence: ParsedPosition;
}

// @public
export interface IndentedCodeSourceInfo {
kind: 'indented';
}

export { Link }

export { ListItem }
Expand Down Expand Up @@ -85,6 +100,7 @@ export type MarkdownRoot = Root;

// @public
export interface MarkdownSourceMap {
getCodeSourceInfo(node: MarkdownCodeNode): CodeSourceInfo;
getFieldSourceRange(node: MarkdownLinkNode | MarkdownDefinitionNode, field: 'url', valueStart: number, valueEnd: number): ParsedPosition;
getRaw(node: MarkdownNode | MarkdownTextNode | MarkdownInlineCodeNode | MarkdownCodeNode | MarkdownLinkNode | MarkdownDefinitionNode): string;
getSourceRange(node: MarkdownTextNode | MarkdownInlineCodeNode | MarkdownCodeNode, valueStart: number, valueEnd: number): ParsedPosition;
Expand Down
1 change: 1 addition & 0 deletions scripts/bench-parse.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -115,6 +115,7 @@ function makeState(source) {
inlineCodeSegments: new WeakMap(),
codeSegments: new WeakMap(),
emptyCodeOffsets: new WeakMap(),
codeSourceInfos: new WeakMap(),
urlSegments: new WeakMap(),
emptyUrlOffsets: new WeakMap(),
};
Expand Down
Loading
Loading