test: markdown parser subsystem (58) + custom-emoji readers (32)
Via subagents, probe-verified against real output, no bugs: - markdown: internal/utils (11), inline/runner (7), inline/parser (21 — bold/ italic/underline/strike/code/spoiler/link, nesting, precedence, URL lookbehind), block/parser (19 — headings/code-fences/quotes/lists/<br>/escapes). Closes the biggest coverage hole (core message rendering). - custom-emoji: PackMetaReader (6), PackImageReader (7), PackImagesReader (4), utils equality+makeImagePacks (5), recent-emoji promote/increment/100-cap (10). Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,112 @@
|
||||
import { test } from 'node:test';
|
||||
import assert from 'node:assert/strict';
|
||||
import { parseBlockMD } from './parser';
|
||||
import { parseInlineMD } from '../inline/parser';
|
||||
|
||||
test('empty string is returned unchanged', () => {
|
||||
assert.equal(parseBlockMD('', parseInlineMD), '');
|
||||
});
|
||||
|
||||
test('plain single line is returned unchanged', () => {
|
||||
assert.equal(parseBlockMD('hello', parseInlineMD), 'hello');
|
||||
});
|
||||
|
||||
test('heading levels', () => {
|
||||
assert.equal(parseBlockMD('# Heading', parseInlineMD), '<h1 data-md="#">Heading</h1>');
|
||||
assert.equal(parseBlockMD('### Three', parseInlineMD), '<h3 data-md="###">Three</h3>');
|
||||
});
|
||||
|
||||
test('heading requires a space after the hashes', () => {
|
||||
// no space => not a heading, falls through to plain text
|
||||
assert.equal(parseBlockMD('#nospace', parseInlineMD), '#nospace');
|
||||
});
|
||||
|
||||
test('inline markdown inside a heading is parsed', () => {
|
||||
assert.equal(
|
||||
parseBlockMD('# **b**', parseInlineMD),
|
||||
'<h1 data-md="#"><strong data-md="**">b</strong></h1>',
|
||||
);
|
||||
});
|
||||
|
||||
test('heading without a parseInline function keeps the raw text', () => {
|
||||
assert.equal(parseBlockMD('# Heading', undefined), '<h1 data-md="#">Heading</h1>');
|
||||
});
|
||||
|
||||
test('fenced code block without info string', () => {
|
||||
assert.equal(
|
||||
parseBlockMD('```\ncode\n```', parseInlineMD),
|
||||
'<pre data-md="```"><code>code\n</code></pre>',
|
||||
);
|
||||
});
|
||||
|
||||
test('fenced code block with a language', () => {
|
||||
assert.equal(
|
||||
parseBlockMD('```js\ncode\n```', parseInlineMD),
|
||||
'<pre data-md="```"><code class="language-js">code\n</code></pre>',
|
||||
);
|
||||
});
|
||||
|
||||
test('fenced code block with a filename adds language and data-label', () => {
|
||||
assert.equal(
|
||||
parseBlockMD('```example.json\ncode\n```', parseInlineMD),
|
||||
'<pre data-md="```"><code class="language-json" data-label="example.json">code\n</code></pre>',
|
||||
);
|
||||
});
|
||||
|
||||
test('blockquote single line', () => {
|
||||
assert.equal(
|
||||
parseBlockMD('> quote', parseInlineMD),
|
||||
'<blockquote data-md=">">quote<br/></blockquote>',
|
||||
);
|
||||
});
|
||||
|
||||
test('blockquote multiple lines', () => {
|
||||
assert.equal(
|
||||
parseBlockMD('> a\n> b', parseInlineMD),
|
||||
'<blockquote data-md=">">a<br/>b<br/></blockquote>',
|
||||
);
|
||||
});
|
||||
|
||||
test('unordered list', () => {
|
||||
assert.equal(parseBlockMD('* item', parseInlineMD), '<ul data-md="*"><li><p>item</p></li></ul>');
|
||||
});
|
||||
|
||||
test('ordered list', () => {
|
||||
assert.equal(
|
||||
parseBlockMD('1. item', parseInlineMD),
|
||||
'<ol data-md="1" start="1"><li><p>item</p></li></ol>',
|
||||
);
|
||||
});
|
||||
|
||||
test('list with multiple items', () => {
|
||||
assert.equal(
|
||||
parseBlockMD('* a\n* b', parseInlineMD),
|
||||
'<ul data-md="*"><li><p>a</p></li><li><p>b</p></li></ul>',
|
||||
);
|
||||
});
|
||||
|
||||
test('nested list opens a child list', () => {
|
||||
assert.equal(
|
||||
parseBlockMD('* a\n * b', parseInlineMD),
|
||||
'<ul data-md="*"><li><p>a</p><ul data-md="*"><li><p>b</p></li></ul></ul>',
|
||||
);
|
||||
});
|
||||
|
||||
test('inline markdown inside list items is parsed', () => {
|
||||
assert.equal(
|
||||
parseBlockMD('1. **b**', parseInlineMD),
|
||||
'<ol data-md="1" start="1"><li><p><strong data-md="**">b</strong></p></li></ol>',
|
||||
);
|
||||
});
|
||||
|
||||
test('newlines are preserved as <br/>', () => {
|
||||
assert.equal(parseBlockMD('line1\nline2', parseInlineMD), 'line1<br/>line2');
|
||||
});
|
||||
|
||||
test('empty lines are preserved as <br/>', () => {
|
||||
assert.equal(parseBlockMD('a\n\nb', parseInlineMD), 'a<br/><br/>b');
|
||||
});
|
||||
|
||||
test('escaped block sequence is unescaped and not treated as a block', () => {
|
||||
assert.equal(parseBlockMD('\\# not heading', parseInlineMD), '# not heading');
|
||||
});
|
||||
@@ -0,0 +1,102 @@
|
||||
import { test } from 'node:test';
|
||||
import assert from 'node:assert/strict';
|
||||
import { parseInlineMD } from './parser';
|
||||
|
||||
test('empty string is returned unchanged', () => {
|
||||
assert.equal(parseInlineMD(''), '');
|
||||
});
|
||||
|
||||
test('plain text without markdown is returned unchanged', () => {
|
||||
assert.equal(parseInlineMD('hello world'), 'hello world');
|
||||
});
|
||||
|
||||
test('bold', () => {
|
||||
assert.equal(parseInlineMD('**bold**'), '<strong data-md="**">bold</strong>');
|
||||
});
|
||||
|
||||
test('italic with asterisk', () => {
|
||||
assert.equal(parseInlineMD('*italic*'), '<i data-md="*">italic</i>');
|
||||
});
|
||||
|
||||
test('italic with underscore', () => {
|
||||
assert.equal(parseInlineMD('_italic_'), '<i data-md="_">italic</i>');
|
||||
});
|
||||
|
||||
test('underline', () => {
|
||||
assert.equal(parseInlineMD('__under__'), '<u data-md="__">under</u>');
|
||||
});
|
||||
|
||||
test('strikethrough', () => {
|
||||
assert.equal(parseInlineMD('~~strike~~'), '<s data-md="~~">strike</s>');
|
||||
});
|
||||
|
||||
test('inline code', () => {
|
||||
assert.equal(parseInlineMD('`code`'), '<code data-md="`">code</code>');
|
||||
});
|
||||
|
||||
test('inline code does not parse markdown inside it', () => {
|
||||
// code is run before the other rules and does not re-parse its content
|
||||
assert.equal(parseInlineMD('`**bold**`'), '<code data-md="`">**bold**</code>');
|
||||
});
|
||||
|
||||
test('spoiler', () => {
|
||||
assert.equal(parseInlineMD('||secret||'), '<span data-md="||" data-mx-spoiler>secret</span>');
|
||||
});
|
||||
|
||||
test('link', () => {
|
||||
assert.equal(
|
||||
parseInlineMD('[alt](https://example.com)'),
|
||||
'<a data-md href="https://example.com">alt</a>',
|
||||
);
|
||||
});
|
||||
|
||||
test('escaped markdown characters are unescaped to literal text', () => {
|
||||
assert.equal(parseInlineMD('\\*notbold\\*'), '*notbold*');
|
||||
assert.equal(parseInlineMD('a\\_b'), 'a_b');
|
||||
});
|
||||
|
||||
test('nesting: italic inside bold', () => {
|
||||
assert.equal(
|
||||
parseInlineMD('**bold *italic***'),
|
||||
'<strong data-md="**">bold <i data-md="*">italic</i></strong>',
|
||||
);
|
||||
});
|
||||
|
||||
test('nesting: bold inside link alt text', () => {
|
||||
assert.equal(
|
||||
parseInlineMD('[**b**](https://e.com)'),
|
||||
'<a data-md href="https://e.com"><strong data-md="**">b</strong></a>',
|
||||
);
|
||||
});
|
||||
|
||||
test('nesting: bold inside spoiler', () => {
|
||||
assert.equal(
|
||||
parseInlineMD('||**b**||'),
|
||||
'<span data-md="||" data-mx-spoiler><strong data-md="**">b</strong></span>',
|
||||
);
|
||||
});
|
||||
|
||||
test('adjacent tokens of different types are both parsed', () => {
|
||||
assert.equal(parseInlineMD('**a**_b_'), '<strong data-md="**">a</strong><i data-md="_">b</i>');
|
||||
});
|
||||
|
||||
test('text surrounding a token is preserved', () => {
|
||||
assert.equal(parseInlineMD('pre **mid** post'), 'pre <strong data-md="**">mid</strong> post');
|
||||
});
|
||||
|
||||
test('two separate tokens are parsed in their text order', () => {
|
||||
assert.equal(parseInlineMD('*a* **b**'), '<i data-md="*">a</i> <strong data-md="**">b</strong>');
|
||||
assert.equal(parseInlineMD('__a__ _b_'), '<u data-md="__">a</u> <i data-md="_">b</i>');
|
||||
});
|
||||
|
||||
test('code takes precedence and is resolved before other inline rules', () => {
|
||||
assert.equal(parseInlineMD('`a` *b*'), '<code data-md="`">a</code> <i data-md="*">b</i>');
|
||||
});
|
||||
|
||||
test('unclosed token is returned as literal text', () => {
|
||||
assert.equal(parseInlineMD('**unclosed'), '**unclosed');
|
||||
});
|
||||
|
||||
test('markdown characters inside a URL are not parsed (negative lookbehind)', () => {
|
||||
assert.equal(parseInlineMD('https://e.com/*path*'), 'https://e.com/*path*');
|
||||
});
|
||||
@@ -0,0 +1,56 @@
|
||||
import { test } from 'node:test';
|
||||
import assert from 'node:assert/strict';
|
||||
import { runInlineRule, runInlineRules } from './runner';
|
||||
import { parseInlineMD } from './parser';
|
||||
import { BoldRule, ItalicRule1, StrikeRule } from './rules';
|
||||
import { InlineMDRule } from './type';
|
||||
|
||||
// A trivial rule that wraps the matched token in <X>...</X>.
|
||||
const makeRule = (token: string, tag: string): InlineMDRule => ({
|
||||
match: (text) => text.match(new RegExp(token)),
|
||||
html: (parse, match) => `<${tag}>${parse(match[0])}</${tag}>`,
|
||||
});
|
||||
|
||||
test('runInlineRule applies a matching rule', () => {
|
||||
assert.equal(runInlineRule('**b**', BoldRule, parseInlineMD), '<strong data-md="**">b</strong>');
|
||||
});
|
||||
|
||||
test('runInlineRule returns undefined when the rule does not match', () => {
|
||||
assert.equal(runInlineRule('plain', BoldRule, parseInlineMD), undefined);
|
||||
});
|
||||
|
||||
test('runInlineRule recursively parses surrounding text', () => {
|
||||
// bold matches in the middle; text on both sides is re-parsed (here it is plain)
|
||||
assert.equal(
|
||||
runInlineRule('pre **mid** post', BoldRule, parseInlineMD),
|
||||
'pre <strong data-md="**">mid</strong> post',
|
||||
);
|
||||
});
|
||||
|
||||
test('runInlineRules returns undefined when no rule matches', () => {
|
||||
assert.equal(runInlineRules('plain text', [BoldRule, ItalicRule1], parseInlineMD), undefined);
|
||||
});
|
||||
|
||||
test('runInlineRules picks the earliest-matching rule regardless of rule order', () => {
|
||||
// italic appears before bold in the text, so italic wins even though Bold is listed first
|
||||
assert.equal(
|
||||
runInlineRules('a *i* **b**', [BoldRule, ItalicRule1], parseInlineMD),
|
||||
'a <i data-md="*">i</i> <strong data-md="**">b</strong>',
|
||||
);
|
||||
});
|
||||
|
||||
test('runInlineRules breaks index ties by rule order (first listed wins)', () => {
|
||||
// Two synthetic rules both match at index 0; the first in the list wins.
|
||||
const aRule = makeRule('a', 'A');
|
||||
const a2Rule = makeRule('a', 'B');
|
||||
assert.equal(runInlineRules('a', [aRule, a2Rule], parseInlineMD), '<A>a</A>');
|
||||
assert.equal(runInlineRules('a', [a2Rule, aRule], parseInlineMD), '<B>a</B>');
|
||||
});
|
||||
|
||||
test('runInlineRules earliest index wins even when a later-listed rule matches sooner', () => {
|
||||
// Strike token "~~s~~" is at index 0; bold token is later in the string.
|
||||
assert.equal(
|
||||
runInlineRules('~~s~~ **b**', [BoldRule, StrikeRule], parseInlineMD),
|
||||
'<s data-md="~~">s</s> <strong data-md="**">b</strong>',
|
||||
);
|
||||
});
|
||||
@@ -0,0 +1,74 @@
|
||||
import { test } from 'node:test';
|
||||
import assert from 'node:assert/strict';
|
||||
import { beforeMatch, afterMatch, replaceMatch } from './utils';
|
||||
|
||||
test('beforeMatch returns the slice before the match', () => {
|
||||
const match = 'abXYcd'.match(/XY/)!;
|
||||
assert.equal(beforeMatch('abXYcd', match), 'ab');
|
||||
});
|
||||
|
||||
test('beforeMatch is empty when the match is at the start', () => {
|
||||
const match = 'XYcd'.match(/XY/)!;
|
||||
assert.equal(beforeMatch('XYcd', match), '');
|
||||
});
|
||||
|
||||
test('beforeMatch on a match without an index treats index as undefined', () => {
|
||||
// text.slice(0, undefined) returns the whole string
|
||||
const fakeMatch = ['X'] as unknown as RegExpMatchArray;
|
||||
assert.equal(beforeMatch('Xabc', fakeMatch), 'Xabc');
|
||||
});
|
||||
|
||||
test('afterMatch returns the slice after the match', () => {
|
||||
const match = 'abXYcd'.match(/XY/)!;
|
||||
assert.equal(afterMatch('abXYcd', match), 'cd');
|
||||
});
|
||||
|
||||
test('afterMatch is empty when the match runs to the end', () => {
|
||||
const match = 'abXY'.match(/XY/)!;
|
||||
assert.equal(afterMatch('abXY', match), '');
|
||||
});
|
||||
|
||||
test('afterMatch handles a match at the start', () => {
|
||||
const match = 'XYcd'.match(/XY/)!;
|
||||
assert.equal(afterMatch('XYcd', match), 'cd');
|
||||
});
|
||||
|
||||
test('afterMatch falls back to index 0 when match.index is missing', () => {
|
||||
// (undefined ?? 0) + match[0].length === 1, so it slices off the first char
|
||||
const fakeMatch = ['X'] as unknown as RegExpMatchArray;
|
||||
assert.equal(afterMatch('Xabc', fakeMatch), 'abc');
|
||||
});
|
||||
|
||||
test('replaceMatch splices content between processed before/after parts', () => {
|
||||
const match = 'abXYcd'.match(/XY/)!;
|
||||
assert.deepEqual(
|
||||
replaceMatch('abXYcd', match, '<R>', (t) => [t]),
|
||||
['ab', '<R>', 'cd'],
|
||||
);
|
||||
});
|
||||
|
||||
test('replaceMatch applies processPart to the surrounding text', () => {
|
||||
const match = 'abXYcd'.match(/XY/)!;
|
||||
assert.deepEqual(
|
||||
replaceMatch('abXYcd', match, '<R>', (t) => [t.toUpperCase()]),
|
||||
['AB', '<R>', 'CD'],
|
||||
);
|
||||
});
|
||||
|
||||
test('replaceMatch keeps empty surrounding parts produced by processPart', () => {
|
||||
const match = 'XY'.match(/XY/)!;
|
||||
// empty before and after still flow through processPart
|
||||
assert.deepEqual(
|
||||
replaceMatch('XY', match, '<R>', (t) => [t]),
|
||||
['', '<R>', ''],
|
||||
);
|
||||
});
|
||||
|
||||
test('replaceMatch supports non-string content types', () => {
|
||||
const match = 'aXb'.match(/X/)!;
|
||||
const node = { type: 'node' };
|
||||
assert.deepEqual(
|
||||
replaceMatch('aXb', match, node, (t) => [t]),
|
||||
['a', node, 'b'],
|
||||
);
|
||||
});
|
||||
Reference in New Issue
Block a user