test: markdown parser subsystem (58) + custom-emoji readers (32)

Via subagents, probe-verified against real output, no bugs:
- markdown: internal/utils (11), inline/runner (7), inline/parser (21 — bold/
  italic/underline/strike/code/spoiler/link, nesting, precedence, URL lookbehind),
  block/parser (19 — headings/code-fences/quotes/lists/<br>/escapes). Closes the
  biggest coverage hole (core message rendering).
- custom-emoji: PackMetaReader (6), PackImageReader (7), PackImagesReader (4),
  utils equality+makeImagePacks (5), recent-emoji promote/increment/100-cap (10).

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
2026-06-30 14:52:48 -04:00
co-authored by Claude Opus 4.8
parent 160c09e525
commit 230ef8ed7c
9 changed files with 719 additions and 0 deletions
@@ -0,0 +1,112 @@
import { test } from 'node:test';
import assert from 'node:assert/strict';
import { parseBlockMD } from './parser';
import { parseInlineMD } from '../inline/parser';
test('empty string is returned unchanged', () => {
assert.equal(parseBlockMD('', parseInlineMD), '');
});
test('plain single line is returned unchanged', () => {
assert.equal(parseBlockMD('hello', parseInlineMD), 'hello');
});
test('heading levels', () => {
assert.equal(parseBlockMD('# Heading', parseInlineMD), '<h1 data-md="#">Heading</h1>');
assert.equal(parseBlockMD('### Three', parseInlineMD), '<h3 data-md="###">Three</h3>');
});
test('heading requires a space after the hashes', () => {
// no space => not a heading, falls through to plain text
assert.equal(parseBlockMD('#nospace', parseInlineMD), '#nospace');
});
test('inline markdown inside a heading is parsed', () => {
assert.equal(
parseBlockMD('# **b**', parseInlineMD),
'<h1 data-md="#"><strong data-md="**">b</strong></h1>',
);
});
test('heading without a parseInline function keeps the raw text', () => {
assert.equal(parseBlockMD('# Heading', undefined), '<h1 data-md="#">Heading</h1>');
});
test('fenced code block without info string', () => {
assert.equal(
parseBlockMD('```\ncode\n```', parseInlineMD),
'<pre data-md="```"><code>code\n</code></pre>',
);
});
test('fenced code block with a language', () => {
assert.equal(
parseBlockMD('```js\ncode\n```', parseInlineMD),
'<pre data-md="```"><code class="language-js">code\n</code></pre>',
);
});
test('fenced code block with a filename adds language and data-label', () => {
assert.equal(
parseBlockMD('```example.json\ncode\n```', parseInlineMD),
'<pre data-md="```"><code class="language-json" data-label="example.json">code\n</code></pre>',
);
});
test('blockquote single line', () => {
assert.equal(
parseBlockMD('> quote', parseInlineMD),
'<blockquote data-md=">">quote<br/></blockquote>',
);
});
test('blockquote multiple lines', () => {
assert.equal(
parseBlockMD('> a\n> b', parseInlineMD),
'<blockquote data-md=">">a<br/>b<br/></blockquote>',
);
});
test('unordered list', () => {
assert.equal(parseBlockMD('* item', parseInlineMD), '<ul data-md="*"><li><p>item</p></li></ul>');
});
test('ordered list', () => {
assert.equal(
parseBlockMD('1. item', parseInlineMD),
'<ol data-md="1" start="1"><li><p>item</p></li></ol>',
);
});
test('list with multiple items', () => {
assert.equal(
parseBlockMD('* a\n* b', parseInlineMD),
'<ul data-md="*"><li><p>a</p></li><li><p>b</p></li></ul>',
);
});
test('nested list opens a child list', () => {
assert.equal(
parseBlockMD('* a\n * b', parseInlineMD),
'<ul data-md="*"><li><p>a</p><ul data-md="*"><li><p>b</p></li></ul></ul>',
);
});
test('inline markdown inside list items is parsed', () => {
assert.equal(
parseBlockMD('1. **b**', parseInlineMD),
'<ol data-md="1" start="1"><li><p><strong data-md="**">b</strong></p></li></ol>',
);
});
test('newlines are preserved as <br/>', () => {
assert.equal(parseBlockMD('line1\nline2', parseInlineMD), 'line1<br/>line2');
});
test('empty lines are preserved as <br/>', () => {
assert.equal(parseBlockMD('a\n\nb', parseInlineMD), 'a<br/><br/>b');
});
test('escaped block sequence is unescaped and not treated as a block', () => {
assert.equal(parseBlockMD('\\# not heading', parseInlineMD), '# not heading');
});
@@ -0,0 +1,102 @@
import { test } from 'node:test';
import assert from 'node:assert/strict';
import { parseInlineMD } from './parser';
test('empty string is returned unchanged', () => {
assert.equal(parseInlineMD(''), '');
});
test('plain text without markdown is returned unchanged', () => {
assert.equal(parseInlineMD('hello world'), 'hello world');
});
test('bold', () => {
assert.equal(parseInlineMD('**bold**'), '<strong data-md="**">bold</strong>');
});
test('italic with asterisk', () => {
assert.equal(parseInlineMD('*italic*'), '<i data-md="*">italic</i>');
});
test('italic with underscore', () => {
assert.equal(parseInlineMD('_italic_'), '<i data-md="_">italic</i>');
});
test('underline', () => {
assert.equal(parseInlineMD('__under__'), '<u data-md="__">under</u>');
});
test('strikethrough', () => {
assert.equal(parseInlineMD('~~strike~~'), '<s data-md="~~">strike</s>');
});
test('inline code', () => {
assert.equal(parseInlineMD('`code`'), '<code data-md="`">code</code>');
});
test('inline code does not parse markdown inside it', () => {
// code is run before the other rules and does not re-parse its content
assert.equal(parseInlineMD('`**bold**`'), '<code data-md="`">**bold**</code>');
});
test('spoiler', () => {
assert.equal(parseInlineMD('||secret||'), '<span data-md="||" data-mx-spoiler>secret</span>');
});
test('link', () => {
assert.equal(
parseInlineMD('[alt](https://example.com)'),
'<a data-md href="https://example.com">alt</a>',
);
});
test('escaped markdown characters are unescaped to literal text', () => {
assert.equal(parseInlineMD('\\*notbold\\*'), '*notbold*');
assert.equal(parseInlineMD('a\\_b'), 'a_b');
});
test('nesting: italic inside bold', () => {
assert.equal(
parseInlineMD('**bold *italic***'),
'<strong data-md="**">bold <i data-md="*">italic</i></strong>',
);
});
test('nesting: bold inside link alt text', () => {
assert.equal(
parseInlineMD('[**b**](https://e.com)'),
'<a data-md href="https://e.com"><strong data-md="**">b</strong></a>',
);
});
test('nesting: bold inside spoiler', () => {
assert.equal(
parseInlineMD('||**b**||'),
'<span data-md="||" data-mx-spoiler><strong data-md="**">b</strong></span>',
);
});
test('adjacent tokens of different types are both parsed', () => {
assert.equal(parseInlineMD('**a**_b_'), '<strong data-md="**">a</strong><i data-md="_">b</i>');
});
test('text surrounding a token is preserved', () => {
assert.equal(parseInlineMD('pre **mid** post'), 'pre <strong data-md="**">mid</strong> post');
});
test('two separate tokens are parsed in their text order', () => {
assert.equal(parseInlineMD('*a* **b**'), '<i data-md="*">a</i> <strong data-md="**">b</strong>');
assert.equal(parseInlineMD('__a__ _b_'), '<u data-md="__">a</u> <i data-md="_">b</i>');
});
test('code takes precedence and is resolved before other inline rules', () => {
assert.equal(parseInlineMD('`a` *b*'), '<code data-md="`">a</code> <i data-md="*">b</i>');
});
test('unclosed token is returned as literal text', () => {
assert.equal(parseInlineMD('**unclosed'), '**unclosed');
});
test('markdown characters inside a URL are not parsed (negative lookbehind)', () => {
assert.equal(parseInlineMD('https://e.com/*path*'), 'https://e.com/*path*');
});
@@ -0,0 +1,56 @@
import { test } from 'node:test';
import assert from 'node:assert/strict';
import { runInlineRule, runInlineRules } from './runner';
import { parseInlineMD } from './parser';
import { BoldRule, ItalicRule1, StrikeRule } from './rules';
import { InlineMDRule } from './type';
// A trivial rule that wraps the matched token in <X>...</X>.
const makeRule = (token: string, tag: string): InlineMDRule => ({
match: (text) => text.match(new RegExp(token)),
html: (parse, match) => `<${tag}>${parse(match[0])}</${tag}>`,
});
test('runInlineRule applies a matching rule', () => {
assert.equal(runInlineRule('**b**', BoldRule, parseInlineMD), '<strong data-md="**">b</strong>');
});
test('runInlineRule returns undefined when the rule does not match', () => {
assert.equal(runInlineRule('plain', BoldRule, parseInlineMD), undefined);
});
test('runInlineRule recursively parses surrounding text', () => {
// bold matches in the middle; text on both sides is re-parsed (here it is plain)
assert.equal(
runInlineRule('pre **mid** post', BoldRule, parseInlineMD),
'pre <strong data-md="**">mid</strong> post',
);
});
test('runInlineRules returns undefined when no rule matches', () => {
assert.equal(runInlineRules('plain text', [BoldRule, ItalicRule1], parseInlineMD), undefined);
});
test('runInlineRules picks the earliest-matching rule regardless of rule order', () => {
// italic appears before bold in the text, so italic wins even though Bold is listed first
assert.equal(
runInlineRules('a *i* **b**', [BoldRule, ItalicRule1], parseInlineMD),
'a <i data-md="*">i</i> <strong data-md="**">b</strong>',
);
});
test('runInlineRules breaks index ties by rule order (first listed wins)', () => {
// Two synthetic rules both match at index 0; the first in the list wins.
const aRule = makeRule('a', 'A');
const a2Rule = makeRule('a', 'B');
assert.equal(runInlineRules('a', [aRule, a2Rule], parseInlineMD), '<A>a</A>');
assert.equal(runInlineRules('a', [a2Rule, aRule], parseInlineMD), '<B>a</B>');
});
test('runInlineRules earliest index wins even when a later-listed rule matches sooner', () => {
// Strike token "~~s~~" is at index 0; bold token is later in the string.
assert.equal(
runInlineRules('~~s~~ **b**', [BoldRule, StrikeRule], parseInlineMD),
'<s data-md="~~">s</s> <strong data-md="**">b</strong>',
);
});
@@ -0,0 +1,74 @@
import { test } from 'node:test';
import assert from 'node:assert/strict';
import { beforeMatch, afterMatch, replaceMatch } from './utils';
test('beforeMatch returns the slice before the match', () => {
const match = 'abXYcd'.match(/XY/)!;
assert.equal(beforeMatch('abXYcd', match), 'ab');
});
test('beforeMatch is empty when the match is at the start', () => {
const match = 'XYcd'.match(/XY/)!;
assert.equal(beforeMatch('XYcd', match), '');
});
test('beforeMatch on a match without an index treats index as undefined', () => {
// text.slice(0, undefined) returns the whole string
const fakeMatch = ['X'] as unknown as RegExpMatchArray;
assert.equal(beforeMatch('Xabc', fakeMatch), 'Xabc');
});
test('afterMatch returns the slice after the match', () => {
const match = 'abXYcd'.match(/XY/)!;
assert.equal(afterMatch('abXYcd', match), 'cd');
});
test('afterMatch is empty when the match runs to the end', () => {
const match = 'abXY'.match(/XY/)!;
assert.equal(afterMatch('abXY', match), '');
});
test('afterMatch handles a match at the start', () => {
const match = 'XYcd'.match(/XY/)!;
assert.equal(afterMatch('XYcd', match), 'cd');
});
test('afterMatch falls back to index 0 when match.index is missing', () => {
// (undefined ?? 0) + match[0].length === 1, so it slices off the first char
const fakeMatch = ['X'] as unknown as RegExpMatchArray;
assert.equal(afterMatch('Xabc', fakeMatch), 'abc');
});
test('replaceMatch splices content between processed before/after parts', () => {
const match = 'abXYcd'.match(/XY/)!;
assert.deepEqual(
replaceMatch('abXYcd', match, '<R>', (t) => [t]),
['ab', '<R>', 'cd'],
);
});
test('replaceMatch applies processPart to the surrounding text', () => {
const match = 'abXYcd'.match(/XY/)!;
assert.deepEqual(
replaceMatch('abXYcd', match, '<R>', (t) => [t.toUpperCase()]),
['AB', '<R>', 'CD'],
);
});
test('replaceMatch keeps empty surrounding parts produced by processPart', () => {
const match = 'XY'.match(/XY/)!;
// empty before and after still flow through processPart
assert.deepEqual(
replaceMatch('XY', match, '<R>', (t) => [t]),
['', '<R>', ''],
);
});
test('replaceMatch supports non-string content types', () => {
const match = 'aXb'.match(/X/)!;
const node = { type: 'node' };
assert.deepEqual(
replaceMatch('aXb', match, node, (t) => [t]),
['a', node, 'b'],
);
});