From 151bf38398bf1079f2c26a66bb6be838ad2443a2 Mon Sep 17 00:00:00 2001 From: Lilleman auf Larv Date: Tue, 29 Sep 2026 18:43:04 +0200 Subject: [PATCH 1/5] cjk - tests for a highlight flush against a script written without spaces --- src/markdown/emit/plain-reduction.test.ts | 2 ++ src/markdown/parse/plain-markdown-to-adf.test.ts | 5 ++++- 2 files changed, 6 insertions(+), 1 deletion(-) diff --git a/src/markdown/emit/plain-reduction.test.ts b/src/markdown/emit/plain-reduction.test.ts index 3c15be7..d162eb3 100644 --- a/src/markdown/emit/plain-reduction.test.ts +++ b/src/markdown/emit/plain-reduction.test.ts @@ -148,6 +148,8 @@ test('spells a highlight as a == pair around the run, whatever its colour', () = assert.equal(plain(paragraph(text('a', highlight('#fff'), code), text('b', highlight('#fff')))), '`a`==b==\n') assert.equal(plain(paragraph(text('=', highlight('#fff')), text(' '), text('a==b', highlight('#fff')))), '==\\=== ==a==b==\n') assert.equal(plain(paragraph(text('x'), text('y', highlight('#fff')), text(' z'))), 'xy z\n') + assert.equal(plain(paragraph(text('この機能は'), text('日本語', highlight('#fff')), text('でのみ'))), 'この機能は==日本語==でのみ\n') + assert.equal(plain(paragraph(text('日==本==語'))), '日\\==本\\==語\n') }) test('escapes text a renderer would take as a flavour marker, and only there', () => { diff --git a/src/markdown/parse/plain-markdown-to-adf.test.ts b/src/markdown/parse/plain-markdown-to-adf.test.ts index 68f73a2..d7b9567 100644 --- a/src/markdown/parse/plain-markdown-to-adf.test.ts +++ b/src/markdown/parse/plain-markdown-to-adf.test.ts @@ -149,6 +149,9 @@ test('reads a == pair to the editor default highlight, Yellow200 #f8e6a0 in @atl assert.deepEqual(read('==`a`==\n'), [paragraph(text('a', code))]) assert.deepEqual(read('x==y==z ==a == b==, (==c==) _d_==e==\n'), [paragraph(text('x==y==z '), text('a == b', highlight), text(', ('), text('c', highlight), text(') '), text('d', em), text('e', highlight))]) assert.deepEqual(read('😀==b== ==c==😀 é==d==\n'), [paragraph(text('😀'), text('b', highlight), text(' '), text('c', highlight), text('😀 é==d=='))]) + assert.deepEqual(read('この機能は==日本語==でのみ、中文==重点==内容、ภาษา==ไทย==ดี a==日本==b\n'), [ + paragraph(text('この機能は'), text('日本語', highlight), text('でのみ、中文'), text('重点', highlight), text('内容、ภาษา'), text('ไทย', highlight), text('ดี a==日本==b')), + ]) assert.deepEqual(read('# ==h==\n\n| ==c== |\n| --- |\n'), [ node('heading', { level: 1 }, text('h', highlight)), bare('table', bare('tableRow', bare('tableHeader', paragraph(text('c', highlight))))), @@ -196,7 +199,7 @@ test('reads what the reduction wrote back to the node it reduced, less the attri test('reads text the writer kept from reading as a marker back as text', () => { const quote = bare('blockquote', said('[!NOTE] x')) const list = bare('bulletList', bare('listItem', said('[x] a')), bare('listItem', said('[ ] b'))) - assert.deepEqual(roundTripped(said('==x== a==b'), quote, list), [said('==x== a==b'), quote, list]) + assert.deepEqual(roundTripped(said('==x== a==b 日==本==語'), quote, list), [said('==x== a==b 日==本==語'), quote, list]) assert.deepEqual(roundTripped(bare('taskList', task('DONE', text('[x] ==a==')))), [bare('taskList', task('DONE', text('[x] ==a==')))]) }) -- 2.52.0 From 6c499127b595937d4c0ac37546f4b670a411cc16 Mon Sep 17 00:00:00 2001 From: Lilleman auf Larv Date: Tue, 29 Sep 2026 18:43:29 +0200 Subject: [PATCH 2/5] cjk - a letter of a script written without spaces bounds a highlight --- README.md | 2 +- docs/decisions.md | 12 +++++++----- src/markdown/plain-conventions.ts | 8 +++++++- 3 files changed, 15 insertions(+), 7 deletions(-) diff --git a/README.md b/README.md index 518bfed..4982ef1 100644 --- a/README.md +++ b/README.md @@ -133,7 +133,7 @@ read replaces mentions, attachments and macros with text. | `panel` | a GitHub alert, `> [!WARNING]`: info `NOTE`, note `IMPORTANT`, tip and success `TIP`, warning `WARNING`, error `CAUTION`, custom `NOTE` | `NOTE` info, `IMPORTANT` note, `TIP` tip, `WARNING` warning, `CAUTION` error, and Obsidian's: hint tip; success, check, done success; attention warning; danger, failure, fail, missing, bug, error error; any other word info — in any case; the rest of the marker's line is the first paragraph | | `expand`, `nestedExpand` | Obsidian's folded callout, `> [!NOTE]- Title` | `-` or `+` after any word, the rest of the marker's line the title; an expand inside an expand is a `nestedExpand` | | `taskList` | `- [x] Done`, `- [ ] Todo` | a bullet list whose every item is so marked, `[X]` too | -| `backgroundColor` | `==text==` | `==text==` on one line, the text touching both delimiters, bounded outside by whitespace, punctuation or a line edge, in the editor's default highlight `#f8e6a0` | +| `backgroundColor` | `==text==` | `==text==` on one line, the text touching both delimiters, bounded outside by whitespace, punctuation, a line edge or a letter of a script written without spaces (Han, Hiragana, Katakana, Thai, Lao, Khmer, Myanmar), in the editor's default highlight `#f8e6a0` | | `table` | a pipe table: the first row its header, a cell's blocks on one line, a span kept under its header by empty cells | — | | `decisionList` | a bullet list | — | | `mention`, `status`, `emoji`, `date` | their text: `@` kept, a mention with none `@` and its id, an emoji its `shortName` without, a date `2026-09-13` in UTC | — | diff --git a/docs/decisions.md b/docs/decisions.md index 49848f9..ad4114e 100644 --- a/docs/decisions.md +++ b/docs/decisions.md @@ -145,14 +145,16 @@ parentheses: `> [!faq]- See [x](http://y)` reads to the title `See x (http://y)` ## The plain flavour's spellings -2026-09-14, panels 2026-09-25, the maintainer. Goal 5. Valid while GitHub's renderer is the one the -audience's markdown is read in. +2026-09-14, panels 2026-09-25 and 2026-09-29, the maintainer. Goal 5. Valid while GitHub's +renderer is the one the audience's markdown is read in. README §Plain markdown's rows come from a survey of GitHub, GitLab, Gitea, Obsidian, Pandoc, MkDocs, Docusaurus, Typora, Joplin, Logseq, Bear, Notion, Azure DevOps and Discord, GitHub's -renderer confirming each shape. Reader panels settled `error` as an error panel and the `==` -bounds (3 of 3), the external image's two forms (6 of 7), the omission notes and a rule opening a -list item dropping (3 of 3), and a list's numbering overflowing into bullets (3 of 3, 5 of 7). +renderer confirming each shape. Reader panels settled `error` as an error panel, the `==` bounds +(3 of 3) and a letter of a script written without spaces bounding them, so `は==日本語==で` +highlights (3 of 3), the external image's two forms (6 of 7), the omission notes and a rule +opening a list item dropping (3 of 3), and a list's numbering overflowing into bullets (3 of 3, +5 of 7). Reading takes other tools' spellings, since it reads their output and writes none of them. Rejected: `~sub~` and `^sup^` (`~2~` is a strike on GitHub, so `subsup` drops), underline and colour spellings, raw HTML (`
`, ``), MkDocs `!!!` and the `:::` admonition family, diff --git a/src/markdown/plain-conventions.ts b/src/markdown/plain-conventions.ts index 308ccbe..ce62cc3 100644 --- a/src/markdown/plain-conventions.ts +++ b/src/markdown/plain-conventions.ts @@ -34,6 +34,8 @@ const panelTypesByWord: Readonly> = { warning: 'warning', } +const unspacedScript = /^[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Khmer}\p{Script=Lao}\p{Script=Myanmar}\p{Script=Thai}]$/u + export function alertMarker(panelType: unknown): string { const word = typeof panelType === 'string' && Object.hasOwn(alertWords, panelType) ? alertWords[panelType] : undefined return `[!${word ?? 'NOTE'}]` @@ -60,7 +62,11 @@ export function highlightFlanking(source: string, index: number): { closes: bool const end = index + highlightDelimiter.length const before = Array.from(source.slice(Math.max(0, index - 2), index)).at(-1) ?? '' const after = Array.from(source.slice(end, end + 2))[0] ?? '' - return { closes: flanks(before) && !isWordCharacter(after), opens: flanks(after) && !isWordCharacter(before) } + return { closes: flanks(before) && bounds(after), opens: flanks(after) && bounds(before) } +} + +function bounds(character: string): boolean { + return !isWordCharacter(character) || unspacedScript.test(character) } function flanks(character: string): boolean { -- 2.52.0 From f105bc9a57d77a0143f948a9860c20459c9422f4 Mon Sep 17 00:00:00 2001 From: Lilleman auf Larv Date: Tue, 29 Sep 2026 18:49:37 +0200 Subject: [PATCH 3/5] cjk - review: tests for Hangul, a long-vowel mark and a Latin word bounding a highlight --- src/conformance/property-harness.ts | 2 +- src/markdown/emit/plain-reduction.test.ts | 1 + src/markdown/parse/plain-markdown-to-adf.test.ts | 7 +++++-- 3 files changed, 7 insertions(+), 3 deletions(-) diff --git a/src/conformance/property-harness.ts b/src/conformance/property-harness.ts index 72bf66e..9862945 100644 --- a/src/conformance/property-harness.ts +++ b/src/conformance/property-harness.ts @@ -23,7 +23,7 @@ export const propertyTimeout = 600000 const depthIdentifier = fc.createDepthIdentifier() const emptyCell: AdfNode = { content: [{ type: 'paragraph' }], type: 'tableCell' } const flatCommonMarkShapeWeight = 4 -export const markdownPieces = fc.constantFrom(...'aZ09 \t\n!"#$%&\'()*+,-./:;<=>?@[\\]^_`{|}~é\xa0🎉', '==', '[!NOTE]', '[x]', 'ab:', 'http://', directivePrefix, `${directivePrefix}a[`, `${directivePrefix}a{`) +export const markdownPieces = fc.constantFrom(...'aZ09 \t\n!"#$%&\'()*+,-./:;<=>?@[\\]^_`{|}~é\xa0🎉日ー한', '𠀀', '==', '[!NOTE]', '[x]', 'ab:', 'http://', directivePrefix, `${directivePrefix}a[`, `${directivePrefix}a{`) const nestingCommonMarkShapeWeight = 21 const spelledTypes = new Set(['text', ...Object.keys(blockNodes), ...Object.keys(inlineNodes), ...Object.keys(markAttributes)]) diff --git a/src/markdown/emit/plain-reduction.test.ts b/src/markdown/emit/plain-reduction.test.ts index d162eb3..7f6ac56 100644 --- a/src/markdown/emit/plain-reduction.test.ts +++ b/src/markdown/emit/plain-reduction.test.ts @@ -149,6 +149,7 @@ test('spells a highlight as a == pair around the run, whatever its colour', () = assert.equal(plain(paragraph(text('=', highlight('#fff')), text(' '), text('a==b', highlight('#fff')))), '==\\=== ==a==b==\n') assert.equal(plain(paragraph(text('x'), text('y', highlight('#fff')), text(' z'))), 'xy z\n') assert.equal(plain(paragraph(text('この機能は'), text('日本語', highlight('#fff')), text('でのみ'))), 'この機能は==日本語==でのみ\n') + assert.equal(plain(paragraph(text('サーバー'), text('停止', highlight('#fff')), text('中 iPhone'), text('専用', highlight('#fff')), text('アプリ 기능은 '), text('한국어', highlight('#fff')), text('에서만'))), 'サーバー==停止==中 iPhone==専用==アプリ 기능은 ==한국어==에서만\n') assert.equal(plain(paragraph(text('日==本==語'))), '日\\==本\\==語\n') }) diff --git a/src/markdown/parse/plain-markdown-to-adf.test.ts b/src/markdown/parse/plain-markdown-to-adf.test.ts index d7b9567..322fbdb 100644 --- a/src/markdown/parse/plain-markdown-to-adf.test.ts +++ b/src/markdown/parse/plain-markdown-to-adf.test.ts @@ -149,8 +149,11 @@ test('reads a == pair to the editor default highlight, Yellow200 #f8e6a0 in @atl assert.deepEqual(read('==`a`==\n'), [paragraph(text('a', code))]) assert.deepEqual(read('x==y==z ==a == b==, (==c==) _d_==e==\n'), [paragraph(text('x==y==z '), text('a == b', highlight), text(', ('), text('c', highlight), text(') '), text('d', em), text('e', highlight))]) assert.deepEqual(read('😀==b== ==c==😀 é==d==\n'), [paragraph(text('😀'), text('b', highlight), text(' '), text('c', highlight), text('😀 é==d=='))]) - assert.deepEqual(read('この機能は==日本語==でのみ、中文==重点==内容、ภาษา==ไทย==ดี a==日本==b\n'), [ - paragraph(text('この機能は'), text('日本語', highlight), text('でのみ、中文'), text('重点', highlight), text('内容、ภาษา'), text('ไทย', highlight), text('ดี a==日本==b')), + assert.deepEqual(read('この機能は==日本語==でのみ、中文==重点==内容、ภาษา==ไทย==ดี 𠀀==𠀁==𠀂 이 기능은 ==한국어==에서만 サーバー==停止==中 このiPhone==専用==アプリ\n'), [ + paragraph( + text('この機能は'), text('日本語', highlight), text('でのみ、中文'), text('重点', highlight), text('内容、ภาษา'), text('ไทย', highlight), text('ดี 𠀀'), text('𠀁', highlight), + text('𠀂 이 기능은 '), text('한국어', highlight), text('에서만 サーバー'), text('停止', highlight), text('中 このiPhone'), text('専用', highlight), text('アプリ'), + ), ]) assert.deepEqual(read('# ==h==\n\n| ==c== |\n| --- |\n'), [ node('heading', { level: 1 }, text('h', highlight)), -- 2.52.0 From e2a660ee8e060bc09cd1a9fb8dbd76629dd21449 Mon Sep 17 00:00:00 2001 From: Lilleman auf Larv Date: Tue, 29 Sep 2026 18:49:55 +0200 Subject: [PATCH 4/5] cjk - review: Hangul, the long-vowel mark and a character inside a delimiter bound a highlight --- README.md | 2 +- docs/decisions.md | 8 ++++---- src/markdown/plain-conventions.ts | 11 ++++++----- 3 files changed, 11 insertions(+), 10 deletions(-) diff --git a/README.md b/README.md index 4982ef1..5ec7dfc 100644 --- a/README.md +++ b/README.md @@ -133,7 +133,7 @@ read replaces mentions, attachments and macros with text. | `panel` | a GitHub alert, `> [!WARNING]`: info `NOTE`, note `IMPORTANT`, tip and success `TIP`, warning `WARNING`, error `CAUTION`, custom `NOTE` | `NOTE` info, `IMPORTANT` note, `TIP` tip, `WARNING` warning, `CAUTION` error, and Obsidian's: hint tip; success, check, done success; attention warning; danger, failure, fail, missing, bug, error error; any other word info — in any case; the rest of the marker's line is the first paragraph | | `expand`, `nestedExpand` | Obsidian's folded callout, `> [!NOTE]- Title` | `-` or `+` after any word, the rest of the marker's line the title; an expand inside an expand is a `nestedExpand` | | `taskList` | `- [x] Done`, `- [ ] Todo` | a bullet list whose every item is so marked, `[X]` too | -| `backgroundColor` | `==text==` | `==text==` on one line, the text touching both delimiters, bounded outside by whitespace, punctuation, a line edge or a letter of a script written without spaces (Han, Hiragana, Katakana, Thai, Lao, Khmer, Myanmar), in the editor's default highlight `#f8e6a0` | +| `backgroundColor` | `==text==` | `==text==` on one line, the text touching both delimiters, bounded outside by whitespace, punctuation or a line edge, or touching a Han, Hangul, Hiragana, Katakana, Thai, Lao, Khmer or Myanmar character on either side, in the editor's default highlight `#f8e6a0` | | `table` | a pipe table: the first row its header, a cell's blocks on one line, a span kept under its header by empty cells | — | | `decisionList` | a bullet list | — | | `mention`, `status`, `emoji`, `date` | their text: `@` kept, a mention with none `@` and its id, an emoji its `shortName` without, a date `2026-09-13` in UTC | — | diff --git a/docs/decisions.md b/docs/decisions.md index ad4114e..123453f 100644 --- a/docs/decisions.md +++ b/docs/decisions.md @@ -151,10 +151,10 @@ renderer is the one the audience's markdown is read in. README §Plain markdown's rows come from a survey of GitHub, GitLab, Gitea, Obsidian, Pandoc, MkDocs, Docusaurus, Typora, Joplin, Logseq, Bear, Notion, Azure DevOps and Discord, GitHub's renderer confirming each shape. Reader panels settled `error` as an error panel, the `==` bounds -(3 of 3) and a letter of a script written without spaces bounding them, so `は==日本語==で` -highlights (3 of 3), the external image's two forms (6 of 7), the omission notes and a rule -opening a list item dropping (3 of 3), and a list's numbering overflowing into bullets (3 of 3, -5 of 7). +(3 of 3) and a Han, Hangul, kana, Thai, Lao, Khmer or Myanmar character on either side bounding a +delimiter, so `は==日本語==で` (3 of 3), `==한국어==에서만` and `iPhone==専用==` (6 of 7) highlight, +the external image's two forms (6 of 7), the omission notes and a rule opening a list item +dropping (3 of 3), and a list's numbering overflowing into bullets (3 of 3, 5 of 7). Reading takes other tools' spellings, since it reads their output and writes none of them. Rejected: `~sub~` and `^sup^` (`~2~` is a strike on GitHub, so `subsup` drops), underline and colour spellings, raw HTML (`
`, ``), MkDocs `!!!` and the `:::` admonition family, diff --git a/src/markdown/plain-conventions.ts b/src/markdown/plain-conventions.ts index ce62cc3..97a9d8a 100644 --- a/src/markdown/plain-conventions.ts +++ b/src/markdown/plain-conventions.ts @@ -16,6 +16,9 @@ const alertWords: Readonly> = { warning: 'WARNING', } +// ー, ー and the kana voicing marks are Script=Common; Script_Extensions would also take Latin combining marks. +const boundingScript = /^[\u3099-\u309c\u30fc\uff70\uff9e\uff9f\p{Script=Han}\p{Script=Hangul}\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Khmer}\p{Script=Lao}\p{Script=Myanmar}\p{Script=Thai}]$/u + const panelTypesByWord: Readonly> = { attention: 'warning', bug: 'error', @@ -34,8 +37,6 @@ const panelTypesByWord: Readonly> = { warning: 'warning', } -const unspacedScript = /^[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Khmer}\p{Script=Lao}\p{Script=Myanmar}\p{Script=Thai}]$/u - export function alertMarker(panelType: unknown): string { const word = typeof panelType === 'string' && Object.hasOwn(alertWords, panelType) ? alertWords[panelType] : undefined return `[!${word ?? 'NOTE'}]` @@ -62,11 +63,11 @@ export function highlightFlanking(source: string, index: number): { closes: bool const end = index + highlightDelimiter.length const before = Array.from(source.slice(Math.max(0, index - 2), index)).at(-1) ?? '' const after = Array.from(source.slice(end, end + 2))[0] ?? '' - return { closes: flanks(before) && bounds(after), opens: flanks(after) && bounds(before) } + return { closes: flanks(before) && bounds(after, before), opens: flanks(after) && bounds(before, after) } } -function bounds(character: string): boolean { - return !isWordCharacter(character) || unspacedScript.test(character) +function bounds(outside: string, inside: string): boolean { + return !isWordCharacter(outside) || boundingScript.test(outside) || boundingScript.test(inside) } function flanks(character: string): boolean { -- 2.52.0 From 160c2a7052b5efa147fa0aae1b0463382736ea3a Mon Sep 17 00:00:00 2001 From: Lilleman auf Larv Date: Tue, 29 Sep 2026 18:52:03 +0200 Subject: [PATCH 5/5] cjk - review: the voicing marks' script stated truly, one spelling for the alphabet --- src/conformance/property-harness.ts | 2 +- src/markdown/plain-conventions.ts | 4 ++-- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/src/conformance/property-harness.ts b/src/conformance/property-harness.ts index 9862945..5013bfa 100644 --- a/src/conformance/property-harness.ts +++ b/src/conformance/property-harness.ts @@ -23,7 +23,7 @@ export const propertyTimeout = 600000 const depthIdentifier = fc.createDepthIdentifier() const emptyCell: AdfNode = { content: [{ type: 'paragraph' }], type: 'tableCell' } const flatCommonMarkShapeWeight = 4 -export const markdownPieces = fc.constantFrom(...'aZ09 \t\n!"#$%&\'()*+,-./:;<=>?@[\\]^_`{|}~é\xa0🎉日ー한', '𠀀', '==', '[!NOTE]', '[x]', 'ab:', 'http://', directivePrefix, `${directivePrefix}a[`, `${directivePrefix}a{`) +export const markdownPieces = fc.constantFrom(...'aZ09 \t\n!"#$%&\'()*+,-./:;<=>?@[\\]^_`{|}~é\xa0🎉日ー한𠀀', '==', '[!NOTE]', '[x]', 'ab:', 'http://', directivePrefix, `${directivePrefix}a[`, `${directivePrefix}a{`) const nestingCommonMarkShapeWeight = 21 const spelledTypes = new Set(['text', ...Object.keys(blockNodes), ...Object.keys(inlineNodes), ...Object.keys(markAttributes)]) diff --git a/src/markdown/plain-conventions.ts b/src/markdown/plain-conventions.ts index 97a9d8a..c5d6f09 100644 --- a/src/markdown/plain-conventions.ts +++ b/src/markdown/plain-conventions.ts @@ -16,8 +16,8 @@ const alertWords: Readonly> = { warning: 'WARNING', } -// ー, ー and the kana voicing marks are Script=Common; Script_Extensions would also take Latin combining marks. -const boundingScript = /^[\u3099-\u309c\u30fc\uff70\uff9e\uff9f\p{Script=Han}\p{Script=Hangul}\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Khmer}\p{Script=Lao}\p{Script=Myanmar}\p{Script=Thai}]$/u +// ー, ー and the kana voicing marks sit outside the kana scripts; Script_Extensions would also take Latin combining marks. +const boundingScript = /^[\u3099\u309a\u30fc\uff70\uff9e\uff9f\p{Script=Han}\p{Script=Hangul}\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Khmer}\p{Script=Lao}\p{Script=Myanmar}\p{Script=Thai}]$/u const panelTypesByWord: Readonly> = { attention: 'warning', -- 2.52.0