Read the leaf blocks, ahead of any inline parsing #31
@@ -111,6 +111,6 @@ export function startsEntityReference(text: string): boolean {
|
||||
return anchoredEntityReference.test(text)
|
||||
}
|
||||
|
||||
function trimSpace(text: string): string {
|
||||
export function trimSpace(text: string): string {
|
||||
return text.replace(/^[ \t]+|[ \t]+$/g, '')
|
||||
}
|
||||
|
||||
@@ -24,6 +24,14 @@ test('keeps the link reference definitions a paragraph gives up, the first of a
|
||||
assert.deepEqual(definitions('[a]: /url\\(x\n'), [['a', { destination: '/url\\(x' }]])
|
||||
assert.deepEqual(definitions('[a]: <>\n'), [['a', { destination: '' }]])
|
||||
assert.deepEqual(definitions('[a]: /url "He said \\"hi\\""\n'), [['a', { destination: '/url', title: 'He said \\"hi\\"' }]])
|
||||
assert.deepEqual(definitions('[a]: /url\\\n[b]: /b\n'), [
|
||||
['a', { destination: '/url\\' }],
|
||||
['b', { destination: '/b' }],
|
||||
])
|
||||
assert.deepEqual(definitions('[\u00a0a]: /one\n[a]: /two\n'), [
|
||||
['\u00a0a', { destination: '/one' }],
|
||||
['a', { destination: '/two' }],
|
||||
])
|
||||
})
|
||||
|
||||
test('leaves the paragraph a line no definition spells', () => {
|
||||
|
||||
@@ -70,7 +70,6 @@ function openBlock(walk: Walk, lines: readonly string[], index: number, line: st
|
||||
return readHtmlBlock(walk, lines, index, html)
|
||||
}
|
||||
|
||||
// The document's last line ending closes its line rather than opening an empty one.
|
||||
function normalizeInput(markdown: string): string {
|
||||
return markdown
|
||||
.replace(/\r\n?/g, '\n')
|
||||
|
||||
@@ -1,11 +1,11 @@
|
||||
import { holdsControlCharacter } from '../commonmark-grammar.ts'
|
||||
import { holdsControlCharacter, isAsciiPunctuation } from '../commonmark-grammar.ts'
|
||||
|
||||
export type LinkDefinition = { destination: string; title?: string }
|
||||
|
||||
type ReadDefinition = { definition: LinkDefinition; label: string; length: number }
|
||||
type ReadValue = { length: number; value: string }
|
||||
|
||||
const bracketedDestination = /^<((?:[^\n<>\\]|\\[\s\S])*)>/
|
||||
const bracketedDestination = /^<((?:[^\n<>\\]|\\[^\n])*)>/
|
||||
const label = /^\[((?:[^[\]\\]|\\[\s\S]){1,999})\]:/
|
||||
const restOfLine = /^[ \t]*(?:\n|$)/
|
||||
const titleClosers: Readonly<Record<string, string>> = { '"': '"', "'": "'", '(': ')' }
|
||||
@@ -24,13 +24,12 @@ export function readLinkDefinitions(definitions: Map<string, LinkDefinition>, te
|
||||
function readDefinition(text: string): ReadDefinition | undefined {
|
||||
const matched = label.exec(text)
|
||||
if (matched === null) return undefined
|
||||
const raw = matched[1] ?? ''
|
||||
if (raw.trim() === '') return undefined
|
||||
const name = normalizeLabel(matched[1] ?? '')
|
||||
if (name === '') return undefined
|
||||
const afterLabel = skipSpace(text, matched[0].length)
|
||||
const destination = readDestination(text, afterLabel)
|
||||
if (destination === undefined) return undefined
|
||||
const afterDestination = afterLabel + destination.length
|
||||
const name = raw.replace(/[ \t\n]+/g, ' ').trim().toLowerCase()
|
||||
const titled = readTitledEnd(text, afterDestination)
|
||||
if (titled !== undefined) return { definition: { destination: destination.value, title: titled.value }, label: name, length: titled.length }
|
||||
const plain = endOfLine(text, afterDestination)
|
||||
@@ -38,6 +37,14 @@ function readDefinition(text: string): ReadDefinition | undefined {
|
||||
return { definition: { destination: destination.value }, label: name, length: plain }
|
||||
}
|
||||
|
||||
// CommonMark's label matching: the whitespace a label holds collapses, and its case folds.
|
||||
function normalizeLabel(raw: string): string {
|
||||
return raw
|
||||
.replace(/^[ \t\n]+|[ \t\n]+$/g, '')
|
||||
.replace(/[ \t\n]+/g, ' ')
|
||||
.toLowerCase()
|
||||
}
|
||||
|
||||
function readTitledEnd(text: string, offset: number): ReadValue | undefined {
|
||||
const afterSpace = skipSpace(text, offset)
|
||||
if (afterSpace === offset) return undefined
|
||||
@@ -56,7 +63,7 @@ function readDestination(text: string, offset: number): ReadValue | undefined {
|
||||
while (index < text.length) {
|
||||
const character = text.charAt(index)
|
||||
if (character === ' ' || holdsControlCharacter(character)) break
|
||||
if (character === '\\') {
|
||||
if (escapesNext(text, index)) {
|
||||
index += 2
|
||||
continue
|
||||
}
|
||||
@@ -77,7 +84,7 @@ function readTitle(text: string, offset: number): ReadValue | undefined {
|
||||
let index = offset + 1
|
||||
while (index < text.length) {
|
||||
const character = text.charAt(index)
|
||||
if (character === '\\') {
|
||||
if (escapesNext(text, index)) {
|
||||
index += 2
|
||||
continue
|
||||
}
|
||||
@@ -88,6 +95,11 @@ function readTitle(text: string, offset: number): ReadValue | undefined {
|
||||
return undefined
|
||||
}
|
||||
|
||||
// A backslash escapes ASCII punctuation only, so a line ending always ends the destination it follows.
|
||||
function escapesNext(text: string, index: number): boolean {
|
||||
return text.charAt(index) === '\\' && isAsciiPunctuation(text.charAt(index + 1))
|
||||
}
|
||||
|
||||
// The label, the destination and the title each take at most one line ending with them.
|
||||
function skipSpace(text: string, offset: number): number {
|
||||
const rest = text.slice(offset)
|
||||
|
||||
@@ -129,8 +129,14 @@ test('gives up the link reference definitions a paragraph opens with', () => {
|
||||
assert.deepEqual(content(markdownToAdf('[a]: /url\n===\n')), [paragraph('===')])
|
||||
})
|
||||
|
||||
test('keeps the whitespace CommonMark strips no more of than a space or a tab', () => {
|
||||
assert.deepEqual(content(markdownToAdf('\u00a0Part.\u00a0\n')), [paragraph('\u00a0Part.\u00a0')])
|
||||
assert.deepEqual(content(markdownToAdf(' \u3000Part.\t\n')), [paragraph('\u3000Part.')])
|
||||
})
|
||||
|
||||
test('normalizes the line endings and the null character CommonMark replaces', () => {
|
||||
assert.deepEqual(content(markdownToAdf('One\r\ntwo.\r\n')), [paragraph('One two.')])
|
||||
assert.deepEqual(content(markdownToAdf('One\rtwo.\r')), [paragraph('One two.')])
|
||||
assert.deepEqual(content(markdownToAdf('```\r\nx\r\n```\r\n')), [{ content: [text('x')], type: 'codeBlock' }])
|
||||
assert.deepEqual(content(markdownToAdf('a\u0000b\n')), [paragraph('a\ufffdb')])
|
||||
})
|
||||
|
||||
@@ -2,6 +2,7 @@ import type { AdfDocument, AdfNode } from '../../adf/document.ts'
|
||||
import type { ClaimedConstruct, LeafBlock } from './blocks.ts'
|
||||
import { failure, success, type ConvertErrorPath, type Result } from '../../result.ts'
|
||||
import { parseBlocks } from './blocks.ts'
|
||||
import { trimSpace } from '../commonmark-grammar.ts'
|
||||
|
||||
export function markdownToAdf(markdown: string): Result<AdfDocument> {
|
||||
const content: AdfNode[] = []
|
||||
@@ -40,7 +41,7 @@ function withContent(node: AdfNode, text: string): AdfNode {
|
||||
function inlineContent(text: string): AdfNode[] {
|
||||
const line = text
|
||||
.split('\n')
|
||||
.map((part) => part.trim())
|
||||
.map((part) => trimSpace(part))
|
||||
.join(' ')
|
||||
return line === '' ? [] : [{ text: line, type: 'text' }]
|
||||
}
|
||||
|
||||
@@ -213,7 +213,8 @@ detail is settled at its own milestone.
|
||||
repeated being two bytes a level, so this is the cheapest way to reach §11's 500. 3b's leaf
|
||||
readers scan the physical line themselves, so a container re-cuts the walk rather than adding
|
||||
to it: the open containers' prefix comes off the line first and the readers take one line at a
|
||||
time, `LeafBlock` renamed with the union they join.
|
||||
time, `LeafBlock` renamed with the union they join and `blockNode`'s chain gaining their
|
||||
branches.
|
||||
**Settled** (the maintainer, 2026-08-27): a claimed line ends lazy continuation, so a
|
||||
closing fence on the line after a blockquote's open paragraph closes its container instead
|
||||
of continuing the paragraph CommonMark would fold it into. Claiming at block level is
|
||||
@@ -225,13 +226,13 @@ detail is settled at its own milestone.
|
||||
into the separation it names, and `spec/flavour.md`'s "none between a nested list and a
|
||||
CommonMark block above it" gaining that exception. Every fixture spelled tight today keeps
|
||||
its bytes, and `nested-list-tight` becomes a round-trip pair.
|
||||
- [ ] **3d — Inline text.** The inline scanner over a block's content: backslash escapes,
|
||||
entity references decoding to their characters, code spans and the literal they hold —
|
||||
directive syntax and `~~` included — CommonMark's own hard breaks, a trailing backslash
|
||||
and two trailing spaces alike, a soft line break as one space, and the raw inline tag,
|
||||
comment and processing instruction refused by name, recognized by the
|
||||
`commonmark-grammar.ts` predicates the emitter already escapes against, under 3b's
|
||||
one-table rule.
|
||||
- [ ] **3d — Inline text.** The inline scanner over a block's content: backslash escapes, entity
|
||||
references decoding to their characters, code spans and the literal they hold — directive
|
||||
syntax and `~~` included — CommonMark's own hard breaks, a trailing backslash and two
|
||||
trailing spaces alike, a soft line break as one space, the fenced info string's own decoding
|
||||
the block walk leaves raw, and the raw inline tag, comment and processing instruction
|
||||
refused by name, recognized by the `commonmark-grammar.ts` predicates the emitter already
|
||||
escapes against, under 3b's one-table rule.
|
||||
- [ ] **3e — Emphasis and links.** `_`, `*` and `~~` runs through `matchEmphasis` to the `em`,
|
||||
`strong` and `strike` marks; links inline and reference, 3b's definitions resolved here,
|
||||
autolinks, and the image gap's named errors — a titled image, and one amid other text.
|
||||
|
||||
Reference in New Issue
Block a user