From 43869a7219a85d7f085a112c0738fd097b949987 Mon Sep 17 00:00:00 2001 From: Lilleman auf Larv Date: Tue, 29 Sep 2026 12:05:04 +0200 Subject: [PATCH] 35b - review: drop every unread highlight in one pass, one task-as-text path, a linear title trim --- README.md | 2 +- src/markdown/emit/adf-to-markdown.ts | 5 ++- src/markdown/emit/inline-line.ts | 5 ++- src/markdown/emit/line-escaping.ts | 26 +++++++------ src/markdown/emit/plain-inline.ts | 19 +++++++--- src/markdown/emit/plain-reduction.ts | 57 +++++++++++++--------------- 6 files changed, 61 insertions(+), 53 deletions(-) diff --git a/README.md b/README.md index e2e65bb..ca48cd6 100644 --- a/README.md +++ b/README.md @@ -147,7 +147,7 @@ read replaces mentions, attachments and macros with text. `_(image not included)_`, `_(jira-issues-table not included)_`, `_(synced block not included)_`, `_(link card not included)_`, `_(extension not included)_`. - `backgroundColor`, `code`, `em`, `link`, `strike` and `strong` stay; every other mark drops, - keeping its text, and so does a mark CommonMark cannot spell where it stands. + keeping its text, and so does a mark the flavour cannot spell where it stands. - A newline in text is a hard break and in an expand's title a space, edge whitespace outside a link or code span is trimmed, carriage returns and null characters are removed, and an empty paragraph drops. diff --git a/src/markdown/emit/adf-to-markdown.ts b/src/markdown/emit/adf-to-markdown.ts index 0603a49..459142d 100644 --- a/src/markdown/emit/adf-to-markdown.ts +++ b/src/markdown/emit/adf-to-markdown.ts @@ -162,7 +162,7 @@ function quoted(text: string): string { .join('\n') } -function tryTaskList(node: AdfNode, path: ConvertErrorPath, depth: number, writing: Writing): Result | undefined { +function tryTaskList(node: AdfNode, path: ConvertErrorPath, depth: number, writing: Writing): Result { const items: PlacedBlock[][] = [] let headroom = largestNesting - depth - 1 // A child other than a task nests in the task before it. @@ -176,7 +176,8 @@ function tryTaskList(node: AdfNode, path: ConvertErrorPath, depth: number, writi else for (const block of walk.value.blocks) previous.push(block) } const lines = items.map((blocks) => tryListItemLines(joinBlocks(blocks, 'list-item'), '- ')) - return lines.includes(undefined) ? undefined : success({ headroom, spelling: 'list', text: lines.join('\n') }) + // The plain reduction leaves no task a list item cannot hold: a directive here would break the flavour. + return lines.includes(undefined) ? failure('unsupported-node-shape', 'a task holds blocks no list item spells', path) : success({ headroom, spelling: 'list', text: lines.join('\n') }) } function placedBlock(node: AdfNode, path: ConvertErrorPath, depth: number, writing: Writing): Result { diff --git a/src/markdown/emit/inline-line.ts b/src/markdown/emit/inline-line.ts index 7ff708c..265da48 100644 --- a/src/markdown/emit/inline-line.ts +++ b/src/markdown/emit/inline-line.ts @@ -38,7 +38,7 @@ type LineAttempt = { fallback: NodeRange | 'opening-link'; line?: undefined } | type LineFallbacks = { carried: Set; flavour: Flavour; openingLinkAsDirective: boolean } -export type PlainLineFallback = { kind: 'claimed-line'; line: number; text: string } | { kind: 'opening-link' } | { kind: 'unspellable-run'; run: MarkRun } +export type PlainLineFallback = { kind: 'claimed-line'; line: number; text: string } | { kind: 'opening-link' } | { kind: 'unspellable-run'; run: MarkRun; runs: MarkRun[] } export function emitInlineLine(nodes: readonly AdfNode[], container: LineContainer, path: ConvertErrorPath, flavour: Flavour): Result { const emitted = emitLine(nodes, container, path, flavour) @@ -129,7 +129,8 @@ function attemptLine(segments: readonly InlineSegment[], container: LineContaine function lineVerdict(segments: readonly InlineSegment[], container: LineContainer, flavour: Flavour): PlainLineFallback | { kind: 'line'; text: string } { const assembled = assembleInlineLine(segments, container, flavour) if (assembled.openingLinkAsDirective) return { kind: 'opening-link' } - if (assembled.unspellableRun !== undefined) return { kind: 'unspellable-run', run: assembled.unspellableRun } + const [run] = assembled.unspellableRuns + if (run !== undefined) return { kind: 'unspellable-run', run, runs: assembled.unspellableRuns } const lines = assembled.line.split('\n') const claimed = container === 'paragraph' ? lines.findIndex((single, index) => claimsLine(single, index === 0 ? 'first' : 'later')) : -1 return claimed === -1 ? { kind: 'line', text: assembled.line } : { kind: 'claimed-line', line: claimed, text: lines[claimed] ?? '' } diff --git a/src/markdown/emit/line-escaping.ts b/src/markdown/emit/line-escaping.ts index 16098a6..253df18 100644 --- a/src/markdown/emit/line-escaping.ts +++ b/src/markdown/emit/line-escaping.ts @@ -23,7 +23,7 @@ export type InlineSegment = | { emphasis?: undefined; escaping: 'none'; highlight?: undefined; nodes: NodeRange; text: string } | { emphasis?: undefined; escaping: InlineEscaping; highlight?: undefined; nodes?: undefined; text: string } -export type AssembledLine = { line: string; openingLinkAsDirective?: true; unspellableRun: MarkRun | undefined } +export type AssembledLine = { line: string; openingLinkAsDirective?: true; unspellableRuns: MarkRun[] } type ScanLine = { position: LinePosition; start: number; text: string } @@ -81,10 +81,10 @@ function escape(segments: readonly InlineSegment[], container: LineContainer, hi output += scan.charAt(index) } if (container === 'paragraph' && opensLinkDefinition(output)) { - if (segments[0]?.nodes !== undefined) return { line: output, openingLinkAsDirective: true, unspellableRun: undefined } - return { line: `\\${output}`, unspellableRun: unspellableRun(segments, output, placements) } + if (segments[0]?.nodes !== undefined) return { line: output, openingLinkAsDirective: true, unspellableRuns: [] } + return { line: `\\${output}`, unspellableRuns: unspellableRuns(segments, output, placements) } } - return { line: output, unspellableRun: unspellableRun(segments, output, placements) } + return { line: output, unspellableRuns: unspellableRuns(segments, output, placements) } } function escapeClaims(scan: string, escapings: readonly InlineEscaping[], container: LineContainer, highlights: boolean): ReadonlySet { @@ -148,25 +148,27 @@ function escapeClosedRuns(scan: string, escapings: readonly InlineEscaping[], cl return escaped } -function unspellableRun(segments: readonly InlineSegment[], output: string, placements: readonly number[]): MarkRun | undefined { +// One emphasis run, the innermost, or every highlight run the line cannot spell. +function unspellableRuns(segments: readonly InlineSegment[], output: string, placements: readonly number[]): MarkRun[] { const { nodes, runs } = emittedRuns(segments, placements, output) const pair = misflanked(runs) ?? unpaired(runs) - return pair === undefined ? unreadHighlight(segments, output, placements) : nodes[pair] + const run = pair === undefined ? undefined : nodes[pair] + return run === undefined ? unreadHighlights(segments, output, placements) : [run] } -// No `==` in text can open or close, and highlights never nest, so a pair reads back where each delimiter flanks and both share a line. -function unreadHighlight(segments: readonly InlineSegment[], output: string, placements: readonly number[]): MarkRun | undefined { +// No `==` in text can open or close, and highlights never nest, so a pair reads back where each delimiter flanks. +function unreadHighlights(segments: readonly InlineSegment[], output: string, placements: readonly number[]): MarkRun[] { + const unread: MarkRun[] = [] let cursor = 0 - let opener = 0 for (const segment of segments) { const start = placements[cursor] ?? 0 cursor += segment.text.length if (segment.highlight === undefined) continue const flanking = highlightFlanking(output, start) - if (segment.highlight === 'open') opener = start - if (segment.highlight === 'open' ? !flanking.opens : !flanking.closes || output.slice(opener, start).includes('\n')) return segment.nodes + const flanks = segment.highlight === 'open' ? flanking.opens : flanking.closes + if (!flanks && unread.at(-1) !== segment.nodes) unread.push(segment.nodes) } - return undefined + return unread } function misflanked(runs: readonly EmittedRun[]): number | undefined { diff --git a/src/markdown/emit/plain-inline.ts b/src/markdown/emit/plain-inline.ts index 9fb8bf2..7da6adc 100644 --- a/src/markdown/emit/plain-inline.ts +++ b/src/markdown/emit/plain-inline.ts @@ -1,5 +1,6 @@ import type { AdfAttributes, AdfMark, AdfNode } from '../../adf/document.ts' import type { LineContainer } from '../line-container.ts' +import type { MarkRun } from './line-escaping.ts' import { blockNodeModel } from '../../adf/block-nodes.ts' import { failure, success, type ConvertErrorPath, type Result } from '../../result.ts' import { largestNesting } from '../../nesting.ts' @@ -130,7 +131,7 @@ function textLeaf(text: string, marks: readonly AdfMark[]): AdfNode { return marks.length === 0 ? { text, type: 'text' } : { marks: [...marks], text, type: 'text' } } -// A highlight goes outermost so its run is one run at depth 0, and code innermost, the only place its spelling holds. +// A highlight goes first, where `highlighted` looks for it, and code last, the only place its spelling holds. function plainMarks(marks: readonly AdfMark[], container: LineContainer, text: string): AdfMark[] { const kept: AdfMark[] = [] for (const mark of marks) { @@ -174,7 +175,7 @@ function highlighted(leaves: readonly AdfNode[]): AdfNode[] { const marks = nodeMarks(leaf) if (marks[0]?.type === highlight) { const held = marks.slice(1) - shared = run.length === 0 ? held.filter((mark) => mark.type !== 'code') : shared.filter((mark) => held.some((other) => sameMark(other, mark))) + shared = run.length === 0 ? held : shared.filter((mark) => held.some((other) => sameMark(other, mark))) run.push(leaf) continue } @@ -265,7 +266,7 @@ function spellableLine(leaves: AdfNode[], container: LineContainer, path: Conver } function withoutFallback(leaves: readonly AdfNode[], fallback: PlainLineFallback): AdfNode[] | undefined { - if (fallback.kind === 'unspellable-run') return withoutMark(leaves, fallback.run.first, fallback.run.last, fallback.run.depth) + if (fallback.kind === 'unspellable-run') return withoutMarks(leaves, fallback.runs) const first = fallback.kind === 'opening-link' ? 0 : lineStart(leaves, fallback.line) const mark = nodeMarks(leaves[first] ?? {})[0] if (mark === undefined || mark.type !== (fallback.kind === 'opening-link' ? 'link' : 'code')) return undefined @@ -274,7 +275,7 @@ function withoutFallback(leaves: readonly AdfNode[], fallback: PlainLineFallback // A code span is what binds the `]` a link definition reads, and dropping it keeps the link target. const spans = leaves.slice(first, last + 1).some((leaf) => nodeMarks(leaf).length > 1 && nodeMarks(leaf).at(-1)?.type === 'code') if (mark.type === 'link' && spans) return leaves.map((leaf, index) => (index < first || index > last ? leaf : withMarks(leaf, nodeMarks(leaf).filter((held) => held.type !== 'code')))) - return withoutMark(leaves, first, last, 0) + return withoutMarks(leaves, [{ depth: 0, first, last }]) } function lineStart(leaves: readonly AdfNode[], line: number): number { @@ -283,6 +284,12 @@ function lineStart(leaves: readonly AdfNode[], line: number): number { return index } -function withoutMark(leaves: readonly AdfNode[], first: number, last: number, depth: number): AdfNode[] { - return leaves.map((leaf, index) => (index < first || index > last ? leaf : withMarks(leaf, nodeMarks(leaf).filter((_, held) => held !== depth)))) +// The runs cover disjoint leaves, so one pass drops them all. +function withoutMarks(leaves: readonly AdfNode[], runs: readonly MarkRun[]): AdfNode[] { + const depths = new Map() + for (const run of runs) for (let index = run.first; index <= run.last; index += 1) depths.set(index, run.depth) + return leaves.map((leaf, index) => { + const depth = depths.get(index) + return depth === undefined ? leaf : withMarks(leaf, nodeMarks(leaf).filter((_, held) => held !== depth)) + }) } diff --git a/src/markdown/emit/plain-reduction.ts b/src/markdown/emit/plain-reduction.ts index b360934..72af133 100644 --- a/src/markdown/emit/plain-reduction.ts +++ b/src/markdown/emit/plain-reduction.ts @@ -164,18 +164,24 @@ function joinedLists(first: AdfNode, second: AdfNode): AdfNode { return { ...head, content: [...nodeContent(head), ...nodeContent(tail)] } } +// A task keeps its marker as text; a list item stands as one, and anything else nests in the item before it. function tasksAsText(list: AdfNode): AdfNode { if (list.type !== 'taskList') return list const items: AdfNode[] = [] for (const child of nodeContent(list)) { - const marker = taskMarker(nodeAttrs(child)['state']) - const previous = child.type === 'taskItem' || child.type === 'blockTaskItem' ? undefined : items.pop() - if (previous !== undefined) items.push(itemOf([...nodeContent(previous), child])) - else items.push(itemOf(child.type === 'taskItem' ? [paragraph(nodeContent(child).length === 0 ? [text(marker)] : [text(`${marker} `), ...nodeContent(child)])] : marked(nodeContent(child), marker))) + const previous = isTask(child) || child.type === 'listItem' ? undefined : items.pop() + items.push(itemOf(previous === undefined ? taskAsText(child) : mergedLists([...nodeContent(previous), child]))) } return { content: items, type: 'bulletList' } } +function taskAsText(child: AdfNode): readonly AdfNode[] { + const marker = taskMarker(nodeAttrs(child)['state']) + if (child.type === 'taskItem') return [paragraph(nodeContent(child).length === 0 ? [text(marker)] : [text(`${marker} `), ...nodeContent(child)])] + if (child.type === 'blockTaskItem') return marked(nodeContent(child), marker) + return child.type === 'listItem' ? nodeContent(child) : [child] +} + function paragraph(content: readonly AdfNode[]): AdfNode { return { content: [...content], type: 'paragraph' } } @@ -209,10 +215,17 @@ function reducePanel(node: AdfNode, reduction: Reduction): Result { function reduceExpand(node: AdfNode, reduction: Reduction): Result { const held = nodeAttrs(node)['title'] - const title = typeof held === 'string' ? oneLine(held).replace(/^[ \t]+|[ \t]+$/g, '') : '' + const title = typeof held === 'string' ? withoutTrailingBlanks(oneLine(held).replace(/^[ \t]+/, '')) : '' return contained(title === '' ? { type: node.type } : { attrs: { title }, type: node.type }, node, reduction) } +// A backward scan: an unanchored-end regex retries from every blank in a long run. +function withoutTrailingBlanks(text: string): string { + let end = text.length + while (end > 0 && (text.charAt(end - 1) === ' ' || text.charAt(end - 1) === '\t')) end -= 1 + return text.slice(0, end) +} + function reduceHeading(node: AdfNode, reduction: Reduction): Result { const level = nodeAttrs(node)['level'] if (typeof level !== 'number' || !Number.isInteger(level) || level < 1 || level > 6) return paragraphOf(nodeContent(node), reduction) @@ -261,16 +274,21 @@ function blankedLines(code: readonly AdfNode[]): AdfNode[] { return blanked === '' ? [] : [text(blanked)] } -// A task list opening with a task and holding tasks and task lists alone keeps its spelling; a list nests in the task before it. +// A task list opening with a task and holding tasks and task lists alone keeps its spelling, a list nesting in the task before it; any other keeps its markers as text. function reduceTaskList(node: AdfNode, reduction: Reduction): Result { const children = nodeContent(node) - if (!isTask(children[0]) || children.some((child) => !isTask(child) && child.type !== 'taskList')) return reduceTasksAsText(node, reduction) + const regular = isTask(children[0]) && children.every((child) => isTask(child) || child.type === 'taskList') const tasks: AdfNode[] = [] let nested: AdfNode[] = [] for (const [index, child] of children.entries()) { const at = childReduction(reduction, index) const reduced = isTask(child) ? reduceTask(child, at) : reduceStanding(child, at) if (!reduced.ok) return reduced + if (!regular) { + const standsAlone = !isTask(child) && child.type !== 'taskList' + for (const block of standsAlone ? [{ content: reduced.value, type: 'listItem' }] : reduced.value) tasks.push(block) + continue + } if (isTask(child)) { nestIn(tasks, nested) nested = [] @@ -278,7 +296,8 @@ function reduceTaskList(node: AdfNode, reduction: Reduction): Result for (const block of reduced.value) (isTask(child) ? tasks : nested).push(block) } nestIn(tasks, nested) - return success([{ content: tasks, type: 'taskList' }]) + if (regular) return success([{ content: tasks, type: 'taskList' }]) + return success(listOf(nodeContent(tasksAsText({ content: tasks, type: 'taskList' })), 'bulletList')) } // The writer nests a list in the task before it, so one closing a block task item's blocks merges with it. @@ -302,28 +321,6 @@ function reduceTask(task: AdfNode, at: Reduction): Result { return blocks.ok ? success([{ attrs, content: blankedCode(blocks.value), type: 'blockTaskItem' }]) : blocks } -function reduceTasksAsText(node: AdfNode, reduction: Reduction): Result { - const items: AdfNode[] = [] - for (const [index, child] of nodeContent(node).entries()) { - const blocks = taskBlocks(child, childReduction(reduction, index)) - if (!blocks.ok) return blocks - const previous = child.type === 'taskList' ? items.pop() : undefined - items.push(itemOf(previous === undefined ? blocks.value : mergedLists([...nodeContent(previous), ...blocks.value]))) - } - return success(listOf(items, 'bulletList')) -} - -function taskBlocks(child: AdfNode, at: Reduction): Result { - const marker = taskMarker(nodeAttrs(child)['state']) - if (child.type === 'taskItem') { - const content = reduceInline(nodeContent(child), 'paragraph', at.path, at.depth) - return content.ok ? success([paragraph(content.value.length === 0 ? [text(marker)] : [text(`${marker} `), ...content.value])]) : content - } - if (child.type !== 'blockTaskItem') return reduceStanding(child, at) - const blocks = reduceBlocks(nodeContent(child), at) - return blocks.ok ? success(marked(blocks.value, marker)) : blocks -} - // The marker leads the first paragraph, or stands as one where the blocks open with another. function marked(blocks: readonly AdfNode[], marker: string): AdfNode[] { const [first, ...rest] = blocks