Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
43 changes: 35 additions & 8 deletions plugin/scripts/lib/md-ast.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -233,33 +233,41 @@ function matchListItem(text, p, col) {
}

// Split a GFM table row into cells honoring backslash escapes.
// Returns [{ text, offset }] — offset = char index of the cell's first
// Returns [{ text, offset, gaps }] — offset = char index of the cell's first
// (untrimmed) char within `text`; `hadPipe` = an unescaped | was seen.
// An escaped pipe `\|` is cell CONTENT and loses its backslash here, before
// inline parsing, so it is unescaped inside a code span too (GFM Tables
// extension, cmark-gfm unescape_pipes). `gaps` = indices in the cell's text
// where one source char (that backslash) was dropped, so makeRow can keep
// source positions honest.
function splitRow(text) {
const cells = [];
let cur = '';
let gaps = [];
let cellStart = 0;
let sawPipe = false;
for (let i = 0; i < text.length; i++) {
const c = text[i];
if (c === '\\' && text[i + 1] === '|') { gaps.push(cur.length); cur += '|'; i++; continue; }
if (c === '\\' && i + 1 < text.length) { cur += c + text[i + 1]; i++; continue; }
if (c === '|') {
sawPipe = true;
cells.push({ text: cur, offset: cellStart });
cells.push({ text: cur, offset: cellStart, gaps });
cur = '';
gaps = [];
cellStart = i + 1;
continue;
}
cur += c;
}
cells.push({ text: cur, offset: cellStart });
cells.push({ text: cur, offset: cellStart, gaps });
// Boundary pipes: a leading | and a trailing | delimit, they don't add cells.
if (cells.length && cells[0].text.trim() === '' && text.trimStart().startsWith('|')) cells.shift();
if (cells.length && cells[cells.length - 1].text.trim() === '' && text.trimEnd().endsWith('|') && !text.trimEnd().endsWith('\\|')) cells.pop();
// Trim each cell, keeping source offsets honest.
const out = cells.map((c) => {
const lead = c.text.length - c.text.trimStart().length;
return { text: c.text.trim(), offset: c.offset + lead };
return { text: c.text.trim(), offset: c.offset + lead, gaps: c.gaps.map((g) => g - lead) };
});
return { cells: out, hadPipe: sawPipe };
}
Expand Down Expand Up @@ -724,16 +732,31 @@ export function parseMarkdown(src) {

function makeRow(cells, rowSrcBase) {
// rowSrcBase = source offset of the row text's char 0 (cells carry offsets
// relative to the SPLIT string, which started at the row's first char)
// relative to the SPLIT string, which started at the row's first char).
// Each unescaped pipe (c.gaps) starts a new seg one source char further on.
const srcLen = (c) => c.text.length + c.gaps.length;
const cellSegs = (c) => {
const segs = [];
let v = 0;
let src = rowSrcBase + c.offset;
for (const g of c.gaps) {
segs.push({ v, src, len: g - v });
src += g - v + 1;
v = g;
}
segs.push({ v, src, len: c.text.length - v });
return segs;
};
const last = cells[cells.length - 1];
return {
type: 'tableRow',
children: cells.map((c) => ({
type: 'tableCell',
children: [],
_raw: { text: c.text, segs: [{ v: 0, src: rowSrcBase + c.offset, len: c.text.length }] },
position: { start: pt(rowSrcBase + c.offset), end: pt(rowSrcBase + c.offset + c.text.length) },
_raw: { text: c.text, segs: cellSegs(c) },
position: { start: pt(rowSrcBase + c.offset), end: pt(rowSrcBase + c.offset + srcLen(c)) },
})),
position: { start: pt(rowSrcBase), end: pt(rowSrcBase + (cells.length ? cells[cells.length - 1].offset + cells[cells.length - 1].text.length : 0)) },
position: { start: pt(rowSrcBase), end: pt(rowSrcBase + (cells.length ? last.offset + srcLen(last) : 0)) },
};
}

Expand Down Expand Up @@ -833,6 +856,10 @@ function parseInlines(raw, definitions, pt) {
else hi = mid - 1;
}
const seg = segs[lo];
// an end exactly where a contiguous seg starts (a table cell's dropped
// backslash, see makeRow) belongs to the previous seg, not past the gap
const prev = segs[lo - 1];
if (end && prev && v === seg.v && prev.v + prev.len === v) return prev.src + prev.len;
const within = Math.min(Math.max(v - seg.v, 0), seg.len);
if (end && v - seg.v > seg.len) return seg.src + seg.len; // in the virtual '\n' gap
return seg.src + within;
Expand Down
43 changes: 35 additions & 8 deletions plugin/skills/doc-structure/lib/md-ast.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -233,33 +233,41 @@ function matchListItem(text, p, col) {
}

// Split a GFM table row into cells honoring backslash escapes.
// Returns [{ text, offset }] — offset = char index of the cell's first
// Returns [{ text, offset, gaps }] — offset = char index of the cell's first
// (untrimmed) char within `text`; `hadPipe` = an unescaped | was seen.
// An escaped pipe `\|` is cell CONTENT and loses its backslash here, before
// inline parsing, so it is unescaped inside a code span too (GFM Tables
// extension, cmark-gfm unescape_pipes). `gaps` = indices in the cell's text
// where one source char (that backslash) was dropped, so makeRow can keep
// source positions honest.
function splitRow(text) {
const cells = [];
let cur = '';
let gaps = [];
let cellStart = 0;
let sawPipe = false;
for (let i = 0; i < text.length; i++) {
const c = text[i];
if (c === '\\' && text[i + 1] === '|') { gaps.push(cur.length); cur += '|'; i++; continue; }
if (c === '\\' && i + 1 < text.length) { cur += c + text[i + 1]; i++; continue; }
if (c === '|') {
sawPipe = true;
cells.push({ text: cur, offset: cellStart });
cells.push({ text: cur, offset: cellStart, gaps });
cur = '';
gaps = [];
cellStart = i + 1;
continue;
}
cur += c;
}
cells.push({ text: cur, offset: cellStart });
cells.push({ text: cur, offset: cellStart, gaps });
// Boundary pipes: a leading | and a trailing | delimit, they don't add cells.
if (cells.length && cells[0].text.trim() === '' && text.trimStart().startsWith('|')) cells.shift();
if (cells.length && cells[cells.length - 1].text.trim() === '' && text.trimEnd().endsWith('|') && !text.trimEnd().endsWith('\\|')) cells.pop();
// Trim each cell, keeping source offsets honest.
const out = cells.map((c) => {
const lead = c.text.length - c.text.trimStart().length;
return { text: c.text.trim(), offset: c.offset + lead };
return { text: c.text.trim(), offset: c.offset + lead, gaps: c.gaps.map((g) => g - lead) };
});
return { cells: out, hadPipe: sawPipe };
}
Expand Down Expand Up @@ -724,16 +732,31 @@ export function parseMarkdown(src) {

function makeRow(cells, rowSrcBase) {
// rowSrcBase = source offset of the row text's char 0 (cells carry offsets
// relative to the SPLIT string, which started at the row's first char)
// relative to the SPLIT string, which started at the row's first char).
// Each unescaped pipe (c.gaps) starts a new seg one source char further on.
const srcLen = (c) => c.text.length + c.gaps.length;
const cellSegs = (c) => {
const segs = [];
let v = 0;
let src = rowSrcBase + c.offset;
for (const g of c.gaps) {
segs.push({ v, src, len: g - v });
src += g - v + 1;
v = g;
}
segs.push({ v, src, len: c.text.length - v });
return segs;
};
const last = cells[cells.length - 1];
return {
type: 'tableRow',
children: cells.map((c) => ({
type: 'tableCell',
children: [],
_raw: { text: c.text, segs: [{ v: 0, src: rowSrcBase + c.offset, len: c.text.length }] },
position: { start: pt(rowSrcBase + c.offset), end: pt(rowSrcBase + c.offset + c.text.length) },
_raw: { text: c.text, segs: cellSegs(c) },
position: { start: pt(rowSrcBase + c.offset), end: pt(rowSrcBase + c.offset + srcLen(c)) },
})),
position: { start: pt(rowSrcBase), end: pt(rowSrcBase + (cells.length ? cells[cells.length - 1].offset + cells[cells.length - 1].text.length : 0)) },
position: { start: pt(rowSrcBase), end: pt(rowSrcBase + (cells.length ? last.offset + srcLen(last) : 0)) },
};
}

Expand Down Expand Up @@ -833,6 +856,10 @@ function parseInlines(raw, definitions, pt) {
else hi = mid - 1;
}
const seg = segs[lo];
// an end exactly where a contiguous seg starts (a table cell's dropped
// backslash, see makeRow) belongs to the previous seg, not past the gap
const prev = segs[lo - 1];
if (end && prev && v === seg.v && prev.v + prev.len === v) return prev.src + prev.len;
const within = Math.min(Math.max(v - seg.v, 0), seg.len);
if (end && v - seg.v > seg.len) return seg.src + seg.len; // in the virtual '\n' gap
return seg.src + within;
Expand Down
6 changes: 6 additions & 0 deletions scripts/fixtures/table-escaped-pipe.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,6 @@
# Table with escaped pipes

| Form | Cell | Note |
| --- | --- | --- |
| code span | `a\|b` | escaped pipe inside a code span |
| plain | a\|b | escaped pipe outside a code span |
43 changes: 35 additions & 8 deletions scripts/lib/md-ast.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -233,33 +233,41 @@ function matchListItem(text, p, col) {
}

// Split a GFM table row into cells honoring backslash escapes.
// Returns [{ text, offset }] — offset = char index of the cell's first
// Returns [{ text, offset, gaps }] — offset = char index of the cell's first
// (untrimmed) char within `text`; `hadPipe` = an unescaped | was seen.
// An escaped pipe `\|` is cell CONTENT and loses its backslash here, before
// inline parsing, so it is unescaped inside a code span too (GFM Tables
// extension, cmark-gfm unescape_pipes). `gaps` = indices in the cell's text
// where one source char (that backslash) was dropped, so makeRow can keep
// source positions honest.
function splitRow(text) {
const cells = [];
let cur = '';
let gaps = [];
let cellStart = 0;
let sawPipe = false;
for (let i = 0; i < text.length; i++) {
const c = text[i];
if (c === '\\' && text[i + 1] === '|') { gaps.push(cur.length); cur += '|'; i++; continue; }
if (c === '\\' && i + 1 < text.length) { cur += c + text[i + 1]; i++; continue; }
if (c === '|') {
sawPipe = true;
cells.push({ text: cur, offset: cellStart });
cells.push({ text: cur, offset: cellStart, gaps });
cur = '';
gaps = [];
cellStart = i + 1;
continue;
}
cur += c;
}
cells.push({ text: cur, offset: cellStart });
cells.push({ text: cur, offset: cellStart, gaps });
// Boundary pipes: a leading | and a trailing | delimit, they don't add cells.
if (cells.length && cells[0].text.trim() === '' && text.trimStart().startsWith('|')) cells.shift();
if (cells.length && cells[cells.length - 1].text.trim() === '' && text.trimEnd().endsWith('|') && !text.trimEnd().endsWith('\\|')) cells.pop();
// Trim each cell, keeping source offsets honest.
const out = cells.map((c) => {
const lead = c.text.length - c.text.trimStart().length;
return { text: c.text.trim(), offset: c.offset + lead };
return { text: c.text.trim(), offset: c.offset + lead, gaps: c.gaps.map((g) => g - lead) };
});
return { cells: out, hadPipe: sawPipe };
}
Expand Down Expand Up @@ -724,16 +732,31 @@ export function parseMarkdown(src) {

function makeRow(cells, rowSrcBase) {
// rowSrcBase = source offset of the row text's char 0 (cells carry offsets
// relative to the SPLIT string, which started at the row's first char)
// relative to the SPLIT string, which started at the row's first char).
// Each unescaped pipe (c.gaps) starts a new seg one source char further on.
const srcLen = (c) => c.text.length + c.gaps.length;
const cellSegs = (c) => {
const segs = [];
let v = 0;
let src = rowSrcBase + c.offset;
for (const g of c.gaps) {
segs.push({ v, src, len: g - v });
Comment thread
coderabbitai[bot] marked this conversation as resolved.
src += g - v + 1;
v = g;
}
segs.push({ v, src, len: c.text.length - v });
return segs;
};
const last = cells[cells.length - 1];
return {
type: 'tableRow',
children: cells.map((c) => ({
type: 'tableCell',
children: [],
_raw: { text: c.text, segs: [{ v: 0, src: rowSrcBase + c.offset, len: c.text.length }] },
position: { start: pt(rowSrcBase + c.offset), end: pt(rowSrcBase + c.offset + c.text.length) },
_raw: { text: c.text, segs: cellSegs(c) },
position: { start: pt(rowSrcBase + c.offset), end: pt(rowSrcBase + c.offset + srcLen(c)) },
})),
position: { start: pt(rowSrcBase), end: pt(rowSrcBase + (cells.length ? cells[cells.length - 1].offset + cells[cells.length - 1].text.length : 0)) },
position: { start: pt(rowSrcBase), end: pt(rowSrcBase + (cells.length ? last.offset + srcLen(last) : 0)) },
};
}

Expand Down Expand Up @@ -833,6 +856,10 @@ function parseInlines(raw, definitions, pt) {
else hi = mid - 1;
}
const seg = segs[lo];
// an end exactly where a contiguous seg starts (a table cell's dropped
// backslash, see makeRow) belongs to the previous seg, not past the gap
const prev = segs[lo - 1];
if (end && prev && v === seg.v && prev.v + prev.len === v) return prev.src + prev.len;
const within = Math.min(Math.max(v - seg.v, 0), seg.len);
if (end && v - seg.v > seg.len) return seg.src + seg.len; // in the virtual '\n' gap
return seg.src + within;
Expand Down
29 changes: 29 additions & 0 deletions scripts/lib/md-checks.test.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -350,3 +350,32 @@ test('doc-unreadable: a NUL byte flags binary/corrupted input instead of a false
const ok = checkDocument('# Title\n\nnormal content\n');
assert.ok(!ok.some((x) => x.check === 'doc-unreadable'));
});

// CWK-153: GFM's Tables extension removes the backslash of an escaped pipe in
// a cell BEFORE inline parsing, so it is gone even inside a code span. Both
// body rows must read `a|b` and still have exactly three cells (the escaped
// pipe is content, never a delimiter).
import { walk, textContent } from './md-ast.mjs';

test('table cells: an escaped pipe loses its backslash, in and out of a code span (CWK-153)', () => {
const src = fs.readFileSync(path.join(FIX, 'table-escaped-pipe.md'), 'utf8');
const tables = [];
walk(parseMarkdown(src), (n) => { if (n.type === 'table') tables.push(n); });
assert.strictEqual(tables.length, 1);
const rows = tables[0].children.map((r) => r.children.map(textContent));
assert.deepStrictEqual(rows, [
['Form', 'Cell', 'Note'],
['code span', 'a|b', 'escaped pipe inside a code span'],
['plain', 'a|b', 'escaped pipe outside a code span'],
]);
assert.strictEqual(tables[0].children[1].children[1].children[0].type, 'inlineCode');
assert.deepStrictEqual(checkDocument(src, { filePath: path.join(FIX, 'table-escaped-pipe.md') }), []);
});

test('table cells: a node ending right before a dropped backslash keeps its own source end (CWK-153)', () => {
const src = '| h |\n| - |\n| [x](url)\\|tail |\n';
const links = [];
walk(parseMarkdown(src), (n) => { if (n.type === 'link') links.push(n); });
assert.strictEqual(links.length, 1);
assert.strictEqual(src.slice(links[0].position.start.offset, links[0].position.end.offset), '[x](url)');
});
Loading