Let a code span carry the emphasis marks, so bold code opens at all

This commit is contained in:
pj committed 2026-08-28 10:19:52 +05:30
1 parent 343b18ff57
commit 1b31e2f189
9 files changed
+130 -15

No files matched your search

+14
View File
@@ -20,6 +20,7 @@ import type { Editor } from "@tiptap/react";
import { EditorState, TextSelection } from "@tiptap/pm/state"; import { EditorState, TextSelection } from "@tiptap/pm/state";
import type { EditorProps as ProseMirrorProps, EditorView } from "@tiptap/pm/view"; import type { EditorProps as ProseMirrorProps, EditorView } from "@tiptap/pm/view";
import type { CalloutKind, HeadingLevel, MarkdownDocument } from "../model/doc"; import type { CalloutKind, HeadingLevel, MarkdownDocument } from "../model/doc";
import { sourceDocument } from "../markdown";
import { marks as markSpecs } from "../model/schema"; import { marks as markSpecs } from "../model/schema";
import type { MarkName } from "../model/schema"; import type { MarkName } from "../model/schema";
import { notify } from "../store/useToast"; import { notify } from "../store/useToast";
@@ -535,8 +536,21 @@ export function DocumentEditor({
return EditorState.create({ doc, plugins: base.plugins }); return EditorState.create({ doc, plugins: base.plugins });
} catch (error) { } catch (error) {
reportContentError(error); reportContentError(error);
// Never an empty document. `check` answers for the whole tree, so one text node the schema
// will not hold used to blank the file on screen, and the file on screen is what the next
// keystroke saves: the debounce would then write those few characters over the bytes on
// disk. The file's own source, in one raw block, is a document that always passes and that
// a save writes back verbatim, so the worst case is a document shown as source rather than
// a document destroyed.
try {
const doc = ed.schema.nodeFromJSON(sourceDocument(source).toJSON());
doc.check();
return EditorState.create({ doc, plugins: base.plugins });
} catch (fallback) {
reportContentError(fallback);
return EditorState.create({ schema: base.schema, plugins: base.plugins }); return EditorState.create({ schema: base.schema, plugins: base.plugins });
} }
}
}; };
const remember = (path: string, entry: Cached) => { const remember = (path: string, entry: Cached) => {
+38
View File
@@ -8,6 +8,7 @@ import type { Node as ProseMirrorNode } from "@tiptap/pm/model";
import { createEditorExtensions } from "./extensions"; import { createEditorExtensions } from "./extensions";
import { schema as contract } from "../model/schema"; import { schema as contract } from "../model/schema";
import { parseMarkdown, serializeMarkdown } from "../markdown"; import { parseMarkdown, serializeMarkdown } from "../markdown";
import { corpus } from "../markdown/corpus/load";
const extensions = () => const extensions = () =>
createEditorExtensions({ documentPath: () => "/notes/a.md", onError: () => {} }); createEditorExtensions({ documentPath: () => "/notes/a.md", onError: () => {} });
@@ -146,6 +147,43 @@ describe("the generated schema", () => {
// the serializer has to read node names rather than node types. This is that, asserted. // the serializer has to read node names rather than node types. This is that, asserted.
expect(serializeMarkdown(parsed, rebound)).toBe(serializeMarkdown(parsed, parsed.doc)); expect(serializeMarkdown(parsed, rebound)).toBe(serializeMarkdown(parsed, parsed.doc));
}); });
// The gate that was missing. Every other sweep reads a document the bridge built, and a document
// the bridge built is not necessarily one the editor will accept: `check()` is what src/editor/
// Editor.tsx asks before it installs the state, and a document that fails it is not partly
// refused, it is replaced by an empty one and the whole file goes blank on screen. Nothing here
// called it on a parsed document, so a paragraph holding "**`x`**" opened as nothing at all.
it("holds every corpus file, checked the way the editor checks it", () => {
const found: string[] = [];
for (const file of corpus()) {
const parsed = parseMarkdown(file.source, `/${file.name}`);
try {
built.nodeFromJSON(parsed.doc.toJSON()).check();
} catch (error) {
found.push(`${file.name}: ${String(error)}`);
}
}
expect(found).toEqual([]);
});
it("holds a code span carrying every mark that can be wrapped around one", () => {
for (const source of ["**`x`**\n", "_`x`_\n", "~~`x`~~\n", "[`x`](./y.md)\n"]) {
const parsed = parseMarkdown(source, "/notes/a.md");
const rebound = built.nodeFromJSON(parsed.doc.toJSON());
expect(() => rebound.check(), source).not.toThrow();
expect(serializeMarkdown(parsed, rebound), source).toBe(source);
}
// All of them on one span. The spelling moves, because marks are a set and MARK_ORDER decides
// the nesting once for every document, so what is asserted is that it settles there and stays.
const source = "**~~[`x`](./y.md)~~**\n";
const parsed = parseMarkdown(source, "/notes/a.md");
expect(() => built.nodeFromJSON(parsed.doc.toJSON()).check()).not.toThrow();
const once = serializeMarkdown(parsed, parsed.doc);
expect(once).toBe("[~~**`x`**~~](./y.md)\n");
const again = parseMarkdown(once, "/notes/a.md");
expect(serializeMarkdown(again, again.doc)).toBe(once);
});
}); });
describe("the editor built from them", () => { describe("the editor built from them", () => {
+9 -5
View File
@@ -327,13 +327,17 @@ const CONTAINERS: Array<[string, (inline: ProseMirrorNode[]) => ProseMirrorNode]
const MARKS = ["link", "strikethrough", "strong", "em", "code"]; const MARKS = ["link", "strikethrough", "strong", "em", "code"];
/** Every pair the schema actually permits: code excludes the formatting marks, and is a leaf. */ /**
* Every pair the schema actually permits: code is a leaf, so it is only ever the inner one.
*
* The inner code pairs were skipped here while the code mark excluded the formatting group, which
* is how `**`x`**` reached a release: the sweep could not build the one document that broke.
*/
function pairs(): Array<[string, string]> { function pairs(): Array<[string, string]> {
const out: Array<[string, string]> = []; const out: Array<[string, string]> = [];
for (const outer of MARKS) { for (const outer of MARKS) {
for (const inner of MARKS) { for (const inner of MARKS) {
if (outer === inner || outer === "code") continue; if (outer === inner || outer === "code") continue;
if (inner === "code" && outer !== "link") continue;
out.push([outer, inner]); out.push([outer, inner]);
} }
} }
@@ -343,8 +347,8 @@ function pairs(): Array<[string, string]> {
describe("still fixed: every mark nested inside every other, in every block", () => { describe("still fixed: every mark nested inside every other, in every block", () => {
// The fourth pass found the delete handler's missing whitespace guard and said why three passes // The fourth pass found the delete handler's missing whitespace guard and said why three passes
// had missed it: the sweeps carried strikethrough and link as sibling snippets and never nested // had missed it: the sweeps carried strikethrough and link as sibling snippets and never nested
// them. This is that gap closed. Thirteen ordered pairs, five spanning shapes, three boundary // them. This is that gap closed. Sixteen ordered pairs, five spanning shapes, three boundary
// paddings and eleven containers, which is 2145 documents, each built from the schema, written, // paddings and eleven containers, which is 2640 documents, each built from the schema, written,
// read back and then saved ten more times. // read back and then saved ten more times.
it("keeps every mark over every span, in every container, without moving or growing", () => { it("keeps every mark over every span, in every container, without moving or growing", () => {
@@ -380,7 +384,7 @@ describe("still fixed: every mark nested inside every other, in every block", ()
} }
} }
expect(checked, "the sweep has to actually be the size it claims").toBe(2145); expect(checked, "the sweep has to actually be the size it claims").toBe(2640);
expect(found).toEqual([]); expect(found).toEqual([]);
}, 120000); }, 120000);
@@ -9,8 +9,9 @@ Strong inside a strikethrough, and a strikethrough inside strong:
Emphasis inside a link, and a link inside emphasis: Emphasis inside a link, and a link inside emphasis:
[a\_b\_c](./x.md) and _a_[_b_](./y.md)_c_. [a\_b\_c](./x.md) and _a_[_b_](./y.md)_c_.
Code inside a link, which is the only pair code takes: Code inside a link, inside each emphasis mark, and inside all of them at once:
[`code span`](./z.md) and [a`b`c](./w.md). [`code span`](./z.md) and [a`b`c](./w.md) and **`bold code`** and _`em code`_ and
~~`struck code`~~ and **~~[`the lot`](./v.md)~~**.
All four at once, over a boundary space: All four at once, over a boundary space:
**q**~~**&#x20;r&#x20;**~~**s** **q**~~**&#x20;r&#x20;**~~**s**
+25 -1
View File
@@ -29,7 +29,8 @@
import type { Node as ProseMirrorNode } from "@tiptap/pm/model"; import type { Node as ProseMirrorNode } from "@tiptap/pm/model";
import { schema } from "../model/schema"; import { schema } from "../model/schema";
import type { MarkdownDocument } from "../model/doc"; import type { MarkdownDocument } from "../model/doc";
import { normaliseSource, splitFrontmatter, withFrontmatter } from "./frontmatter"; import { rawNode } from "../model/doc";
import { BOM, normaliseSource, splitFrontmatter, withFrontmatter } from "./frontmatter";
import { parseToMdast } from "./handlers"; import { parseToMdast } from "./handlers";
import { buildDoc } from "./parse"; import { buildDoc } from "./parse";
import { serializeBody } from "./serialize"; import { serializeBody } from "./serialize";
@@ -79,6 +80,29 @@ export function serializeMarkdown(document: MarkdownDocument, doc: ProseMirrorNo
return withFrontmatter(document.frontmatter, serializeBody(doc, document.frontmatter === null)); return withFrontmatter(document.frontmatter, serializeBody(doc, document.frontmatter === null));
} }
/**
* The whole body as one raw block: the file, shown as its own source.
*
* The last resort for a document the editor will not hold. The bridge refuses a construct by making
* a raw block of it, which is the same answer at a smaller scale, and a raw block the user has not
* typed in is written back byte for byte, so a file opened this way and saved is the file that was
* read. The alternative that was here, an empty document, is the one outcome the module's fourth
* invariant exists to forbid: the file looks empty on screen and the first keystroke saves it that
* way over the bytes on disk.
*/
export function sourceDocument(document: MarkdownDocument): ProseMirrorNode {
const { text } = normaliseSource(document.source);
// The prefix is the frontmatter as it will be written back, and it carries the byte order mark
// when there is one. `text` has already had that mark taken off, so putting it into the slice
// offset would cut the body's first character off with it.
const head = document.frontmatter ?? "";
const prefix = head.startsWith(BOM) ? head.slice(BOM.length) : head;
const body = text.slice(prefix.length);
const doc = schema.nodes.doc.createAndFill(null, body ? [rawNode(body)] : []);
if (!doc) throw new Error("the source could not be held as a raw block");
return doc;
}
/** /**
* A .txt file, which the tree marks editable and the editor opens alongside markdown. * A .txt file, which the tree marks editable and the editor opens alongside markdown.
* *
+5 -2
View File
@@ -446,7 +446,10 @@ function inlineFrom(nodes: PhrasingContent[], marks: readonly Mark[]): ProseMirr
break; break;
} }
case "inlineCode": { case "inlineCode": {
if (node.value) out.push(schema.text(node.value, [...marks, m.code.create()])); // addToSet for the same reason the wrappers use it: a spread is sorted by `Mark.setFrom`
// but is not put through the marks' own exclusion rules, so it can build a set the schema
// does not allow and that `doc.check()` refuses, which blanks the whole document.
if (node.value) out.push(schema.text(node.value, m.code.create().addToSet(marks)));
break; break;
} }
case "link": { case "link": {
@@ -456,7 +459,7 @@ function inlineFrom(nodes: PhrasingContent[], marks: readonly Mark[]): ProseMirr
// dropping a destination quietly is the one thing this bridge is here not to do. // dropping a destination quietly is the one thing this bridge is here not to do.
if (marks.some((mark) => mark.type === m.link)) return null; if (marks.some((mark) => mark.type === m.link)) return null;
const mark = m.link.create({ href: node.url, title: node.title ?? null }); const mark = m.link.create({ href: node.url, title: node.title ?? null });
const inner = inlineFrom(node.children, [...marks, mark]); const inner = inlineFrom(node.children, mark.addToSet(marks));
// A mark needs something to sit on, so a link with no text at all, `[](./x.md)`, has // A mark needs something to sit on, so a link with no text at all, `[](./x.md)`, has
// nowhere to live in the document and would be dropped along with its destination. // nowhere to live in the document and would be dropped along with its destination.
if (!inner || inner.length === 0) return null; if (!inner || inner.length === 0) return null;
+23 -1
View File
@@ -5,7 +5,18 @@ import { describe, expect, it } from "vitest";
import type { Node as ProseMirrorNode } from "@tiptap/pm/model"; import type { Node as ProseMirrorNode } from "@tiptap/pm/model";
import { schema } from "../model/schema"; import { schema } from "../model/schema";
import { corpus } from "./corpus/load"; import { corpus } from "./corpus/load";
import { parseMarkdown, serializeMarkdown } from "./index"; import { parseMarkdown, serializeMarkdown, sourceDocument } from "./index";
import { BOM, normaliseSource } from "./frontmatter";
/**
* The source as a save of it writes it back: CRLF collapsed, the byte order mark still there, and
* the one line ending the house style ends a file with. Everything else is the file's own bytes.
*/
function settled(source: string): string {
const { text, bom } = normaliseSource(source);
const body = text === "" || text.endsWith("\n") ? text : `${text}\n`;
return bom ? BOM + body : body;
}
const files = corpus(); const files = corpus();
@@ -187,6 +198,17 @@ describe("opening a document", () => {
} }
}); });
// The fallback the editor installs when a document is one it cannot hold. It has to be worth
// more than the empty document it replaced, which means the bytes have to survive a save.
it("can be shown as its own source, and written back as the bytes that were read", () => {
for (const file of files) {
const document = parseMarkdown(file.source, `/corpus/${file.name}`);
const shown = sourceDocument(document);
expect(shown.childCount, file.name).toBeLessThanOrEqual(1);
expect(serializeMarkdown(document, shown), file.name).toBe(settled(file.source));
}
});
it("reaches nothing outside itself", () => { it("reaches nothing outside itself", () => {
const sources = import.meta.glob("./*.ts", { query: "?raw", import: "default", eager: true }) as Record<string, string>; const sources = import.meta.glob("./*.ts", { query: "?raw", import: "default", eager: true }) as Record<string, string>;
for (const [name, text] of Object.entries(sources)) { for (const [name, text] of Object.entries(sources)) {
+5 -2
View File
@@ -240,10 +240,13 @@ describe("inline", () => {
expect(link.toJSON()).toEqual({ type: "link", attrs: { href: "./a.md", title: "A" } }); expect(link.toJSON()).toEqual({ type: "link", attrs: { href: "./a.md", title: "A" } });
}); });
it("lets a code span sit inside a link but not inside emphasis", () => { it("lets a code span sit inside a link and inside emphasis", () => {
const code = m.code.create(); const code = m.code.create();
expect(code.isInSet(m.link.create({ href: "./a.md" }).addToSet([code]))).toBeTruthy(); expect(code.isInSet(m.link.create({ href: "./a.md" }).addToSet([code]))).toBeTruthy();
expect(m.strong.create().addToSet([code])).toEqual([code]); for (const outer of [m.strong, m.em, m.strikethrough]) {
const set = outer.create().addToSet([code]);
expect([outer.name, set.map((mark) => mark.type.name)]).toEqual([outer.name, [outer.name, "code"]]);
}
}); });
it("treats math and images as inline atoms", () => { it("treats math and images as inline atoms", () => {
+8 -2
View File
@@ -433,10 +433,16 @@ export const marks: { [name in MarkName]: MarkSpec } = {
toDOM: () => ["s", 0], toDOM: () => ["s", 0],
}, },
// A code span is literal, so it excludes the emphasis marks, but not link: [`x`](y) is valid. // A code span's own content is literal, so nothing can be emphasised inside one, but a code span
// can itself be emphasised or linked: `**\`x\`**` and `[\`x\`](y)` are both ordinary markdown and
// both render on GitHub. Marks are a flat set here, so those two readings are the same set and
// the direction is decided once, by the serializer: src/markdown/serialize.ts puts code innermost
// and writes the emphasis around it. Excluding the formatting group instead, which this mark did
// until a document holding `**\`x\`**` could not be opened at all, is not a narrower rule but a
// wrong one: the bridge kept building the set the file described and every such document failed
// `doc.check()` on the way into the editor.
code: { code: {
code: true, code: true,
excludes: "formatting code",
parseDOM: [{ tag: "code" }], parseDOM: [{ tag: "code" }],
toDOM: () => ["code", 0], toDOM: () => ["code", 0],
}, },