Skip to content
Open
Show file tree
Hide file tree
Changes from 6 commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 5 additions & 0 deletions .changeset/fix-markdown-empty-doc-fallback.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,5 @@
---
"@tiptap/markdown": patch
---

Fix `MarkdownManager.parse()` returning a document with empty `content` for markdown that yields no renderable blocks — whitespace-only input, or input whose only token has no registered handler (e.g. a leading-whitespace-indented line parsed as a code block when no code-block extension is present). A `doc` node requires at least one block child, so the empty document made `setContent` throw `RangeError: Invalid content for node doc: <>`. `parse()` now falls back to a single empty paragraph in that case, matching how an empty markdown string is represented.
153 changes: 153 additions & 0 deletions packages/markdown/__tests__/empty-doc-fallback.spec.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,153 @@
import { Editor, Extension, Node } from '@tiptap/core'
import { Document } from '@tiptap/extension-document'
import { Paragraph } from '@tiptap/extension-paragraph'
import { Text } from '@tiptap/extension-text'
import { Markdown, MarkdownManager } from '@tiptap/markdown'
import { afterEach, describe, expect, it } from 'vitest'

// A custom extension whose markdown handler returns a bare top-level text
// node instead of wrapping it in a block. Simulates a third-party extension
// misbehaving, to prove the doc-validity fallback doesn't just check
// `content.length > 0`.
const BareTextBlock = Extension.create({
name: 'bareTextBlock',
markdownTokenName: 'bareTextBlock',
parseMarkdown: token => ({ type: 'text', text: token.text || '' }),
markdownTokenizer: {
name: 'bareTextBlock',
level: 'block',
start: ':::bare',
tokenize: (src: string) => {
const match = src.match(/^:::bare\s+(.+?)\s+:::/)
if (!match) {
return undefined
}
return { type: 'bareTextBlock', raw: match[0], text: match[1] }
},
},
})

// A real block node, registered via `registerExtension()` after construction
// rather than passed to the constructor. Proves the block-node cache reads
// `this.extensions` (every extension ever registered) and not just
// `this.baseExtensions` (only the constructor's initial list).
const DirectlyRegisteredCallout = Node.create({
name: 'callout',
group: 'block',
content: 'text*',
markdownTokenName: 'callout',
parseMarkdown: (token, helpers) => ({
type: 'callout',
content: helpers.parseInline(token.tokens || []),
}),
markdownTokenizer: {
name: 'callout',
level: 'block',
start: ':::callout',
tokenize: (src: string, _tokens: unknown, lexer: any) => {
const match = src.match(/^:::callout\s+(.+?)\s+:::/)
if (!match) {
return undefined
}
return {
type: 'callout',
raw: match[0],
text: match[1],
tokens: lexer.inlineTokens(match[1]),
}
},
},
})

/**
* Regression tests for #7914.
*
* A `doc` node requires at least one block child, so markdown that yields no
* renderable blocks must not parse to a doc with empty content — that makes
* `setContent` throw `RangeError: Invalid content for node doc: <>`.
*/
describe('markdown parse never yields an empty document (#7914)', () => {
let editor: Editor | undefined
afterEach(() => editor?.destroy())

const mm = new MarkdownManager({ extensions: [Document, Paragraph, Text] })

it.each([[' / '], [' \\'], [' '], ['']])(
'parse(%j) returns a valid doc with at least one block',
input => {
const json = mm.parse(input)
expect(json.type).toBe('doc')
expect(json.content!.length).toBeGreaterThanOrEqual(1)
expect(json.content![0].type).toBe('paragraph')
},
)

it('setContent with whitespace+slash markdown does not throw', () => {
expect(() => {
editor = new Editor({
extensions: [Document, Paragraph, Text, Markdown],
content: ' / ',
contentType: 'markdown',
})
}).not.toThrow()
expect(editor!.getJSON().content![0].type).toBe('paragraph')
})
Comment thread
coderabbitai[bot] marked this conversation as resolved.

it('editor.commands.setContent with whitespace+slash markdown does not throw', () => {
editor = new Editor({ extensions: [Document, Paragraph, Text, Markdown] })

expect(() => {
editor!.commands.setContent(' / ', { contentType: 'markdown' })
}).not.toThrow()
expect(editor!.getJSON().content![0].type).toBe('paragraph')
})

it('still parses meaningful single-character content', () => {
expect(mm.parse('/')).toMatchObject({
type: 'doc',
content: [{ type: 'paragraph', content: [{ type: 'text', text: '/' }] }],
})
})

it('falls back to a paragraph when a custom handler yields a bare top-level text node', () => {
const bareTextManager = new MarkdownManager({
extensions: [Document, Paragraph, Text, BareTextBlock],
})
const json = bareTextManager.parse(':::bare hello :::')

expect(json.type).toBe('doc')
expect(json.content!.length).toBeGreaterThanOrEqual(1)
expect(json.content![0].type).toBe('paragraph')
})

it('setContent does not throw when a custom handler yields a bare top-level text node', () => {
expect(() => {
editor = new Editor({
extensions: [Document, Paragraph, Text, Markdown, BareTextBlock],
content: ':::bare hello :::',
contentType: 'markdown',
})
}).not.toThrow()
expect(editor!.getJSON().content![0].type).toBe('paragraph')
})

it('recognizes a block extension registered directly via registerExtension, not just the constructor', () => {
const directManager = new MarkdownManager({ extensions: [Document, Paragraph, Text] })
directManager.registerExtension(DirectlyRegisteredCallout)

const json = directManager.parse(':::callout hello :::')

expect(json.content![0].type).toBe('callout')
})

it('keeps a hardcoded token-type result (e.g. a setext heading) even without a matching registered extension', () => {
// parseToken's 'heading' case builds a { type: 'heading' } node directly —
// it doesn't depend on a Heading extension being registered. The fallback
// check must not mistake "no extension named 'heading'" for "not block
// content" and discard it.
const noHeadingManager = new MarkdownManager({ extensions: [Document, Paragraph, Text] })
const json = noHeadingManager.parse('Title\n---')

expect(json.content![0].type).toBe('heading')
})
})
11 changes: 10 additions & 1 deletion packages/markdown/src/MarkdownManager.ts
Original file line number Diff line number Diff line change
Expand Up @@ -348,10 +348,19 @@ export class MarkdownManager {
// Convert tokens to Tiptap JSON
const content = this.parseTokens(tokens, true)

// A document requires at least one block child. A non-empty `content`
// array isn't sufficient proof of that: a bare top-level `text` node
// (the only inline type parseToken ever returns unwrapped — see the
// 'text'/'escape' cases above) would still violate the schema's
// `block+` requirement. Every other type parseTokens can produce,
// whether from a hardcoded case like 'heading' or from a registered
// extension's own handler, is meant to stand alone at the top level.
const hasBlockContent = content.some(node => node.type !== 'text')

// Return a document node containing the parsed content

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Do not delete this comment

return {
type: 'doc',
content,
content: hasBlockContent ? content : [{ type: 'paragraph' }],
}
} finally {
this.activeParseLexer = previousParseLexer
Expand Down
Loading