816 lines
27 KiB
Rust
816 lines
27 KiB
Rust
use comrak::nodes::{AstNode, ListType, NodeValue, TableAlignment};
|
||
use comrak::{parse_document, Arena, Options};
|
||
use serde_json::{json, Map, Value};
|
||
|
||
#[derive(Debug, Clone)]
|
||
pub struct ParsedLocalMarkdownPage {
|
||
pub title: String,
|
||
pub body: String,
|
||
}
|
||
|
||
#[derive(Debug, Clone)]
|
||
struct MarkdownAstDocument {
|
||
blocks: Vec<MarkdownBlock>,
|
||
}
|
||
|
||
#[derive(Debug, Clone)]
|
||
enum MarkdownBlock {
|
||
Paragraph(Vec<MarkdownInline>),
|
||
Heading {
|
||
level: u8,
|
||
content: Vec<MarkdownInline>,
|
||
},
|
||
Quote(Vec<MarkdownInline>),
|
||
CodeBlock {
|
||
language: String,
|
||
text: String,
|
||
},
|
||
Divider,
|
||
BulletListItem(Vec<MarkdownInline>),
|
||
NumberedListItem(Vec<MarkdownInline>),
|
||
Todo {
|
||
checked: bool,
|
||
content: Vec<MarkdownInline>,
|
||
},
|
||
Table {
|
||
alignments: Vec<TableAlignment>,
|
||
rows: Vec<MarkdownTableRow>,
|
||
},
|
||
Image {
|
||
alt: String,
|
||
source_path: String,
|
||
},
|
||
Mindmap {
|
||
name: String,
|
||
source_path: String,
|
||
},
|
||
Media {
|
||
name: String,
|
||
source_path: String,
|
||
},
|
||
}
|
||
|
||
#[derive(Debug, Clone)]
|
||
struct MarkdownTableRow {
|
||
is_header: bool,
|
||
cells: Vec<MarkdownTableCell>,
|
||
}
|
||
|
||
#[derive(Debug, Clone)]
|
||
struct MarkdownTableCell {
|
||
content: Vec<MarkdownInline>,
|
||
}
|
||
|
||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||
struct MarkdownInline {
|
||
text: String,
|
||
styles: MarkdownInlineStyles,
|
||
}
|
||
|
||
#[derive(Debug, Clone, Default, PartialEq, Eq)]
|
||
struct MarkdownInlineStyles {
|
||
bold: bool,
|
||
italic: bool,
|
||
strike: bool,
|
||
underline: bool,
|
||
code: bool,
|
||
link: Option<String>,
|
||
}
|
||
|
||
pub fn parse_markdown_page(markdown: &str, file_name: &str) -> ParsedLocalMarkdownPage {
|
||
let (_, body) = split_frontmatter(markdown);
|
||
let title = file_stem_title(file_name);
|
||
ParsedLocalMarkdownPage {
|
||
title,
|
||
body: body.to_string(),
|
||
}
|
||
}
|
||
|
||
pub fn markdown_to_blocks(markdown: &str) -> Value {
|
||
markdown_ast_document_to_blocks(&parse_markdown_ast_document(markdown))
|
||
}
|
||
|
||
pub fn parse_markdown_attachment_link(trimmed: &str) -> Option<(String, String)> {
|
||
let value = trimmed
|
||
.strip_prefix('!')
|
||
.unwrap_or(trimmed)
|
||
.strip_prefix('[')?;
|
||
let (label, rest) = value.split_once("](")?;
|
||
let raw_target = rest.strip_suffix(')')?.trim();
|
||
let target = raw_target
|
||
.strip_prefix('<')
|
||
.and_then(|value| value.strip_suffix('>'))
|
||
.unwrap_or(raw_target)
|
||
.trim();
|
||
if target.is_empty()
|
||
|| target.starts_with("http://")
|
||
|| target.starts_with("https://")
|
||
|| target.starts_with('#')
|
||
|| target.starts_with("mailto:")
|
||
{
|
||
return None;
|
||
}
|
||
let target_path = std::path::Path::new(target);
|
||
let extension = target_path.extension().and_then(|value| value.to_str())?;
|
||
if extension.eq_ignore_ascii_case("md") || extension.eq_ignore_ascii_case("markdown") {
|
||
return None;
|
||
}
|
||
let fallback_name = target_path
|
||
.file_name()
|
||
.and_then(|value| value.to_str())
|
||
.unwrap_or(target)
|
||
.trim();
|
||
let name = if label.trim().is_empty() {
|
||
fallback_name
|
||
} else {
|
||
label.trim()
|
||
};
|
||
Some((name.to_string(), target.to_string()))
|
||
}
|
||
|
||
fn parse_markdown_ast_document(markdown: &str) -> MarkdownAstDocument {
|
||
let arena = Arena::new();
|
||
let mut options = Options::default();
|
||
options.extension.table = true;
|
||
options.extension.tasklist = true;
|
||
options.extension.strikethrough = true;
|
||
options.extension.autolink = true;
|
||
options.extension.front_matter_delimiter = Some("---".to_string());
|
||
options.parse.tasklist_in_table = true;
|
||
|
||
let root = parse_document(&arena, markdown, &options);
|
||
let mut blocks = Vec::new();
|
||
for node in root.children() {
|
||
append_ast_block(node, &mut blocks);
|
||
}
|
||
MarkdownAstDocument { blocks }
|
||
}
|
||
|
||
fn append_ast_block<'a>(node: &'a AstNode<'a>, blocks: &mut Vec<MarkdownBlock>) {
|
||
match node.data.borrow().value.clone() {
|
||
NodeValue::Paragraph => append_ast_paragraph(node, blocks),
|
||
NodeValue::Heading(heading) => blocks.push(MarkdownBlock::Heading {
|
||
level: heading.level,
|
||
content: collect_inline_children(node),
|
||
}),
|
||
NodeValue::BlockQuote => blocks.push(MarkdownBlock::Quote(collect_inline_children(node))),
|
||
NodeValue::ThematicBreak => blocks.push(MarkdownBlock::Divider),
|
||
NodeValue::CodeBlock(code_block) => blocks.push(MarkdownBlock::CodeBlock {
|
||
language: code_block.info.clone(),
|
||
text: code_block.literal.clone(),
|
||
}),
|
||
NodeValue::List(list) => {
|
||
for item in node.children() {
|
||
append_ast_list_item(item, list.list_type == ListType::Ordered, blocks);
|
||
}
|
||
}
|
||
NodeValue::Table(table) => blocks.push(ast_table_to_ir(node, table.alignments)),
|
||
NodeValue::HtmlBlock(html) => blocks.push(MarkdownBlock::Paragraph(vec![MarkdownInline {
|
||
text: html.literal.clone(),
|
||
styles: MarkdownInlineStyles::default(),
|
||
}])),
|
||
NodeValue::FrontMatter(_) => {}
|
||
_ => {
|
||
let text = collect_plain_text(node);
|
||
if !text.is_empty() {
|
||
blocks.push(MarkdownBlock::Paragraph(vec![MarkdownInline {
|
||
text,
|
||
styles: MarkdownInlineStyles::default(),
|
||
}]));
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
fn append_ast_paragraph<'a>(node: &'a AstNode<'a>, blocks: &mut Vec<MarkdownBlock>) {
|
||
if let Some((alt, source_path)) = paragraph_image(node) {
|
||
blocks.push(MarkdownBlock::Image { alt, source_path });
|
||
return;
|
||
}
|
||
if let Some((name, source_path)) = paragraph_mindmap(node) {
|
||
blocks.push(MarkdownBlock::Mindmap { name, source_path });
|
||
return;
|
||
}
|
||
if let Some((name, source_path)) = paragraph_attachment_media(node) {
|
||
blocks.push(MarkdownBlock::Media { name, source_path });
|
||
return;
|
||
}
|
||
if let Some((name, source_path, remaining)) = paragraph_leading_attachment_media(node) {
|
||
blocks.push(MarkdownBlock::Media { name, source_path });
|
||
let content = merge_adjacent_inline_nodes(remaining);
|
||
if !content.is_empty() {
|
||
blocks.push(MarkdownBlock::Paragraph(content));
|
||
}
|
||
return;
|
||
}
|
||
blocks.push(MarkdownBlock::Paragraph(collect_inline_children(node)));
|
||
}
|
||
|
||
fn append_ast_list_item<'a>(node: &'a AstNode<'a>, ordered: bool, blocks: &mut Vec<MarkdownBlock>) {
|
||
let (is_task, checked) = match node.data.borrow().value.clone() {
|
||
NodeValue::TaskItem(task_item) => (true, task_item.symbol.is_some()),
|
||
NodeValue::Item(_) => (false, false),
|
||
_ => return append_ast_block(node, blocks),
|
||
};
|
||
let mut content = Vec::new();
|
||
for child in node.children() {
|
||
match child.data.borrow().value.clone() {
|
||
NodeValue::Paragraph => content.extend(collect_inline_children(child)),
|
||
_ => append_ast_block(child, blocks),
|
||
}
|
||
}
|
||
|
||
if is_task {
|
||
blocks.push(MarkdownBlock::Todo { checked, content });
|
||
} else if ordered {
|
||
blocks.push(MarkdownBlock::NumberedListItem(content));
|
||
} else {
|
||
blocks.push(MarkdownBlock::BulletListItem(content));
|
||
}
|
||
}
|
||
|
||
fn ast_table_to_ir<'a>(node: &'a AstNode<'a>, alignments: Vec<TableAlignment>) -> MarkdownBlock {
|
||
let rows = node
|
||
.children()
|
||
.map(|row| {
|
||
let is_header = matches!(row.data.borrow().value, NodeValue::TableRow(true));
|
||
let cells = row
|
||
.children()
|
||
.map(|cell| MarkdownTableCell {
|
||
content: collect_inline_children(cell),
|
||
})
|
||
.collect::<Vec<_>>();
|
||
MarkdownTableRow { is_header, cells }
|
||
})
|
||
.collect::<Vec<_>>();
|
||
MarkdownBlock::Table { alignments, rows }
|
||
}
|
||
|
||
fn collect_inline_children<'a>(node: &'a AstNode<'a>) -> Vec<MarkdownInline> {
|
||
let mut nodes = Vec::new();
|
||
for child in node.children() {
|
||
collect_inline_nodes(child, &MarkdownInlineStyles::default(), &mut nodes);
|
||
}
|
||
merge_adjacent_inline_nodes(nodes)
|
||
}
|
||
|
||
fn collect_inline_nodes<'a>(
|
||
node: &'a AstNode<'a>,
|
||
active_styles: &MarkdownInlineStyles,
|
||
nodes: &mut Vec<MarkdownInline>,
|
||
) {
|
||
match node.data.borrow().value.clone() {
|
||
NodeValue::Text(text) => push_inline_text_node(nodes, text.as_ref(), active_styles),
|
||
NodeValue::TaskItem(task_item) => {
|
||
let marker = if task_item.symbol.is_some() {
|
||
"[x] "
|
||
} else {
|
||
"[ ] "
|
||
};
|
||
push_inline_text_node(nodes, marker, active_styles);
|
||
for child in node.children() {
|
||
collect_inline_nodes(child, active_styles, nodes);
|
||
}
|
||
}
|
||
NodeValue::Code(code) => {
|
||
let mut styles = active_styles.clone();
|
||
styles.code = true;
|
||
push_inline_text_node(nodes, &code.literal, &styles);
|
||
}
|
||
NodeValue::Strong => {
|
||
let mut styles = active_styles.clone();
|
||
styles.bold = true;
|
||
collect_inline_children_with_styles(node, &styles, nodes);
|
||
}
|
||
NodeValue::Emph => {
|
||
let mut styles = active_styles.clone();
|
||
styles.italic = true;
|
||
collect_inline_children_with_styles(node, &styles, nodes);
|
||
}
|
||
NodeValue::Strikethrough => {
|
||
let mut styles = active_styles.clone();
|
||
styles.strike = true;
|
||
collect_inline_children_with_styles(node, &styles, nodes);
|
||
}
|
||
NodeValue::Underline => {
|
||
let mut styles = active_styles.clone();
|
||
styles.underline = true;
|
||
collect_inline_children_with_styles(node, &styles, nodes);
|
||
}
|
||
NodeValue::Link(link) => {
|
||
let mut styles = active_styles.clone();
|
||
styles.link = Some(link.url.clone());
|
||
collect_inline_children_with_styles(node, &styles, nodes);
|
||
}
|
||
NodeValue::SoftBreak | NodeValue::LineBreak => {
|
||
push_inline_text_node(nodes, " ", active_styles);
|
||
}
|
||
NodeValue::HtmlInline(text) => push_inline_text_node(nodes, text.as_ref(), active_styles),
|
||
NodeValue::Image(link) => {
|
||
let mut styles = active_styles.clone();
|
||
styles.link = Some(link.url.clone());
|
||
push_inline_text_node(nodes, link.url.as_str(), &styles);
|
||
}
|
||
_ => collect_inline_children_with_styles(node, active_styles, nodes),
|
||
}
|
||
}
|
||
|
||
fn collect_inline_children_with_styles<'a>(
|
||
node: &'a AstNode<'a>,
|
||
active_styles: &MarkdownInlineStyles,
|
||
nodes: &mut Vec<MarkdownInline>,
|
||
) {
|
||
for child in node.children() {
|
||
collect_inline_nodes(child, active_styles, nodes);
|
||
}
|
||
}
|
||
|
||
fn push_inline_text_node(
|
||
nodes: &mut Vec<MarkdownInline>,
|
||
text: &str,
|
||
styles: &MarkdownInlineStyles,
|
||
) {
|
||
if text.is_empty() {
|
||
return;
|
||
}
|
||
nodes.push(MarkdownInline {
|
||
text: text.to_string(),
|
||
styles: styles.clone(),
|
||
});
|
||
}
|
||
|
||
fn merge_adjacent_inline_nodes(nodes: Vec<MarkdownInline>) -> Vec<MarkdownInline> {
|
||
let mut merged = Vec::<MarkdownInline>::new();
|
||
for node in nodes {
|
||
if let Some(last) = merged.last_mut() {
|
||
if last.styles == node.styles {
|
||
last.text.push_str(&node.text);
|
||
continue;
|
||
}
|
||
}
|
||
merged.push(node);
|
||
}
|
||
merged
|
||
}
|
||
|
||
fn paragraph_attachment_media<'a>(node: &'a AstNode<'a>) -> Option<(String, String)> {
|
||
let mut children = node.children();
|
||
let first = children.next()?;
|
||
if children.next().is_some() {
|
||
return None;
|
||
}
|
||
link_attachment_media(first)
|
||
}
|
||
|
||
fn paragraph_mindmap<'a>(node: &'a AstNode<'a>) -> Option<(String, String)> {
|
||
let mut children = node.children();
|
||
let first = children.next()?;
|
||
if children.next().is_some() {
|
||
return None;
|
||
}
|
||
link_mindmap(first)
|
||
}
|
||
|
||
fn paragraph_image<'a>(node: &'a AstNode<'a>) -> Option<(String, String)> {
|
||
let mut children = node.children();
|
||
let first = children.next()?;
|
||
if children.next().is_some() {
|
||
return None;
|
||
}
|
||
let NodeValue::Image(link) = &first.data.borrow().value else {
|
||
return None;
|
||
};
|
||
let alt = collect_plain_text(first).trim().to_string();
|
||
Some((alt, link.url.clone()))
|
||
}
|
||
|
||
fn paragraph_leading_attachment_media<'a>(
|
||
node: &'a AstNode<'a>,
|
||
) -> Option<(String, String, Vec<MarkdownInline>)> {
|
||
let mut children = node.children();
|
||
let first = children.next()?;
|
||
let (name, source_path) = link_attachment_media(first)?;
|
||
let second = children.next()?;
|
||
if !matches!(
|
||
second.data.borrow().value,
|
||
NodeValue::SoftBreak | NodeValue::LineBreak
|
||
) {
|
||
return None;
|
||
}
|
||
let mut remaining = Vec::new();
|
||
for child in children {
|
||
collect_inline_nodes(child, &MarkdownInlineStyles::default(), &mut remaining);
|
||
}
|
||
Some((name, source_path, remaining))
|
||
}
|
||
|
||
fn link_attachment_media<'a>(node: &'a AstNode<'a>) -> Option<(String, String)> {
|
||
let NodeValue::Link(link) = &node.data.borrow().value else {
|
||
return None;
|
||
};
|
||
parse_markdown_attachment_link(&format!("[{}]({})", collect_plain_text(node), link.url))
|
||
}
|
||
|
||
fn link_mindmap<'a>(node: &'a AstNode<'a>) -> Option<(String, String)> {
|
||
let NodeValue::Link(link) = &node.data.borrow().value else {
|
||
return None;
|
||
};
|
||
let target = link.url.trim();
|
||
let file_name = std::path::Path::new(target)
|
||
.file_name()
|
||
.and_then(|value| value.to_str())
|
||
.unwrap_or(target)
|
||
.trim();
|
||
let lower = file_name.to_ascii_lowercase();
|
||
let is_mindmap = lower.ends_with(".mindmap.json")
|
||
|| (file_name.starts_with("思维导图") && lower.ends_with(".json"));
|
||
if !is_mindmap {
|
||
return None;
|
||
}
|
||
let name = collect_plain_text(node).trim().to_string();
|
||
Some((
|
||
if name.is_empty() {
|
||
"思维导图".to_string()
|
||
} else {
|
||
name
|
||
},
|
||
target.to_string(),
|
||
))
|
||
}
|
||
|
||
fn markdown_ast_document_to_blocks(document: &MarkdownAstDocument) -> Value {
|
||
Value::Array(
|
||
document
|
||
.blocks
|
||
.iter()
|
||
.enumerate()
|
||
.map(|(index, block)| markdown_block_to_json(block, index + 1))
|
||
.collect(),
|
||
)
|
||
}
|
||
|
||
fn markdown_block_to_json(block: &MarkdownBlock, block_number: usize) -> Value {
|
||
match block {
|
||
MarkdownBlock::Paragraph(content) => json_block(
|
||
"paragraph",
|
||
legacy_inline_nodes_to_json(content),
|
||
block_number,
|
||
),
|
||
MarkdownBlock::Heading { level, content } => json!({
|
||
"id": format!("local-block-{block_number}"),
|
||
"type": "heading",
|
||
"props": { "level": level },
|
||
"content": legacy_inline_nodes_to_json(content),
|
||
"children": [],
|
||
}),
|
||
MarkdownBlock::Quote(content) => {
|
||
json_block("quote", legacy_inline_nodes_to_json(content), block_number)
|
||
}
|
||
MarkdownBlock::CodeBlock { language, text } => json!({
|
||
"id": format!("local-block-{block_number}"),
|
||
"type": "codeBlock",
|
||
"props": { "language": language },
|
||
"content": [{ "type": "text", "text": text, "styles": {} }],
|
||
"children": [],
|
||
}),
|
||
MarkdownBlock::Divider => json!({
|
||
"id": format!("local-block-{block_number}"),
|
||
"type": "divider",
|
||
"content": [],
|
||
"children": [],
|
||
}),
|
||
MarkdownBlock::BulletListItem(content) => json_block(
|
||
"bulletListItem",
|
||
legacy_inline_nodes_to_json(content),
|
||
block_number,
|
||
),
|
||
MarkdownBlock::NumberedListItem(content) => json_block(
|
||
"numberedListItem",
|
||
legacy_inline_nodes_to_json(content),
|
||
block_number,
|
||
),
|
||
MarkdownBlock::Todo { checked, content } => json!({
|
||
"id": format!("local-block-{block_number}"),
|
||
"type": "todo",
|
||
"props": { "checked": checked },
|
||
"content": legacy_inline_nodes_to_json(content),
|
||
"children": [],
|
||
}),
|
||
MarkdownBlock::Table { alignments, rows } => {
|
||
table_block_to_json(alignments, rows, block_number)
|
||
}
|
||
MarkdownBlock::Image { alt, source_path } => json!({
|
||
"id": format!("local-block-{block_number}"),
|
||
"type": "image",
|
||
"props": {
|
||
"src": source_path,
|
||
"alt": alt,
|
||
"title": alt,
|
||
},
|
||
"content": [],
|
||
"children": [],
|
||
}),
|
||
MarkdownBlock::Mindmap { name, source_path } => json!({
|
||
"id": format!("local-block-{block_number}"),
|
||
"type": "mindmap",
|
||
"props": {
|
||
"name": name,
|
||
"sourcePath": source_path,
|
||
"mindmapId": source_path,
|
||
"rootNodeId": "root",
|
||
},
|
||
"content": [],
|
||
"children": [],
|
||
}),
|
||
MarkdownBlock::Media { name, source_path } => json!({
|
||
"id": format!("local-block-{block_number}"),
|
||
"type": "media",
|
||
"props": {
|
||
"name": name,
|
||
"sourcePath": source_path,
|
||
},
|
||
"content": [],
|
||
"children": [],
|
||
}),
|
||
}
|
||
}
|
||
|
||
fn table_block_to_json(
|
||
alignments: &[TableAlignment],
|
||
rows: &[MarkdownTableRow],
|
||
block_number: usize,
|
||
) -> Value {
|
||
let content_rows = rows
|
||
.iter()
|
||
.map(|row| {
|
||
let cells = row
|
||
.cells
|
||
.iter()
|
||
.enumerate()
|
||
.map(|(column_index, cell)| {
|
||
let cell_type = if row.is_header {
|
||
"tableHeader"
|
||
} else {
|
||
"tableCell"
|
||
};
|
||
let text_align = match alignments
|
||
.get(column_index)
|
||
.copied()
|
||
.unwrap_or(TableAlignment::None)
|
||
{
|
||
TableAlignment::Left => "left",
|
||
TableAlignment::Center => "center",
|
||
TableAlignment::Right => "right",
|
||
TableAlignment::None => "",
|
||
};
|
||
json!({
|
||
"type": cell_type,
|
||
"attrs": {
|
||
"colspan": 1,
|
||
"rowspan": 1,
|
||
"colwidth": null,
|
||
"textAlign": text_align,
|
||
},
|
||
"content": [{
|
||
"type": "paragraph",
|
||
"content": tiptap_inline_nodes_to_json(&cell.content),
|
||
}]
|
||
})
|
||
})
|
||
.collect::<Vec<_>>();
|
||
json!({
|
||
"type": "tableRow",
|
||
"content": cells,
|
||
})
|
||
})
|
||
.collect::<Vec<_>>();
|
||
json!({
|
||
"id": format!("local-block-{block_number}"),
|
||
"type": "table",
|
||
"props": {
|
||
"tiptapTable": {
|
||
"type": "table",
|
||
"attrs": { "blockId": format!("local-block-{block_number}") },
|
||
"content": content_rows,
|
||
}
|
||
},
|
||
"content": [],
|
||
"children": [],
|
||
})
|
||
}
|
||
|
||
fn json_block(block_type: &str, content: Vec<Value>, block_number: usize) -> Value {
|
||
json!({
|
||
"id": format!("local-block-{block_number}"),
|
||
"type": block_type,
|
||
"content": content,
|
||
"children": [],
|
||
})
|
||
}
|
||
|
||
fn legacy_inline_nodes_to_json(nodes: &[MarkdownInline]) -> Vec<Value> {
|
||
nodes
|
||
.iter()
|
||
.map(|node| {
|
||
json!({
|
||
"type": "text",
|
||
"text": node.text,
|
||
"styles": legacy_styles_to_json(&node.styles),
|
||
})
|
||
})
|
||
.collect()
|
||
}
|
||
|
||
fn legacy_styles_to_json(styles: &MarkdownInlineStyles) -> Value {
|
||
let mut object = Map::new();
|
||
if styles.bold {
|
||
object.insert("bold".to_string(), Value::Bool(true));
|
||
}
|
||
if styles.italic {
|
||
object.insert("italic".to_string(), Value::Bool(true));
|
||
}
|
||
if styles.strike {
|
||
object.insert("strike".to_string(), Value::Bool(true));
|
||
}
|
||
if styles.underline {
|
||
object.insert("underline".to_string(), Value::Bool(true));
|
||
}
|
||
if styles.code {
|
||
object.insert("code".to_string(), Value::Bool(true));
|
||
}
|
||
if let Some(href) = styles
|
||
.link
|
||
.as_deref()
|
||
.map(str::trim)
|
||
.filter(|href| !href.is_empty())
|
||
{
|
||
object.insert("link".to_string(), Value::String(href.to_string()));
|
||
}
|
||
Value::Object(object)
|
||
}
|
||
|
||
fn tiptap_inline_nodes_to_json(nodes: &[MarkdownInline]) -> Vec<Value> {
|
||
nodes
|
||
.iter()
|
||
.filter(|node| !node.text.is_empty())
|
||
.map(|node| {
|
||
let mut object = Map::new();
|
||
object.insert("type".to_string(), Value::String("text".to_string()));
|
||
object.insert("text".to_string(), Value::String(node.text.clone()));
|
||
let marks = tiptap_marks_from_styles(&node.styles);
|
||
if !marks.is_empty() {
|
||
object.insert("marks".to_string(), Value::Array(marks));
|
||
}
|
||
Value::Object(object)
|
||
})
|
||
.collect()
|
||
}
|
||
|
||
fn tiptap_marks_from_styles(styles: &MarkdownInlineStyles) -> Vec<Value> {
|
||
let mut marks = Vec::new();
|
||
if styles.bold {
|
||
marks.push(json!({ "type": "bold" }));
|
||
}
|
||
if styles.italic {
|
||
marks.push(json!({ "type": "italic" }));
|
||
}
|
||
if styles.underline {
|
||
marks.push(json!({ "type": "underline" }));
|
||
}
|
||
if styles.strike {
|
||
marks.push(json!({ "type": "strike" }));
|
||
}
|
||
if styles.code {
|
||
marks.push(json!({ "type": "code" }));
|
||
}
|
||
if let Some(href) = styles
|
||
.link
|
||
.as_deref()
|
||
.map(str::trim)
|
||
.filter(|href| !href.is_empty())
|
||
{
|
||
marks.push(json!({ "type": "link", "attrs": { "href": href } }));
|
||
}
|
||
marks
|
||
}
|
||
|
||
fn collect_plain_text<'a>(node: &'a AstNode<'a>) -> String {
|
||
let mut parts = Vec::new();
|
||
for descendant in node.descendants() {
|
||
if let NodeValue::Text(text) = &descendant.data.borrow().value {
|
||
parts.push(text.as_ref().to_string());
|
||
}
|
||
}
|
||
parts.join("")
|
||
}
|
||
|
||
pub(crate) fn split_frontmatter(markdown: &str) -> (Option<String>, &str) {
|
||
let normalized = markdown.strip_prefix('\u{feff}').unwrap_or(markdown);
|
||
if !normalized.starts_with("---\n") {
|
||
return (None, normalized);
|
||
}
|
||
let rest = &normalized[4..];
|
||
if let Some(end) = rest.find("\n---\n") {
|
||
let frontmatter = rest[..end].to_string();
|
||
let body = &rest[end + 5..];
|
||
return (Some(frontmatter), body);
|
||
}
|
||
(None, normalized)
|
||
}
|
||
|
||
pub(crate) fn file_stem_title(file_name: &str) -> String {
|
||
std::path::Path::new(file_name)
|
||
.file_stem()
|
||
.and_then(|stem| stem.to_str())
|
||
.map(str::trim)
|
||
.filter(|value| !value.is_empty())
|
||
.unwrap_or(file_name)
|
||
.to_string()
|
||
}
|
||
|
||
#[cfg(test)]
|
||
mod tests {
|
||
use super::markdown_to_blocks;
|
||
|
||
#[test]
|
||
fn markdown_image_parses_as_image_block() {
|
||
let blocks = markdown_to_blocks("\n");
|
||
let first = blocks
|
||
.as_array()
|
||
.and_then(|items| items.first())
|
||
.expect("first block");
|
||
|
||
assert_eq!(first["type"].as_str(), Some("image"));
|
||
assert_eq!(first["props"]["src"].as_str(), Some("assets/photo.jpg"));
|
||
assert_eq!(first["props"]["alt"].as_str(), Some("示例图片"));
|
||
}
|
||
|
||
/// 固定空引用块行为:`>` 在 GFM AST 中产生 BlockQuote 节点,
|
||
/// collect_inline_children 返回空 vec → 输出 type=quote content=[]。
|
||
#[test]
|
||
fn markdown_empty_blockquote_parse() {
|
||
let blocks = markdown_to_blocks("> \n\n>");
|
||
let array = blocks.as_array().expect("blocks");
|
||
assert!(!array.is_empty(), "应至少产生一个引用块");
|
||
for block in array.iter() {
|
||
if block["type"].as_str() == Some("quote") {
|
||
let content = block["content"].as_array().expect("quote content");
|
||
assert!(
|
||
content.is_empty(),
|
||
"空引用块 content 应为空,实际 {:?}",
|
||
content
|
||
);
|
||
}
|
||
}
|
||
}
|
||
|
||
/// 固定空表格单元格行为:comrak 解析空单元格为空的 tiptap paragraph,
|
||
/// content 字段为 []。
|
||
#[test]
|
||
fn markdown_table_empty_cells_parse() {
|
||
let blocks = markdown_to_blocks("| 左 | |\n| --- | --- |\n| | 右 |\n");
|
||
let array = blocks.as_array().expect("blocks");
|
||
let table = array
|
||
.iter()
|
||
.find(|b| b["type"].as_str() == Some("table"))
|
||
.expect("should have table block");
|
||
let tiptap_table = table["props"]["tiptapTable"]
|
||
.as_object()
|
||
.expect("tiptapTable object");
|
||
let content_rows = tiptap_table["content"]
|
||
.as_array()
|
||
.expect("table content rows");
|
||
// 表头行:两个单元格
|
||
let header_row = &content_rows[0]["content"].as_array().expect("header row cells");
|
||
assert_eq!(header_row.len(), 2);
|
||
// 表头第一个单元格(非空)
|
||
let h1_cell = &header_row[0]["content"].as_array().expect("header cell paragraphs");
|
||
assert!(!h1_cell.is_empty(), "表头第一个单元格不应为空");
|
||
// 表头第二个单元格(空):单元格的 content 为 [{type:"paragraph", content:[]}]
|
||
// 段落层的 content 应为空数组
|
||
let h2_par_content = &header_row[1]["content"][0]["content"];
|
||
let h2_inline = h2_par_content.as_array().expect("header cell para content");
|
||
assert!(
|
||
h2_inline.is_empty(),
|
||
"空表头单元格的 paragraph content 应为空,实际值: {}",
|
||
serde_json::to_string_pretty(h2_par_content).unwrap()
|
||
);
|
||
|
||
// 数据行
|
||
let data_row = &content_rows[1]["content"].as_array().expect("data row cells");
|
||
assert_eq!(data_row.len(), 2);
|
||
// 数据行第一个单元格(空):段落层的 content 应为空数组
|
||
let d1_par_content = &data_row[0]["content"][0]["content"];
|
||
let d1_inline = d1_par_content.as_array().expect("data cell para content");
|
||
assert!(
|
||
d1_inline.is_empty(),
|
||
"空数据单元格的 paragraph content 应为空,实际值: {}",
|
||
serde_json::to_string_pretty(d1_par_content).unwrap()
|
||
);
|
||
// 数据行第二个单元格(非空):段落层的 content 不应为空
|
||
let d2_par_content = &data_row[1]["content"][0]["content"];
|
||
let d2_inline = d2_par_content.as_array().expect("data cell para content");
|
||
assert!(!d2_inline.is_empty(), "非空数据单元格不应为空");
|
||
}
|
||
}
|