diff --git a/src/parse.rs b/src/parse.rs index 2273196..6750db6 100644 --- a/src/parse.rs +++ b/src/parse.rs @@ -1,4 +1,7 @@ -use typst_syntax::ast::{self, AstNode, Expr}; +use typst_syntax::{ + ast::{self, AstNode, Expr}, + is_newline, +}; use crate::Block; @@ -68,12 +71,28 @@ pub fn parse(source: &str) -> Vec { // ---- inline elements (accumulate into paragraph) ---- Expr::Space(_) => { let text = node_text(&expr); - // Two consecutive Space("\n") nodes arise only when a LineComment - // was dropped between them (bare \n\n is always a Parbreak token, - // never two Space nodes). Skip the second \n to avoid producing - // \n\n in the reconstructed source, which Typst would treat as a - // paragraph break. - if !(text == "\n" && paragraph_buf.ends_with('\n')) { + // Two consecutive Space nodes spanning a newline arise only when + // a comment was dropped between them (bare \n\n is always a + // Parbreak token, never two Space nodes). Reconstructing them + // verbatim would leave a blank or whitespace-only line, which + // Typst treats as a paragraph break. Drop the buffer's trailing + // indent and the duplicate newline so the two lines stay in one + // paragraph. The buffer may end in a newline plus indent when + // the dropped comment was itself indented. + let joined = match strip_newline_prefix(&text) { + Some(rest) => { + let line_start = paragraph_buf.trim_end_matches([' ', '\t']); + if line_start.chars().next_back().is_some_and(is_newline) { + paragraph_buf.truncate(line_start.len()); + paragraph_buf.push_str(rest); + true + } else { + false + } + } + None => false, + }; + if !joined { paragraph_buf.push_str(&text); } } @@ -139,6 +158,16 @@ fn node_text(expr: &Expr<'_>) -> String { } } +/// Strip one Typst newline, treating CRLF as a single sequence. +fn strip_newline_prefix(text: &str) -> Option<&str> { + let first = text.chars().next().filter(|&c| is_newline(c))?; + let mut len = first.len_utf8(); + if first == '\r' && text[len..].starts_with('\n') { + len += '\n'.len_utf8(); + } + Some(&text[len..]) +} + /// Flush the paragraph buffer into blocks if non-empty. fn flush_paragraph(buf: &mut String, blocks: &mut Vec) { let trimmed = buf.trim(); @@ -240,6 +269,67 @@ mod tests { } } + #[test] + fn test_parse_line_comment_keeps_single_paragraph() { + let blocks = parse("First.\n// a comment\nSecond.\n"); + assert_eq!(blocks.len(), 1, "expected 1 block, got: {blocks:?}"); + assert!( + matches!(&blocks[0], Block::Paragraph { source_text } if source_text == "First.\nSecond.") + ); + } + + #[test] + fn test_parse_indented_line_comment_keeps_single_paragraph() { + let blocks = parse("First.\n // indented comment\nSecond.\n"); + assert_eq!(blocks.len(), 1, "expected 1 block, got: {blocks:?}"); + assert!( + matches!(&blocks[0], Block::Paragraph { source_text } if source_text == "First.\nSecond.") + ); + } + + #[test] + fn test_parse_indented_line_comment_with_each_newline_form() { + for newline in [ + "\n", "\x0B", "\x0C", "\r", "\r\n", "\u{0085}", "\u{2028}", "\u{2029}", + ] { + let source = format!("First.{newline} // indented comment{newline}Second.{newline}"); + let blocks = parse(&source); + let expected = format!("First.{newline}Second."); + assert_eq!( + blocks, + vec![Block::Paragraph { + source_text: expected, + }], + "newline: {newline:?}" + ); + } + } + + #[test] + fn test_parse_consecutive_indented_comments_keep_single_paragraph() { + let blocks = parse("First.\n // c1\n // c2\nSecond.\n"); + assert_eq!(blocks.len(), 1, "expected 1 block, got: {blocks:?}"); + assert!( + matches!(&blocks[0], Block::Paragraph { source_text } if source_text == "First.\nSecond.") + ); + } + + #[test] + fn test_parse_blank_line_before_comment_keeps_paragraph_break() { + let blocks = parse("First.\n\n// a comment\nSecond.\n"); + assert!( + blocks + .iter() + .any(|b| matches!(b, Block::Paragraph { source_text } if source_text == "First.")) + ); + assert!(blocks.iter().any(|b| matches!(b, Block::Parbreak))); + assert!( + blocks + .iter() + .any(|b| matches!(b, Block::Paragraph { source_text } if source_text == "Second.")) + ); + } + #[test] fn test_parse_label_at_paragraph_start_as_own_block() { let blocks = parse("= Sample\n\n\nAlpha beta.\n"); diff --git a/tests/integration.rs b/tests/integration.rs index f52c925..dacad43 100644 --- a/tests/integration.rs +++ b/tests/integration.rs @@ -198,3 +198,14 @@ after the field log was reconciled. assert!(output.contains("#ref(, supplement: [])")); assert!(!output.contains("#ref(\\")); } + +#[test] +fn test_indented_line_comment_does_not_split_paragraph() { + let old = "First sentence.\n // indented comment\nSecond sentence.\n"; + let new = "First sentence.\n // indented comment\nSecond sentence changed.\n"; + let output = run_diff(old, new); + + assert!(output.contains("First sentence.\nSecond sentence")); + // A whitespace-only line would be treated as a paragraph break by Typst. + assert!(!output.contains("\n \n")); +}