diff --git a/src/parser/markdown.rs b/src/parser/markdown.rs index 3978d053..90f40cd1 100644 --- a/src/parser/markdown.rs +++ b/src/parser/markdown.rs @@ -20,28 +20,32 @@ static FENCED_LANG_RE: LazyLock = /// CommonMark list marker: 0–3 spaces, then `-`/`*`/`+` or 1–9 digits plus /// `.`/`)`, then a space, a tab, or the empty rest of the line (spec 0.31.2 -/// sec 5.2). Ten or more digits is prose. Four or more spaces is indented +/// sec 5.2). Pandoc example lists use `(@)` / `(@label)` the same way. +/// Ten or more digits is prose. Four or more spaces is indented /// code, not a list (spec 0.31.2 ex. 289). Empty markers do not interrupt /// a paragraph (`-` after prose is setext; GitHub #326). Tab-padded (`-\t`) /// and two-space-padded (`- `) empty markers are list items so the next /// unindented line is not lazy-joined (GitHub #337). One-space and /// three-or-more spaces stay as in #329. -static LIST_ITEM_RE: LazyLock = - LazyLock::new(|| Regex::new(r"^( {0,3}(?:[-*+]|\d{1,9}[.)])(?:[ \t]|$))(.*)$").unwrap()); +static LIST_ITEM_RE: LazyLock = LazyLock::new(|| { + Regex::new(r"^( {0,3}(?:[-*+]|\d{1,9}[.)]|\(@[A-Za-z0-9]*\))(?:[ \t]|$))(.*)$").unwrap() +}); /// List-looking line at any indent (including 4+ spaces). LIST_ITEM_RE is /// 0–3 only; a 4-space dash is indented code, but after a blank we still /// need the shape so hang-relative close can hand it to snapper-tupp. /// Digit cap matches LIST_ITEM_RE (CommonMark 1–9). Empty rest of line, /// including a tab after the marker, is still list-looking (sec 5.2 / #337). -static LIST_LOOKING_RE: LazyLock = - LazyLock::new(|| Regex::new(r"^[\t ]*(?:[-*+]|\d{1,9}[.)])(?:[ \t]|$)").unwrap()); +static LIST_LOOKING_RE: LazyLock = LazyLock::new(|| { + Regex::new(r"^[\t ]*(?:[-*+]|\d{1,9}[.)]|\(@[A-Za-z0-9]*\))(?:[ \t]|$)").unwrap() +}); /// List item at any indent. LIST_ITEM_RE is 0–3 only so a 4-space dash is /// document-level indented code. Inside a parent item, hang ≤ indent < hang+4 /// is a nested list (pulldown / CM 5.2), not code. -static LIST_ITEM_ANY_INDENT_RE: LazyLock = - LazyLock::new(|| Regex::new(r"^([\t ]*(?:[-*+]|\d{1,9}[.)])(?:[ \t]|$))(.*)$").unwrap()); +static LIST_ITEM_ANY_INDENT_RE: LazyLock = LazyLock::new(|| { + Regex::new(r"^([\t ]*(?:[-*+]|\d{1,9}[.)]|\(@[A-Za-z0-9]*\))(?:[ \t]|$))(.*)$").unwrap() +}); /// Match a markdown table row: line whose trimmed form starts and ends with `|`. /// Also matches separator rows like `|---|---|`. @@ -1360,6 +1364,11 @@ fn list_opener_hang(line: &str) -> Option { return None; } let token = marker.trim(); + // Pandoc `(@)` / `(@label)` interrupts like a bullet. It is not an + // ordered marker, so the start-at-1 rule does not apply. + if token.starts_with("(@") && token.ends_with(')') { + return Some(list_marker_hang(marker)); + } if !token.starts_with(['-', '*', '+']) { let digits = token.trim_end_matches(['.', ')']); if digits.parse::().ok()? != 1 { @@ -1727,6 +1736,20 @@ fn is_gfm_table_row(line: &str) -> bool { gfm_table_cells(line).is_some() } +/// Pandoc line block: a line that starts with `|` and a space, and does +/// not end with a pipe. A flanking-pipe row is a table ([`TABLE_ROW_RE`]), +/// not a line block. `|cell` (no space) is prose. +fn is_pandoc_line_block(line: &str) -> bool { + if line_indent(line) > 3 { + return false; + } + if TABLE_ROW_RE.is_match(line) { + return false; + } + let rest = line.trim_start(); + rest.starts_with("| ") || rest == "|" +} + /// Last line of a GFM table starting at `start` (header), if the next line /// is a delimiter with a matching cell count. Data rows may omit flanking /// pipes (GFM 4.10 / pulldown `ENABLE_TABLES`, ex. 199). @@ -1757,7 +1780,7 @@ fn is_setext_title_line(line: &str) -> bool { if HEADING_RE.is_match(line) { return false; } - if TABLE_ROW_RE.is_match(line) { + if TABLE_ROW_RE.is_match(line) || is_pandoc_line_block(line) { return false; } if list_interrupts_paragraph(line) || quote_marker_depth(line) > 0 { @@ -2643,6 +2666,27 @@ impl FormatParser for MarkdownParser { continue; } + // Pandoc line block. No closing pipe, so TABLE_ROW_RE misses it + // and the line would otherwise be prose. The whole line stays + // Structure: an interior period must not sentence-split it or + // join the next `|` line. A later prose paragraph still splits. + if is_pandoc_line_block(line_text) { + close_list_item( + &mut in_list_item, + &mut list_hang, + &mut current_prose, + &mut prose_span, + &mut list_term, + &mut in_definition_list, + input, + &mut regions, + ); + flush_prose_spanned(&mut current_prose, &mut prose_span, &mut regions); + regions.push(SpannedRegion::structure(input, line.span())); + i += 1; + continue; + } + // HTML comment block (not a snapper pragma — those are handled above). if starts_html_comment(line_text) { close_list_item( diff --git a/src/reflow.rs b/src/reflow.rs index 17ee5074..bd46fb12 100644 --- a/src/reflow.rs +++ b/src/reflow.rs @@ -1235,8 +1235,16 @@ fn hanging_indent_width(s: &str) -> usize { let is_ordered = (core.ends_with('.') || core.ends_with(')')) && core.len() > 1 && core[..core.len() - 1].bytes().all(|b| b.is_ascii_digit()); + // Pandoc example list (`(@) `, `(@good) `). Not an ordered marker: + // the body after `@` is a label, not digits. + let is_example = core.starts_with("(@") + && core.ends_with(')') + && core.len() >= 3 + && core[2..core.len() - 1] + .bytes() + .all(|b| b.is_ascii_alphanumeric()); // LaTeX `\item` / `\item[label]` (not `\itemize`). - if is_bullet || is_ordered || is_rst_autoenum || is_latex_item_core(core) { + if is_bullet || is_ordered || is_example || is_rst_autoenum || is_latex_item_core(core) { s.chars().count() } else { 0 @@ -1979,6 +1987,16 @@ They are endowed with reason and conscience and should act towards one another i assert_eq!(result, "- One.\n Two.\n"); } + #[test] + fn pandoc_example_list_hangs_second_sentence() { + let result = reflow_regions(vec![ + Region::Structure("(@) ".to_string()), + Region::Prose("Example one. Second sentence.".to_string()), + Region::Structure("\n".to_string()), + ]); + assert_eq!(result, "(@) Example one.\n Second sentence.\n"); + } + #[test] fn md_definition_list_hangs_second_sentence() { let result = reflow_regions(vec![ diff --git a/src/sentence/unicode.rs b/src/sentence/unicode.rs index 35823ceb..82d2834f 100644 --- a/src/sentence/unicode.rs +++ b/src/sentence/unicode.rs @@ -45,16 +45,28 @@ static INLINE_TOKEN_RE: LazyLock = LazyLock::new(|| { r"\[[^\]]+\]\([^)]+\)", // Markdown links: [text](url) r"!\[[^\]]*\]\([^)]+\)", // Markdown images: ![alt](url) // CommonMark 0.31.2 §6.3 full / collapsed reference links. - // The label must follow the text immediately. Shortcut `[text]` - // is not matched: that would swallow every bracket group. + // The label must follow the text immediately. A bare `[text]` + // with no interior sentence punctuation is not a token: that + // would swallow `[fn:1]` and every other bracket group. + // A shortcut whose label contains `.` `!` or `?` is one span. r"!\[[^\]]*\]\[[^\]]*\]", // Markdown reference images: ![alt][ref] r"\[[^\]]+\]\[[^\]]*\]", // Markdown reference links: [text][ref] + // Pandoc citation: `[@doe2020]` / `[see @doe2020, pp. 33]`. + // `@` is the key, not an email glued to the previous word. + // Org `[cite:…]` is already a token above. + r"\[(?:[^\[\]\n]*[;\s])?-?@[A-Za-z][^\[\]\n]*\]", + // Shortcut reference whose label holds a sentence boundary. + // No look-around: the regex crate rejects it. `[fn:1]` has no + // `.` `!` or `?`, and `[cite:…]` / `[fn::…]` match earlier. + r"\[[^\[\]\n]*[.!?][^\[\]\n]*\]", r"\$\$[^$\n]+\$\$", // Display math: $$...$$ // org-element-latex-fragment-parser: after `$` the next char // is not space/tab/newline/`,`/`.`/`;`; the char before the - // closer is not space/tab/newline/`,`/`.`. `$ x. Next $` is - // leftover prose. `$a. b$` stays a fragment. - r"\$[^\s,.;$\n](?:[^$\n]*[^\s,.$\n])?\$", + // closer is not space/tab/newline/`,`. A `.` may sit against + // the closer (`$See this. Then that.$`, pandoc tex_math_dollars). + // `$ x. Next $` is leftover prose. `$a. b$` and `$x = 3.14$` + // stay fragments. + r"\$[^\s,.;$\n](?:[^$\n]*[^\s,$\n])?\$", // org-element-latex-fragment-parser: \(...\) / \[...\] search // to the closer. `[^\\\n]` dropped interior `\alpha` / `\beta`. r"\\\([^\n]+?\\\)", // LaTeX inline math: \(...\) @@ -7746,6 +7758,60 @@ mod tests { ); } + #[test] + fn dollar_math_closed_by_period_stays_one_span() { + let math = "$See this. Then that.$"; + let text = "See $See this. Then that.$ today. Next sentence."; + let (_, placeholders) = protect_inline_tokens(text); + assert!( + placeholders.iter().any(|p| p == math), + "period-closed math must be one token, got {placeholders:?}" + ); + assert_eq!( + split(text), + vec![ + "See $See this. Then that.$ today.".to_string(), + "Next sentence.".to_string() + ] + ); + } + + #[test] + fn shortcut_reference_label_stays_one_span() { + let link = "[Theorem. Proof]"; + let text = "See [Theorem. Proof] for details. Next sentence."; + let (_, placeholders) = protect_inline_tokens(text); + assert!( + placeholders.iter().any(|p| p == link), + "shortcut label must be one token, got {placeholders:?}" + ); + assert_eq!( + split(text), + vec![ + "See [Theorem. Proof] for details.".to_string(), + "Next sentence.".to_string() + ] + ); + } + + #[test] + fn pandoc_citation_stays_one_span() { + let cite = "[@doe2020, see this. Then that]"; + let text = "See [@doe2020, see this. Then that] for details. Next sentence."; + let (_, placeholders) = protect_inline_tokens(text); + assert!( + placeholders.iter().any(|p| p == cite), + "pandoc citation must be one token, got {placeholders:?}" + ); + assert_eq!( + split(text), + vec![ + "See [@doe2020, see this. Then that] for details.".to_string(), + "Next sentence.".to_string() + ] + ); + } + #[test] fn inline_markdown_link_preserved() { assert_eq!( diff --git a/tests/md_dollar_period.rs b/tests/md_dollar_period.rs new file mode 100644 index 00000000..388ebdab --- /dev/null +++ b/tests/md_dollar_period.rs @@ -0,0 +1,47 @@ +//! `$…$` whose closer is preceded by `.` stays one span, as does +//! `$x = 3.14$`. A sentence outside the dollars still splits. + +use snapper_fmt::format::Format; +use snapper_fmt::{FormatConfig, format_text}; + +fn md_cfg() -> FormatConfig { + FormatConfig { + format: Format::Markdown, + max_width: 0, + ..Default::default() + } + .without_safety_backstops() +} + +#[test] +fn dollar_closed_by_period_stays_whole_and_outside_splits() { + let input = "See $See this. Then that.$ today. Next sentence.\n"; + let out = format_text(input, &md_cfg()).unwrap(); + assert!( + out.contains("$See this. Then that.$"), + "period-closed math must stay one span, got:\n{out}" + ); + assert!( + !out.contains("$See this.\n"), + "must not split inside the dollars, got:\n{out}" + ); + assert!( + out.contains("today.\nNext sentence.\n"), + "a sentence outside the dollars must still split, got:\n{out}" + ); + assert_eq!(format_text(&out, &md_cfg()).unwrap(), out); +} + +#[test] +fn decimal_dollar_math_stays_whole() { + let input = "The value $x = 3.14$ matters. Next sentence.\n"; + let out = format_text(input, &md_cfg()).unwrap(); + assert!( + out.contains("$x = 3.14$"), + "decimal math must stay one span, got:\n{out}" + ); + assert!( + out.contains("matters.\nNext sentence.\n"), + "a sentence outside the dollars must still split, got:\n{out}" + ); +} diff --git a/tests/md_example_list.rs b/tests/md_example_list.rs new file mode 100644 index 00000000..598bb631 --- /dev/null +++ b/tests/md_example_list.rs @@ -0,0 +1,51 @@ +//! Pandoc example list `(@)` is a list marker. The marker stays on the +//! item line. A second sentence may hang on a continuation line and +//! must not drop to column 0. + +use snapper_fmt::format::Format; +use snapper_fmt::parser::markdown::MarkdownParser; +use snapper_fmt::parser::{FormatParser, Region}; +use snapper_fmt::{FormatConfig, format_text}; + +fn md_cfg() -> FormatConfig { + FormatConfig { + format: Format::Markdown, + max_width: 0, + ..Default::default() + } + .without_safety_backstops() +} + +#[test] +fn example_marker_is_structure_and_body_stays_prose() { + let input = "(@) Example one. Second sentence.\n"; + let regions = MarkdownParser.parse(input); + assert!( + regions + .iter() + .any(|r| matches!(r, Region::Structure(s) if s == "(@) ")), + "(@) marker must be Structure, got {regions:?}" + ); + assert!( + regions.iter().any(|r| matches!( + r, + Region::Prose(p) if p.contains("Example one.") && p.contains("Second sentence.") + )), + "item body must stay Prose, got {regions:?}" + ); +} + +#[test] +fn example_second_sentence_hangs_inside_the_item() { + let input = "(@) Example one. Second sentence.\n"; + let out = format_text(input, &md_cfg()).unwrap(); + assert_eq!( + out, "(@) Example one.\n Second sentence.\n", + "second sentence must hang under (@), got:\n{out}" + ); + assert!( + !out.contains("\nSecond sentence."), + "second sentence must not leave the item, got:\n{out}" + ); + assert_eq!(format_text(&out, &md_cfg()).unwrap(), out); +} diff --git a/tests/md_pandoc_citation.rs b/tests/md_pandoc_citation.rs new file mode 100644 index 00000000..61a2f49b --- /dev/null +++ b/tests/md_pandoc_citation.rs @@ -0,0 +1,33 @@ +//! Pandoc citation `[@doe2020, see this. Then that]` is one span. +//! A sentence outside it still splits. + +use snapper_fmt::format::Format; +use snapper_fmt::{FormatConfig, format_text}; + +fn md_cfg() -> FormatConfig { + FormatConfig { + format: Format::Markdown, + max_width: 0, + ..Default::default() + } + .without_safety_backstops() +} + +#[test] +fn pandoc_citation_stays_one_span_and_outside_splits() { + let input = "See [@doe2020, see this. Then that] for details. Next sentence.\n"; + let out = format_text(input, &md_cfg()).unwrap(); + assert!( + out.contains("[@doe2020, see this. Then that]"), + "citation must stay one span, got:\n{out}" + ); + assert!( + !out.contains("see this.\n"), + "must not split inside the citation, got:\n{out}" + ); + assert!( + out.contains("for details.\nNext sentence.\n"), + "a sentence outside the citation must still split, got:\n{out}" + ); + assert_eq!(format_text(&out, &md_cfg()).unwrap(), out); +} diff --git a/tests/md_pandoc_line_block.rs b/tests/md_pandoc_line_block.rs new file mode 100644 index 00000000..42bc219f --- /dev/null +++ b/tests/md_pandoc_line_block.rs @@ -0,0 +1,78 @@ +//! Pandoc line block: `| ` with no closing pipe is not a table row and +//! not a paragraph. The line stays whole. A following prose sentence +//! still splits. A flanking-pipe row stays a table. + +use snapper_fmt::format::Format; +use snapper_fmt::parser::markdown::MarkdownParser; +use snapper_fmt::parser::{FormatParser, Region}; +use snapper_fmt::{FormatConfig, format_text}; + +fn md_cfg() -> FormatConfig { + FormatConfig { + format: Format::Markdown, + max_width: 0, + ..Default::default() + } + .without_safety_backstops() +} + +fn ticket_fixture() -> &'static str { + concat!( + "| See Dr. Smith. He left.\n", + "After the block. Still prose.\n", + ) +} + +#[test] +fn line_block_is_structure_not_prose() { + let regions = MarkdownParser.parse(ticket_fixture()); + assert!( + regions.iter().any(|r| matches!( + r, + Region::Structure(s) if s.trim_end() == "| See Dr. Smith. He left." + )), + "line block must be one Structure line, got {regions:?}" + ); + assert!( + !regions.iter().any(|r| matches!( + r, + Region::Prose(p) if p.contains("Dr. Smith") + )), + "line block must not be Prose, got {regions:?}" + ); +} + +#[test] +fn line_block_stays_one_line_and_following_prose_splits() { + let input = ticket_fixture(); + let out = format_text(input, &md_cfg()).unwrap(); + assert!( + out.contains("| See Dr. Smith. He left.\n"), + "line block must stay one line, got:\n{out}" + ); + assert!( + !out.contains("| See Dr. Smith.\n"), + "must not sentence-split the line block, got:\n{out}" + ); + assert!( + out.contains("After the block.\nStill prose.\n"), + "following prose must still split, got:\n{out}" + ); + assert_eq!(format_text(&out, &md_cfg()).unwrap(), out); +} + +#[test] +fn flanking_pipe_table_row_stays_a_table() { + let input = concat!( + "| Dr. Smith | He left. |\n", + "| --- | --- |\n", + "| A. | B. |\n", + ); + let regions = MarkdownParser.parse(input); + assert!( + regions.iter().all(|r| matches!(r, Region::Structure(_))), + "closing-pipe rows must stay Structure, got {regions:?}" + ); + let out = format_text(input, &md_cfg()).unwrap(); + assert_eq!(out, input, "table bytes must stay, got:\n{out}"); +} diff --git a/tests/md_shortcut_ref_label.rs b/tests/md_shortcut_ref_label.rs new file mode 100644 index 00000000..9997eac2 --- /dev/null +++ b/tests/md_shortcut_ref_label.rs @@ -0,0 +1,34 @@ +//! CommonMark shortcut reference `[Theorem. Proof]` is one span. +//! A sentence outside it still splits. Full and collapsed references +//! stay on their existing path. + +use snapper_fmt::format::Format; +use snapper_fmt::{FormatConfig, format_text}; + +fn md_cfg() -> FormatConfig { + FormatConfig { + format: Format::Markdown, + max_width: 0, + ..Default::default() + } + .without_safety_backstops() +} + +#[test] +fn shortcut_label_stays_one_span_and_outside_splits() { + let input = "See [Theorem. Proof] for details. Next sentence.\n"; + let out = format_text(input, &md_cfg()).unwrap(); + assert!( + out.contains("[Theorem. Proof]"), + "shortcut label must stay one span, got:\n{out}" + ); + assert!( + !out.contains("[Theorem.\n") && !out.contains("Proof]\nfor"), + "must not split inside the shortcut label, got:\n{out}" + ); + assert!( + out.contains("for details.\nNext sentence.\n"), + "a sentence outside the label must still split, got:\n{out}" + ); + assert_eq!(format_text(&out, &md_cfg()).unwrap(), out); +}