Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
60 changes: 52 additions & 8 deletions src/parser/markdown.rs
Original file line number Diff line number Diff line change
Expand Up @@ -20,28 +20,32 @@ static FENCED_LANG_RE: LazyLock<Regex> =

/// CommonMark list marker: 0–3 spaces, then `-`/`*`/`+` or 1–9 digits plus
/// `.`/`)`, then a space, a tab, or the empty rest of the line (spec 0.31.2
/// sec 5.2). Ten or more digits is prose. Four or more spaces is indented
/// sec 5.2). Pandoc example lists use `(@)` / `(@label)` the same way.
/// Ten or more digits is prose. Four or more spaces is indented
/// code, not a list (spec 0.31.2 ex. 289). Empty markers do not interrupt
/// a paragraph (`-` after prose is setext; GitHub #326). Tab-padded (`-\t`)
/// and two-space-padded (`- `) empty markers are list items so the next
/// unindented line is not lazy-joined (GitHub #337). One-space and
/// three-or-more spaces stay as in #329.
static LIST_ITEM_RE: LazyLock<Regex> =
LazyLock::new(|| Regex::new(r"^( {0,3}(?:[-*+]|\d{1,9}[.)])(?:[ \t]|$))(.*)$").unwrap());
static LIST_ITEM_RE: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(r"^( {0,3}(?:[-*+]|\d{1,9}[.)]|\(@[A-Za-z0-9]*\))(?:[ \t]|$))(.*)$").unwrap()
});

/// List-looking line at any indent (including 4+ spaces). LIST_ITEM_RE is
/// 0–3 only; a 4-space dash is indented code, but after a blank we still
/// need the shape so hang-relative close can hand it to snapper-tupp.
/// Digit cap matches LIST_ITEM_RE (CommonMark 1–9). Empty rest of line,
/// including a tab after the marker, is still list-looking (sec 5.2 / #337).
static LIST_LOOKING_RE: LazyLock<Regex> =
LazyLock::new(|| Regex::new(r"^[\t ]*(?:[-*+]|\d{1,9}[.)])(?:[ \t]|$)").unwrap());
static LIST_LOOKING_RE: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(r"^[\t ]*(?:[-*+]|\d{1,9}[.)]|\(@[A-Za-z0-9]*\))(?:[ \t]|$)").unwrap()
});

/// List item at any indent. LIST_ITEM_RE is 0–3 only so a 4-space dash is
/// document-level indented code. Inside a parent item, hang ≤ indent < hang+4
/// is a nested list (pulldown / CM 5.2), not code.
static LIST_ITEM_ANY_INDENT_RE: LazyLock<Regex> =
LazyLock::new(|| Regex::new(r"^([\t ]*(?:[-*+]|\d{1,9}[.)])(?:[ \t]|$))(.*)$").unwrap());
static LIST_ITEM_ANY_INDENT_RE: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(r"^([\t ]*(?:[-*+]|\d{1,9}[.)]|\(@[A-Za-z0-9]*\))(?:[ \t]|$))(.*)$").unwrap()
});

/// Match a markdown table row: line whose trimmed form starts and ends with `|`.
/// Also matches separator rows like `|---|---|`.
Expand Down Expand Up @@ -1360,6 +1364,11 @@ fn list_opener_hang(line: &str) -> Option<usize> {
return None;
}
let token = marker.trim();
// Pandoc `(@)` / `(@label)` interrupts like a bullet. It is not an
// ordered marker, so the start-at-1 rule does not apply.
if token.starts_with("(@") && token.ends_with(')') {
return Some(list_marker_hang(marker));
}
if !token.starts_with(['-', '*', '+']) {
let digits = token.trim_end_matches(['.', ')']);
if digits.parse::<u32>().ok()? != 1 {
Expand Down Expand Up @@ -1727,6 +1736,20 @@ fn is_gfm_table_row(line: &str) -> bool {
gfm_table_cells(line).is_some()
}

/// Pandoc line block: a line that starts with `|` and a space, and does
/// not end with a pipe. A flanking-pipe row is a table ([`TABLE_ROW_RE`]),
/// not a line block. `|cell` (no space) is prose.
fn is_pandoc_line_block(line: &str) -> bool {
if line_indent(line) > 3 {
return false;
}
if TABLE_ROW_RE.is_match(line) {
return false;
}
let rest = line.trim_start();
rest.starts_with("| ") || rest == "|"
}

/// Last line of a GFM table starting at `start` (header), if the next line
/// is a delimiter with a matching cell count. Data rows may omit flanking
/// pipes (GFM 4.10 / pulldown `ENABLE_TABLES`, ex. 199).
Expand Down Expand Up @@ -1757,7 +1780,7 @@ fn is_setext_title_line(line: &str) -> bool {
if HEADING_RE.is_match(line) {
return false;
}
if TABLE_ROW_RE.is_match(line) {
if TABLE_ROW_RE.is_match(line) || is_pandoc_line_block(line) {
return false;
}
if list_interrupts_paragraph(line) || quote_marker_depth(line) > 0 {
Expand Down Expand Up @@ -2643,6 +2666,27 @@ impl FormatParser for MarkdownParser {
continue;
}

// Pandoc line block. No closing pipe, so TABLE_ROW_RE misses it
// and the line would otherwise be prose. The whole line stays
// Structure: an interior period must not sentence-split it or
// join the next `|` line. A later prose paragraph still splits.
if is_pandoc_line_block(line_text) {
close_list_item(
&mut in_list_item,
&mut list_hang,
&mut current_prose,
&mut prose_span,
&mut list_term,
&mut in_definition_list,
input,
&mut regions,
);
flush_prose_spanned(&mut current_prose, &mut prose_span, &mut regions);
regions.push(SpannedRegion::structure(input, line.span()));
i += 1;
continue;
}

// HTML comment block (not a snapper pragma — those are handled above).
if starts_html_comment(line_text) {
close_list_item(
Expand Down
20 changes: 19 additions & 1 deletion src/reflow.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1235,8 +1235,16 @@ fn hanging_indent_width(s: &str) -> usize {
let is_ordered = (core.ends_with('.') || core.ends_with(')'))
&& core.len() > 1
&& core[..core.len() - 1].bytes().all(|b| b.is_ascii_digit());
// Pandoc example list (`(@) `, `(@good) `). Not an ordered marker:
// the body after `@` is a label, not digits.
let is_example = core.starts_with("(@")
&& core.ends_with(')')
&& core.len() >= 3
&& core[2..core.len() - 1]
.bytes()
.all(|b| b.is_ascii_alphanumeric());
// LaTeX `\item` / `\item[label]` (not `\itemize`).
if is_bullet || is_ordered || is_rst_autoenum || is_latex_item_core(core) {
if is_bullet || is_ordered || is_example || is_rst_autoenum || is_latex_item_core(core) {
s.chars().count()
} else {
0
Expand Down Expand Up @@ -1979,6 +1987,16 @@ They are endowed with reason and conscience and should act towards one another i
assert_eq!(result, "- One.\n Two.\n");
}

#[test]
fn pandoc_example_list_hangs_second_sentence() {
let result = reflow_regions(vec![
Region::Structure("(@) ".to_string()),
Region::Prose("Example one. Second sentence.".to_string()),
Region::Structure("\n".to_string()),
]);
assert_eq!(result, "(@) Example one.\n Second sentence.\n");
}

#[test]
fn md_definition_list_hangs_second_sentence() {
let result = reflow_regions(vec![
Expand Down
76 changes: 71 additions & 5 deletions src/sentence/unicode.rs
Original file line number Diff line number Diff line change
Expand Up @@ -45,16 +45,28 @@ static INLINE_TOKEN_RE: LazyLock<Regex> = LazyLock::new(|| {
r"\[[^\]]+\]\([^)]+\)", // Markdown links: [text](url)
r"!\[[^\]]*\]\([^)]+\)", // Markdown images: ![alt](url)
// CommonMark 0.31.2 §6.3 full / collapsed reference links.
// The label must follow the text immediately. Shortcut `[text]`
// is not matched: that would swallow every bracket group.
// The label must follow the text immediately. A bare `[text]`
// with no interior sentence punctuation is not a token: that
// would swallow `[fn:1]` and every other bracket group.
// A shortcut whose label contains `.` `!` or `?` is one span.
r"!\[[^\]]*\]\[[^\]]*\]", // Markdown reference images: ![alt][ref]
r"\[[^\]]+\]\[[^\]]*\]", // Markdown reference links: [text][ref]
// Pandoc citation: `[@doe2020]` / `[see @doe2020, pp. 33]`.
// `@` is the key, not an email glued to the previous word.
// Org `[cite:…]` is already a token above.
r"\[(?:[^\[\]\n]*[;\s])?-?@[A-Za-z][^\[\]\n]*\]",
// Shortcut reference whose label holds a sentence boundary.
// No look-around: the regex crate rejects it. `[fn:1]` has no
// `.` `!` or `?`, and `[cite:…]` / `[fn::…]` match earlier.
r"\[[^\[\]\n]*[.!?][^\[\]\n]*\]",
r"\$\$[^$\n]+\$\$", // Display math: $$...$$
// org-element-latex-fragment-parser: after `$` the next char
// is not space/tab/newline/`,`/`.`/`;`; the char before the
// closer is not space/tab/newline/`,`/`.`. `$ x. Next $` is
// leftover prose. `$a. b$` stays a fragment.
r"\$[^\s,.;$\n](?:[^$\n]*[^\s,.$\n])?\$",
// closer is not space/tab/newline/`,`. A `.` may sit against
// the closer (`$See this. Then that.$`, pandoc tex_math_dollars).
// `$ x. Next $` is leftover prose. `$a. b$` and `$x = 3.14$`
// stay fragments.
r"\$[^\s,.;$\n](?:[^$\n]*[^\s,$\n])?\$",
// org-element-latex-fragment-parser: \(...\) / \[...\] search
// to the closer. `[^\\\n]` dropped interior `\alpha` / `\beta`.
r"\\\([^\n]+?\\\)", // LaTeX inline math: \(...\)
Expand Down Expand Up @@ -7746,6 +7758,60 @@ mod tests {
);
}

#[test]
fn dollar_math_closed_by_period_stays_one_span() {
let math = "$See this. Then that.$";
let text = "See $See this. Then that.$ today. Next sentence.";
let (_, placeholders) = protect_inline_tokens(text);
assert!(
placeholders.iter().any(|p| p == math),
"period-closed math must be one token, got {placeholders:?}"
);
assert_eq!(
split(text),
vec![
"See $See this. Then that.$ today.".to_string(),
"Next sentence.".to_string()
]
);
}

#[test]
fn shortcut_reference_label_stays_one_span() {
let link = "[Theorem. Proof]";
let text = "See [Theorem. Proof] for details. Next sentence.";
let (_, placeholders) = protect_inline_tokens(text);
assert!(
placeholders.iter().any(|p| p == link),
"shortcut label must be one token, got {placeholders:?}"
);
assert_eq!(
split(text),
vec![
"See [Theorem. Proof] for details.".to_string(),
"Next sentence.".to_string()
]
);
}

#[test]
fn pandoc_citation_stays_one_span() {
let cite = "[@doe2020, see this. Then that]";
let text = "See [@doe2020, see this. Then that] for details. Next sentence.";
let (_, placeholders) = protect_inline_tokens(text);
assert!(
placeholders.iter().any(|p| p == cite),
"pandoc citation must be one token, got {placeholders:?}"
);
assert_eq!(
split(text),
vec![
"See [@doe2020, see this. Then that] for details.".to_string(),
"Next sentence.".to_string()
]
);
}

#[test]
fn inline_markdown_link_preserved() {
assert_eq!(
Expand Down
47 changes: 47 additions & 0 deletions tests/md_dollar_period.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,47 @@
//! `$…$` whose closer is preceded by `.` stays one span, as does
//! `$x = 3.14$`. A sentence outside the dollars still splits.

use snapper_fmt::format::Format;
use snapper_fmt::{FormatConfig, format_text};

fn md_cfg() -> FormatConfig {
FormatConfig {
format: Format::Markdown,
max_width: 0,
..Default::default()
}
.without_safety_backstops()
}

#[test]
fn dollar_closed_by_period_stays_whole_and_outside_splits() {
let input = "See $See this. Then that.$ today. Next sentence.\n";
let out = format_text(input, &md_cfg()).unwrap();
assert!(
out.contains("$See this. Then that.$"),
"period-closed math must stay one span, got:\n{out}"
);
assert!(
!out.contains("$See this.\n"),
"must not split inside the dollars, got:\n{out}"
);
assert!(
out.contains("today.\nNext sentence.\n"),
"a sentence outside the dollars must still split, got:\n{out}"
);
assert_eq!(format_text(&out, &md_cfg()).unwrap(), out);
}

#[test]
fn decimal_dollar_math_stays_whole() {
let input = "The value $x = 3.14$ matters. Next sentence.\n";
let out = format_text(input, &md_cfg()).unwrap();
assert!(
out.contains("$x = 3.14$"),
"decimal math must stay one span, got:\n{out}"
);
assert!(
out.contains("matters.\nNext sentence.\n"),
"a sentence outside the dollars must still split, got:\n{out}"
);
}
51 changes: 51 additions & 0 deletions tests/md_example_list.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,51 @@
//! Pandoc example list `(@)` is a list marker. The marker stays on the
//! item line. A second sentence may hang on a continuation line and
//! must not drop to column 0.

use snapper_fmt::format::Format;
use snapper_fmt::parser::markdown::MarkdownParser;
use snapper_fmt::parser::{FormatParser, Region};
use snapper_fmt::{FormatConfig, format_text};

fn md_cfg() -> FormatConfig {
FormatConfig {
format: Format::Markdown,
max_width: 0,
..Default::default()
}
.without_safety_backstops()
}

#[test]
fn example_marker_is_structure_and_body_stays_prose() {
let input = "(@) Example one. Second sentence.\n";
let regions = MarkdownParser.parse(input);
assert!(
regions
.iter()
.any(|r| matches!(r, Region::Structure(s) if s == "(@) ")),
"(@) marker must be Structure, got {regions:?}"
);
assert!(
regions.iter().any(|r| matches!(
r,
Region::Prose(p) if p.contains("Example one.") && p.contains("Second sentence.")
)),
"item body must stay Prose, got {regions:?}"
);
}

#[test]
fn example_second_sentence_hangs_inside_the_item() {
let input = "(@) Example one. Second sentence.\n";
let out = format_text(input, &md_cfg()).unwrap();
assert_eq!(
out, "(@) Example one.\n Second sentence.\n",
"second sentence must hang under (@), got:\n{out}"
);
assert!(
!out.contains("\nSecond sentence."),
"second sentence must not leave the item, got:\n{out}"
);
assert_eq!(format_text(&out, &md_cfg()).unwrap(), out);
}
33 changes: 33 additions & 0 deletions tests/md_pandoc_citation.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,33 @@
//! Pandoc citation `[@doe2020, see this. Then that]` is one span.
//! A sentence outside it still splits.

use snapper_fmt::format::Format;
use snapper_fmt::{FormatConfig, format_text};

fn md_cfg() -> FormatConfig {
FormatConfig {
format: Format::Markdown,
max_width: 0,
..Default::default()
}
.without_safety_backstops()
}

#[test]
fn pandoc_citation_stays_one_span_and_outside_splits() {
let input = "See [@doe2020, see this. Then that] for details. Next sentence.\n";
let out = format_text(input, &md_cfg()).unwrap();
assert!(
out.contains("[@doe2020, see this. Then that]"),
"citation must stay one span, got:\n{out}"
);
assert!(
!out.contains("see this.\n"),
"must not split inside the citation, got:\n{out}"
);
assert!(
out.contains("for details.\nNext sentence.\n"),
"a sentence outside the citation must still split, got:\n{out}"
);
assert_eq!(format_text(&out, &md_cfg()).unwrap(), out);
}
Loading
Loading