diff --git a/docs/orgmode/howto/vale-integration.org b/docs/orgmode/howto/vale-integration.org index 65905ff9..d6315b39 100644 --- a/docs/orgmode/howto/vale-integration.org +++ b/docs/orgmode/howto/vale-integration.org @@ -40,8 +40,8 @@ Level: warning. - =LongProseLine= :: Flags prose lines exceeding 120 characters. Level: suggestion. - =SnapperCheck= :: External action. - Runs =snapper --check --output-format json= and reports =fused= / =wrap= on the matching lines when =snapper= is on =PATH=. - Advisory =long= stays on =LongProseLine=. + Runs =snapper --check --output-format json= and reports =fused= / =wrap= on the matching lines when =snapper= is on =PATH=. + Advisory =long= stays on =LongProseLine=. * Precise CI enforcement :PROPERTIES: diff --git a/src/parser/org.rs b/src/parser/org.rs index f5cb89c8..4f2888c5 100644 --- a/src/parser/org.rs +++ b/src/parser/org.rs @@ -17,6 +17,55 @@ static LIST_ITEM_RE: LazyLock = LazyLock::new(|| { Regex::new(r"^(\s*(?:[-+]|\d+[.)])(?:[ \t]|$)|[ \t]+\*(?:[ \t]|$))(.*)$").unwrap() }); +/// org-syntax item: `tag :: description`. +/// +/// Bytes of `after_bullet` through the last `[ \t]+::` and the spaces +/// that follow. The tag stays on the item line; the description hangs +/// as prose. +/// `None` when the item has no tag (`::` needs whitespace before it, +/// and whitespace or end after it). +pub(crate) fn org_item_tag_end(after_bullet: &str) -> Option { + let bytes = after_bullet.as_bytes(); + let mut i = 0; + let mut found = None; + while i < bytes.len() { + if bytes[i] == b' ' || bytes[i] == b'\t' { + let mut j = i; + while j < bytes.len() && (bytes[j] == b' ' || bytes[j] == b'\t') { + j += 1; + } + if j + 1 < bytes.len() && bytes[j] == b':' && bytes[j + 1] == b':' { + let mut k = j + 2; + if k == bytes.len() || bytes[k] == b' ' || bytes[k] == b'\t' { + while k < bytes.len() && (bytes[k] == b' ' || bytes[k] == b'\t') { + k += 1; + } + found = Some(k); + } + } + i = j; + continue; + } + i += 1; + } + found +} + +/// Hang width when `s` is a list marker plus an item tag and nothing else. +/// Description reflow stays under `tag ::`. +pub(crate) fn org_item_tag_hang_width(s: &str) -> Option { + if !s.ends_with([' ', '\t']) { + return None; + } + let marker_len = LIST_ITEM_RE.captures(s)?.get(1)?.end(); + let tag_end = org_item_tag_end(&s[marker_len..])?; + if marker_len + tag_end == s.len() { + Some(s.chars().count()) + } else { + None + } +} + /// org-element-export-snippet-parser prefix: `@@BACKEND:VALUE@@`. /// Backend is `[-A-Za-z0-9]+`. Value runs to the next `@@` (may contain /// a single `@`; GitHub #354). @@ -1354,7 +1403,8 @@ impl FormatParser for OrgParser { footnote_saw_blank = false; } - // List item: marker is structure, rest is prose + // List item: bullet is structure. An item tag (`tag ::`) + // stays on that line; the description after ` :: ` hangs. if let Some(caps) = LIST_ITEM_RE.captures(line_text) { flush_prose_spanned(&mut current_prose, &mut prose_span, &mut regions); let marker = caps.get(1).unwrap().as_str(); @@ -1370,9 +1420,12 @@ impl FormatParser for OrgParser { list_saw_blank = false; in_footnote_def = false; footnote_saw_blank = false; - let marker_span = ByteSpan::new(line.start, line.start + marker.len()); + let struct_len = org_item_tag_end(&line_text[marker.len()..]) + .map(|n| marker.len() + n) + .unwrap_or(marker.len()); + let marker_span = ByteSpan::new(line.start, line.start + struct_len); regions.push(SpannedRegion::structure(input, marker_span)); - Self::emit_hung_text(input, &line, marker.len(), &mut regions); + Self::emit_hung_text(input, &line, struct_len, &mut regions); continue; } @@ -1428,6 +1481,24 @@ impl FormatParser for OrgParser { continue; } if is_term { + // A tag-only item has Structure then a newline and + // no Prose. The next indented line is the description. + let prev_is_prose = regions + .iter() + .rev() + .nth(1) + .is_some_and(|r| matches!(r.region, Region::Prose(_))); + if !prev_is_prose { + if leading > 0 { + regions.push(SpannedRegion::structure( + input, + ByteSpan::new(line.start, line.start + leading), + )); + } + Self::emit_hung_text(input, &line, leading, &mut regions); + list_saw_blank = false; + continue; + } regions.pop(); if let Some(break_at) = org_line_break_at(line_text) { let content = line_text[..break_at].trim(); diff --git a/src/parser/rst.rs b/src/parser/rst.rs index 87ae98c6..9e4f96a5 100644 --- a/src/parser/rst.rs +++ b/src/parser/rst.rs @@ -935,7 +935,9 @@ fn rst_directive_name(trimmed: &str) -> Option { /// arguments (epigraph / highlights / pull-quote / compound / header / /// footer / parsed-literal / line-block): same-line text after `::` is /// the first nested-parsed body paragraph (GitHub #349 / #351 / #422 / -/// #426 / #430). +/// #426 / #430). `title` is the same no-argument class here. +/// `admonition` / `rubric` / `topic` / `sidebar` / `list-table` / +/// `contents` take a title argument and are not in this set. fn is_rst_specific_admonition(name: &str) -> bool { matches!( name, @@ -956,16 +958,19 @@ fn is_rst_specific_admonition(name: &str) -> bool { | "footer" | "parsed-literal" | "line-block" - | "rubric" - | "topic" - | "sidebar" - | "admonition" - | "list-table" - | "contents" | "title" ) } +/// Directives whose same-line text is a title argument, not a body +/// paragraph. The title stays on the directive line. +fn is_rst_title_argument_directive(name: &str) -> bool { + matches!( + name, + "admonition" | "rubric" | "topic" | "sidebar" | "list-table" | "contents" + ) +} + /// Docutils admonitions plus figure/topic/sidebar/container: bodies /// nested-parse, so hang + reflow. Leftover body.py names in /// `is_rst_specific_admonition` (including parsed-literal / line-block) @@ -975,10 +980,8 @@ fn is_rst_specific_admonition(name: &str) -> bool { /// nested-parsed paragraph (GitHub #434). fn is_rst_container_directive(name: &str) -> bool { is_rst_specific_admonition(name) - || matches!( - name, - "admonition" | "figure" | "topic" | "sidebar" | "container" | "class" | "list-table" - ) + || is_rst_title_argument_directive(name) + || matches!(name, "figure" | "container" | "class") } /// Docutils `meta` directive. Takes no argument; the body is a field diff --git a/src/reflow.rs b/src/reflow.rs index bd46fb12..9e91338d 100644 --- a/src/reflow.rs +++ b/src/reflow.rs @@ -1219,6 +1219,11 @@ fn hanging_indent_width(s: &str) -> usize { if crate::parser::org::org_caption_marker_len(s) == Some(s.len()) { return s.chars().count(); } + // Org item tag (`- tag :: `): hang at the tag so the description + // stays on the item and a period in the tag does not split it. + if let Some(width) = crate::parser::org::org_item_tag_hang_width(s) { + return width; + } // Markdown definition marker (`: ` / ` : `): hang at marker width // so the body stays inside the definition (GitHub #210). if crate::parser::markdown::md_definition_list_marker_len(s) == Some(s.len()) { diff --git a/tests/org_item_tag.rs b/tests/org_item_tag.rs new file mode 100644 index 00000000..6ff7a7dc --- /dev/null +++ b/tests/org_item_tag.rs @@ -0,0 +1,123 @@ +//! Org list item tag (`tag :: description`) stays on the item line. +//! The description after ` :: ` reflows. A sentence in the tag does not +//! split away from the bullet. + +use snapper_fmt::format::Format; +use snapper_fmt::parser::org::OrgParser; +use snapper_fmt::parser::{FormatParser, Region}; +use snapper_fmt::{FormatConfig, format_text}; + +fn org_cfg() -> FormatConfig { + FormatConfig { + format: Format::Org, + max_width: 0, + ..Default::default() + } + .without_safety_backstops() +} + +#[test] +fn item_tag_stays_on_the_item_line_description_reflows() { + let input = "- Alpha. Beta :: One. Two.\n"; + let regions = OrgParser.parse(input); + assert!( + regions.iter().any(|r| matches!( + r, + Region::Structure(s) if s == "- Alpha. Beta :: " + )), + "tag and separator stay Structure, got {regions:?}" + ); + assert!( + regions + .iter() + .any(|r| matches!(r, Region::Prose(p) if p == "One. Two.")), + "description after :: is Prose, got {regions:?}" + ); + assert!( + !regions + .iter() + .any(|r| matches!(r, Region::Prose(p) if p.contains("Alpha."))), + "a sentence in the tag must not be Prose, got {regions:?}" + ); + let out = format_text(input, &org_cfg()).unwrap(); + let hang = " ".repeat("- Alpha. Beta :: ".chars().count()); + assert_eq!( + out, + format!("- Alpha. Beta :: One.\n{hang}Two.\n"), + "tag stays; description splits under the separator, got:\n{out}" + ); + assert!( + !out.contains("- Alpha.\n"), + "tag sentence must not leave the item marker, got:\n{out}" + ); + assert_eq!(format_text(&out, &org_cfg()).unwrap(), out); +} + +#[test] +fn item_tag_uses_the_last_separator() { + let input = "- Alpha :: Beta. Gamma :: One. Two.\n"; + let regions = OrgParser.parse(input); + assert!( + regions.iter().any(|r| matches!( + r, + Region::Structure(s) if s == "- Alpha :: Beta. Gamma :: " + )), + "the tag runs through the last separator, got {regions:?}" + ); + assert!( + regions + .iter() + .any(|r| matches!(r, Region::Prose(p) if p == "One. Two.")), + "only the text after the last separator is Prose, got {regions:?}" + ); + let out = format_text(input, &org_cfg()).unwrap(); + assert!( + out.starts_with("- Alpha :: Beta. Gamma :: One.\n"), + "a sentence inside the tag stays on the item line, got:\n{out}" + ); + assert!( + out.contains("Two.\n"), + "description still splits, got:\n{out}" + ); + assert!( + !out.contains("- Alpha.\n"), + "tag must not split, got:\n{out}" + ); +} + +#[test] +fn tagged_item_description_on_the_next_line_is_kept() { + let input = "- Alpha. Beta ::\n One. Two.\n"; + let regions = OrgParser.parse(input); + assert!( + regions.iter().any(|r| matches!( + r, + Region::Structure(s) if s.contains("- Alpha. Beta ::") + )), + "tag stays Structure, got {regions:?}" + ); + assert!( + regions + .iter() + .any(|r| matches!(r, Region::Prose(p) if p.contains("One. Two."))), + "next-line description must be Prose, got {regions:?}" + ); + let out = format_text(input, &org_cfg()).unwrap(); + assert!( + out.contains("- Alpha. Beta ::"), + "tag line stays, got:\n{out}" + ); + assert!( + out.contains("One."), + "first description sentence stays, got:\n{out}" + ); + assert!( + out.contains("Two."), + "second description sentence stays, got:\n{out}" + ); + assert!( + !out.contains("One. Two."), + "description must still split, got:\n{out}" + ); + assert_eq!(format_text(&out, &org_cfg()).unwrap(), out); +} diff --git a/tests/rst_container_directives.rs b/tests/rst_container_directives.rs index ad033b58..fdeaae6f 100644 --- a/tests/rst_container_directives.rs +++ b/tests/rst_container_directives.rs @@ -163,8 +163,8 @@ fn leftover_list_table_title_hangs_cells_still_split() { ); let out = format_text(input, &rst_cfg()).unwrap(); assert!( - !out.contains(".. list-table:: fig. 1 is here. After."), - "list-table title must still split, got:\n{out}" + out.contains(".. list-table:: fig. 1 is here. After."), + "list-table title stays on the directive line, got:\n{out}" ); assert!( out.contains("After.\nNext."), @@ -178,8 +178,8 @@ fn leftover_contents_same_line_title_hangs_and_splits() { let input = concat!(".. contents:: fig. 1 is here. After.\n", "After. Next.\n",); let out = format_text(input, &rst_cfg()).unwrap(); assert!( - !out.contains(".. contents:: fig. 1 is here. After."), - "contents title must still split, got:\n{out}" + out.contains(".. contents:: fig. 1 is here. After."), + "contents title stays on the directive line, got:\n{out}" ); assert!( out.contains("After.\nNext."), @@ -235,14 +235,14 @@ fn leftover_topic_same_line_title_hangs_and_splits() { assert!( regions.iter().any(|r| matches!( r, - Region::Prose(p) if p.contains("fig. 1 is here.") && p.contains("After.") + Region::Structure(s) if s.contains(".. topic:: fig. 1 is here. After.") )), - "topic title leftover must be Prose, got {regions:?}" + "topic title stays Structure on the directive line, got {regions:?}" ); let out = format_text(input, &rst_cfg()).unwrap(); assert!( - !out.contains(".. topic:: fig. 1 is here. After."), - "topic title must still split, got:\n{out}" + out.contains(".. topic:: fig. 1 is here. After."), + "topic title stays on the directive line, got:\n{out}" ); assert!( out.contains("After.\nNext."), @@ -264,14 +264,14 @@ fn leftover_rubric_same_line_hangs_and_splits() { assert!( regions.iter().any(|r| matches!( r, - Region::Prose(p) if p.contains("fig. 1 is here.") && p.contains("After.") + Region::Structure(s) if s.contains(".. rubric:: fig. 1 is here. After.") )), - "rubric argument must be leftover Prose, got {regions:?}" + "rubric title stays Structure on the directive line, got {regions:?}" ); let out = format_text(input, &rst_cfg()).unwrap(); assert!( - !out.contains(".. rubric:: fig. 1 is here. After."), - "rubric argument must still split, got:\n{out}" + out.contains(".. rubric:: fig. 1 is here. After."), + "rubric title stays on the directive line, got:\n{out}" ); assert!( out.contains("After.\nNext."), diff --git a/tests/rst_directive_title.rs b/tests/rst_directive_title.rs new file mode 100644 index 00000000..71ecf9e0 --- /dev/null +++ b/tests/rst_directive_title.rs @@ -0,0 +1,105 @@ +//! Title arguments of admonition, rubric, topic, sidebar, list-table, +//! and contents stay on the directive line. Body prose still reflows. +//! `note` / `warning` same-line text is a body paragraph and still splits. + +use snapper_fmt::format::Format; +use snapper_fmt::parser::rst::RstParser; +use snapper_fmt::parser::{FormatParser, Region}; +use snapper_fmt::{FormatConfig, format_text}; + +fn rst_cfg() -> FormatConfig { + FormatConfig { + format: Format::Rst, + max_width: 0, + ..Default::default() + } + .without_safety_backstops() +} + +#[test] +fn title_argument_stays_on_directive_line_body_reflows() { + for name in [ + "admonition", + "rubric", + "topic", + "sidebar", + "list-table", + "contents", + ] { + let input = format!( + ".. {name}:: First title. Second title.\n\n Body one. Body two.\n\nAfter. Next.\n" + ); + let regions = RstParser.parse(&input); + assert!( + regions.iter().any(|r| matches!( + r, + Region::Structure(s) if s.contains(&format!(".. {name}:: First title. Second title.")) + )), + "{name} title must stay Structure, got {regions:?}" + ); + assert!( + !regions.iter().any(|r| matches!( + r, + Region::Prose(p) if p.contains("First title.") + )), + "{name} title must not be Prose, got {regions:?}" + ); + let out = format_text(&input, &rst_cfg()).unwrap(); + assert!( + out.contains(&format!(".. {name}:: First title. Second title.")), + "{name} title stays on the directive line, got:\n{out}" + ); + assert!( + !out.contains("First title.\n"), + "{name} title must not split, got:\n{out}" + ); + assert!( + out.contains(" Body one.\n Body two."), + "{name} body must still reflow, got:\n{out}" + ); + assert!( + out.contains("After.\nNext."), + "prose after {name} must still reflow, got:\n{out}" + ); + assert_eq!(format_text(&out, &rst_cfg()).unwrap(), out); + } +} + +#[test] +fn indented_line_after_same_line_title_still_splits() { + let input = ".. admonition:: Title\n Body here. Second body.\nAfter markup. Next sentence.\n"; + let out = format_text(input, &rst_cfg()).unwrap(); + assert!( + out.contains(".. admonition:: Title\n"), + "same-line title stays on the directive line, got:\n{out}" + ); + assert!( + out.contains(" Body here.\n Second body.\n"), + "indented body must hang and split, got:\n{out}" + ); + assert!( + out.contains("After markup.\nNext sentence.\n"), + "flush prose after the directive must split, got:\n{out}" + ); + assert_eq!(format_text(&out, &rst_cfg()).unwrap(), out); +} + +#[test] +fn note_and_warning_same_line_body_still_reflows() { + for name in ["note", "warning"] { + let input = format!(".. {name}:: Body one. Body two.\n"); + let out = format_text(&input, &rst_cfg()).unwrap(); + assert!( + out.contains(&format!(".. {name}:: Body one.\n")), + "{name} same-line body must still split, got:\n{out}" + ); + assert!( + out.contains("Body two."), + "{name} second sentence must remain, got:\n{out}" + ); + assert!( + !out.contains(&format!(".. {name}:: Body one. Body two.")), + "{name} same-line body must not stay one line, got:\n{out}" + ); + } +}