Skip to content
4 changes: 2 additions & 2 deletions docs/orgmode/howto/vale-integration.org
Original file line number Diff line number Diff line change
Expand Up @@ -40,8 +40,8 @@ Level: warning.
- =LongProseLine= :: Flags prose lines exceeding 120 characters.
Level: suggestion.
- =SnapperCheck= :: External action.
Runs =snapper --check --output-format json= and reports =fused= / =wrap= on the matching lines when =snapper= is on =PATH=.
Advisory =long= stays on =LongProseLine=.
Runs =snapper --check --output-format json= and reports =fused= / =wrap= on the matching lines when =snapper= is on =PATH=.
Advisory =long= stays on =LongProseLine=.

* Precise CI enforcement
:PROPERTIES:
Expand Down
77 changes: 74 additions & 3 deletions src/parser/org.rs
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,55 @@ static LIST_ITEM_RE: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(r"^(\s*(?:[-+]|\d+[.)])(?:[ \t]|$)|[ \t]+\*(?:[ \t]|$))(.*)$").unwrap()
});

/// org-syntax item: `tag :: description`.
///
/// Bytes of `after_bullet` through the last `[ \t]+::` and the spaces
/// that follow. The tag stays on the item line; the description hangs
/// as prose.
/// `None` when the item has no tag (`::` needs whitespace before it,
/// and whitespace or end after it).
pub(crate) fn org_item_tag_end(after_bullet: &str) -> Option<usize> {
let bytes = after_bullet.as_bytes();
let mut i = 0;
let mut found = None;
while i < bytes.len() {
if bytes[i] == b' ' || bytes[i] == b'\t' {
let mut j = i;
while j < bytes.len() && (bytes[j] == b' ' || bytes[j] == b'\t') {
j += 1;
}
if j + 1 < bytes.len() && bytes[j] == b':' && bytes[j + 1] == b':' {
let mut k = j + 2;
if k == bytes.len() || bytes[k] == b' ' || bytes[k] == b'\t' {
while k < bytes.len() && (bytes[k] == b' ' || bytes[k] == b'\t') {
k += 1;
}
found = Some(k);
}
}
i = j;
continue;
}
i += 1;
}
found
}

/// Hang width when `s` is a list marker plus an item tag and nothing else.
/// Description reflow stays under `tag ::`.
pub(crate) fn org_item_tag_hang_width(s: &str) -> Option<usize> {
if !s.ends_with([' ', '\t']) {
return None;
}
let marker_len = LIST_ITEM_RE.captures(s)?.get(1)?.end();
let tag_end = org_item_tag_end(&s[marker_len..])?;
if marker_len + tag_end == s.len() {
Some(s.chars().count())
} else {
None
}
}

/// org-element-export-snippet-parser prefix: `@@BACKEND:VALUE@@`.
/// Backend is `[-A-Za-z0-9]+`. Value runs to the next `@@` (may contain
/// a single `@`; GitHub #354).
Expand Down Expand Up @@ -1354,7 +1403,8 @@ impl FormatParser for OrgParser {
footnote_saw_blank = false;
}

// List item: marker is structure, rest is prose
// List item: bullet is structure. An item tag (`tag ::`)
// stays on that line; the description after ` :: ` hangs.
if let Some(caps) = LIST_ITEM_RE.captures(line_text) {
flush_prose_spanned(&mut current_prose, &mut prose_span, &mut regions);
let marker = caps.get(1).unwrap().as_str();
Expand All @@ -1370,9 +1420,12 @@ impl FormatParser for OrgParser {
list_saw_blank = false;
in_footnote_def = false;
footnote_saw_blank = false;
let marker_span = ByteSpan::new(line.start, line.start + marker.len());
let struct_len = org_item_tag_end(&line_text[marker.len()..])
.map(|n| marker.len() + n)
.unwrap_or(marker.len());
let marker_span = ByteSpan::new(line.start, line.start + struct_len);
regions.push(SpannedRegion::structure(input, marker_span));
Self::emit_hung_text(input, &line, marker.len(), &mut regions);
Self::emit_hung_text(input, &line, struct_len, &mut regions);
continue;
}

Expand Down Expand Up @@ -1428,6 +1481,24 @@ impl FormatParser for OrgParser {
continue;
}
if is_term {
// A tag-only item has Structure then a newline and
// no Prose. The next indented line is the description.
let prev_is_prose = regions
.iter()
.rev()
.nth(1)
.is_some_and(|r| matches!(r.region, Region::Prose(_)));
if !prev_is_prose {
if leading > 0 {
regions.push(SpannedRegion::structure(
input,
ByteSpan::new(line.start, line.start + leading),
));
}
Self::emit_hung_text(input, &line, leading, &mut regions);
list_saw_blank = false;
continue;
}
regions.pop();
if let Some(break_at) = org_line_break_at(line_text) {
let content = line_text[..break_at].trim();
Expand Down
25 changes: 14 additions & 11 deletions src/parser/rst.rs
Original file line number Diff line number Diff line change
Expand Up @@ -935,7 +935,9 @@ fn rst_directive_name(trimmed: &str) -> Option<String> {
/// arguments (epigraph / highlights / pull-quote / compound / header /
/// footer / parsed-literal / line-block): same-line text after `::` is
/// the first nested-parsed body paragraph (GitHub #349 / #351 / #422 /
/// #426 / #430).
/// #426 / #430). `title` is the same no-argument class here.
/// `admonition` / `rubric` / `topic` / `sidebar` / `list-table` /
/// `contents` take a title argument and are not in this set.
fn is_rst_specific_admonition(name: &str) -> bool {
matches!(
name,
Expand All @@ -956,16 +958,19 @@ fn is_rst_specific_admonition(name: &str) -> bool {
| "footer"
| "parsed-literal"
| "line-block"
| "rubric"
| "topic"
| "sidebar"
| "admonition"
| "list-table"
| "contents"
| "title"
)
}

/// Directives whose same-line text is a title argument, not a body
/// paragraph. The title stays on the directive line.
fn is_rst_title_argument_directive(name: &str) -> bool {
matches!(
name,
"admonition" | "rubric" | "topic" | "sidebar" | "list-table" | "contents"
)
}

/// Docutils admonitions plus figure/topic/sidebar/container: bodies
/// nested-parse, so hang + reflow. Leftover body.py names in
/// `is_rst_specific_admonition` (including parsed-literal / line-block)
Expand All @@ -975,10 +980,8 @@ fn is_rst_specific_admonition(name: &str) -> bool {
/// nested-parsed paragraph (GitHub #434).
fn is_rst_container_directive(name: &str) -> bool {
is_rst_specific_admonition(name)
|| matches!(
name,
"admonition" | "figure" | "topic" | "sidebar" | "container" | "class" | "list-table"
)
|| is_rst_title_argument_directive(name)
|| matches!(name, "figure" | "container" | "class")
}

/// Docutils `meta` directive. Takes no argument; the body is a field
Expand Down
5 changes: 5 additions & 0 deletions src/reflow.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1219,6 +1219,11 @@ fn hanging_indent_width(s: &str) -> usize {
if crate::parser::org::org_caption_marker_len(s) == Some(s.len()) {
return s.chars().count();
}
// Org item tag (`- tag :: `): hang at the tag so the description
// stays on the item and a period in the tag does not split it.
if let Some(width) = crate::parser::org::org_item_tag_hang_width(s) {
return width;
}
// Markdown definition marker (`: ` / ` : `): hang at marker width
// so the body stays inside the definition (GitHub #210).
if crate::parser::markdown::md_definition_list_marker_len(s) == Some(s.len()) {
Expand Down
123 changes: 123 additions & 0 deletions tests/org_item_tag.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,123 @@
//! Org list item tag (`tag :: description`) stays on the item line.
//! The description after ` :: ` reflows. A sentence in the tag does not
//! split away from the bullet.

use snapper_fmt::format::Format;
use snapper_fmt::parser::org::OrgParser;
use snapper_fmt::parser::{FormatParser, Region};
use snapper_fmt::{FormatConfig, format_text};

fn org_cfg() -> FormatConfig {
FormatConfig {
format: Format::Org,
max_width: 0,
..Default::default()
}
.without_safety_backstops()
}

#[test]
fn item_tag_stays_on_the_item_line_description_reflows() {
let input = "- Alpha. Beta :: One. Two.\n";
let regions = OrgParser.parse(input);
assert!(
regions.iter().any(|r| matches!(
r,
Region::Structure(s) if s == "- Alpha. Beta :: "
)),
"tag and separator stay Structure, got {regions:?}"
);
assert!(
regions
.iter()
.any(|r| matches!(r, Region::Prose(p) if p == "One. Two.")),
"description after :: is Prose, got {regions:?}"
);
assert!(
!regions
.iter()
.any(|r| matches!(r, Region::Prose(p) if p.contains("Alpha."))),
"a sentence in the tag must not be Prose, got {regions:?}"
);
let out = format_text(input, &org_cfg()).unwrap();
let hang = " ".repeat("- Alpha. Beta :: ".chars().count());
assert_eq!(
out,
format!("- Alpha. Beta :: One.\n{hang}Two.\n"),
"tag stays; description splits under the separator, got:\n{out}"
);
assert!(
!out.contains("- Alpha.\n"),
"tag sentence must not leave the item marker, got:\n{out}"
);
assert_eq!(format_text(&out, &org_cfg()).unwrap(), out);
}

#[test]
fn item_tag_uses_the_last_separator() {
let input = "- Alpha :: Beta. Gamma :: One. Two.\n";
let regions = OrgParser.parse(input);
assert!(
regions.iter().any(|r| matches!(
r,
Region::Structure(s) if s == "- Alpha :: Beta. Gamma :: "
)),
"the tag runs through the last separator, got {regions:?}"
);
assert!(
regions
.iter()
.any(|r| matches!(r, Region::Prose(p) if p == "One. Two.")),
"only the text after the last separator is Prose, got {regions:?}"
);
let out = format_text(input, &org_cfg()).unwrap();
assert!(
out.starts_with("- Alpha :: Beta. Gamma :: One.\n"),
"a sentence inside the tag stays on the item line, got:\n{out}"
);
assert!(
out.contains("Two.\n"),
"description still splits, got:\n{out}"
);
assert!(
!out.contains("- Alpha.\n"),
"tag must not split, got:\n{out}"
);
}

#[test]
fn tagged_item_description_on_the_next_line_is_kept() {
let input = "- Alpha. Beta ::\n One. Two.\n";
let regions = OrgParser.parse(input);
assert!(
regions.iter().any(|r| matches!(
r,
Region::Structure(s) if s.contains("- Alpha. Beta ::")
)),
"tag stays Structure, got {regions:?}"
);
assert!(
regions
.iter()
.any(|r| matches!(r, Region::Prose(p) if p.contains("One. Two."))),
"next-line description must be Prose, got {regions:?}"
);
let out = format_text(input, &org_cfg()).unwrap();
assert!(
out.contains("- Alpha. Beta ::"),
"tag line stays, got:\n{out}"
);
assert!(
out.contains("One."),
"first description sentence stays, got:\n{out}"
);
assert!(
out.contains("Two."),
"second description sentence stays, got:\n{out}"
);
assert!(
!out.contains("One. Two."),
"description must still split, got:\n{out}"
);
assert_eq!(format_text(&out, &org_cfg()).unwrap(), out);
}
24 changes: 12 additions & 12 deletions tests/rst_container_directives.rs
Original file line number Diff line number Diff line change
Expand Up @@ -163,8 +163,8 @@ fn leftover_list_table_title_hangs_cells_still_split() {
);
let out = format_text(input, &rst_cfg()).unwrap();
assert!(
!out.contains(".. list-table:: fig. 1 is here. After."),
"list-table title must still split, got:\n{out}"
out.contains(".. list-table:: fig. 1 is here. After."),
"list-table title stays on the directive line, got:\n{out}"
);
assert!(
out.contains("After.\nNext."),
Expand All @@ -178,8 +178,8 @@ fn leftover_contents_same_line_title_hangs_and_splits() {
let input = concat!(".. contents:: fig. 1 is here. After.\n", "After. Next.\n",);
let out = format_text(input, &rst_cfg()).unwrap();
assert!(
!out.contains(".. contents:: fig. 1 is here. After."),
"contents title must still split, got:\n{out}"
out.contains(".. contents:: fig. 1 is here. After."),
"contents title stays on the directive line, got:\n{out}"
);
assert!(
out.contains("After.\nNext."),
Expand Down Expand Up @@ -235,14 +235,14 @@ fn leftover_topic_same_line_title_hangs_and_splits() {
assert!(
regions.iter().any(|r| matches!(
r,
Region::Prose(p) if p.contains("fig. 1 is here.") && p.contains("After.")
Region::Structure(s) if s.contains(".. topic:: fig. 1 is here. After.")
)),
"topic title leftover must be Prose, got {regions:?}"
"topic title stays Structure on the directive line, got {regions:?}"
);
let out = format_text(input, &rst_cfg()).unwrap();
assert!(
!out.contains(".. topic:: fig. 1 is here. After."),
"topic title must still split, got:\n{out}"
out.contains(".. topic:: fig. 1 is here. After."),
"topic title stays on the directive line, got:\n{out}"
);
assert!(
out.contains("After.\nNext."),
Expand All @@ -264,14 +264,14 @@ fn leftover_rubric_same_line_hangs_and_splits() {
assert!(
regions.iter().any(|r| matches!(
r,
Region::Prose(p) if p.contains("fig. 1 is here.") && p.contains("After.")
Region::Structure(s) if s.contains(".. rubric:: fig. 1 is here. After.")
)),
"rubric argument must be leftover Prose, got {regions:?}"
"rubric title stays Structure on the directive line, got {regions:?}"
);
let out = format_text(input, &rst_cfg()).unwrap();
assert!(
!out.contains(".. rubric:: fig. 1 is here. After."),
"rubric argument must still split, got:\n{out}"
out.contains(".. rubric:: fig. 1 is here. After."),
"rubric title stays on the directive line, got:\n{out}"
);
assert!(
out.contains("After.\nNext."),
Expand Down
Loading
Loading