diff --git a/docs/orgmode/reference/formats.org b/docs/orgmode/reference/formats.org index 454d0adc..30531ed9 100644 --- a/docs/orgmode/reference/formats.org +++ b/docs/orgmode/reference/formats.org @@ -143,7 +143,8 @@ These tokens within prose are not split across lines: - Extra names from =[latex].verbatim_envs= are code regions too - Body follows the same comment-reflow and optional =--format-code= rules as other formats when language is known (=minted= / =minted*= language arg, =lstlisting= / =lstlisting*= =language== option) - Inline leftover =\verb= / =\verb*= / leftover =\lstinline= / leftover =\spverb= / leftover =\mintinline= / leftover =\mint= / leftover fancyvrb =\Verb= / =\Verb*= / leftover =\SaveVerb= / leftover =\UseVerb= / leftover =\UseVerbatim= / leftover =\LUseVerbatim= / leftover =\BUseVerbatim= / leftover =\DefineShortVerb= / =\UndefineShortVerb= / leftover fvextra =\EscVerb= / leftover fvextra =\VerbatimInsertBuffer= / =\VerbatimClearBuffer= / =\InsertBuffer= / =\IterateBuffer= / =\VerbatimInput= / =\BVerbatimInput= / =\LVerbatimInput= / piton.sty leftover =\piton= / =\PitonInputFile= / =\PitonInputFileT= / =\PitonInputFileF= / =\PitonInputFileTF= / listings.sty =\lstinputlisting= / minted.sty =\inputminted= / tools/verbatim.sty =\verbatiminput= / tcolorbox =\tcbinputlisting= / pythontex.sty =\inputpy= / =\inputpycon= / leftover inline =\py= / =\pyc= / =\pys= / =\pyb= / =\pyv= / =\pycon= and twins / =\sympy= / =\pylab= and twins / leftover usefamily =\ruby= / =\rb= / =\julia= / =\jl= / =\matlab= / =\octave= / =\bash= / =\sage= / =\rust= / =\rs= / =\R= / =\perl= / =\pl= / =\perlsix= / =\psix= / =\javascript= / =\js= and twins / =\inputpygments= / =\pygment= / leftover =\pythontexcustomc= / pythonhighlight.sty =\inputpython= / =\inputpythonfile= / leftover =\pyth= / catchfilebetweentags.sty =\CatchFileBetweenTags= / =\CatchFileBetweenDelims= / =\ExecuteMetaData= / catchfile.sty leftover =\CatchFileDef= / =\CatchFileEdef= / moreverb =\listinginput= / leftover =\verbatimtabinput= / =\verbatimtabinput*= / leftover =\verbatimwrite= / =\verbatimwrite*= / leftover =\listingcont= / sagetex =\sageinput= / leftover inline =\sageplot= / =\sagestr= / scontents leftover =\Scontents= / =\Scontents*= / =\typestored= / =\getstored= / =\mergesc= / =\meaningsc= / =\foreachsc= (and extra =[latex].verbatim_commands=) stay atomic; inner =.!?%= do not split or comment. - leftover =\verb= / =\verb*= / =\lstinline= / =\spverb= take a delimiter or, for =\lstinline=, optional =[...]= then a delimiter or ={...}=; a flush following sentence stays on its own line. + leftover =\verb= / =\verb*= / =\lstinline= / =\spverb= take a delimiter or, for =\lstinline=, optional =[...]= then a delimiter or ={...}=; spaces after the control word are not the delimiter, so =\verb x...x= closes on =x=; a flush following sentence stays on its own line. + url.sty =\url= / =\path= and hyperref =\nolinkurl= take a non-brace delimiter the same way (=\url|...|=); a braced argument stays on the generic command path; a flush following sentence stays on its own line. A width wrap keeps a space inside that delimiter on the same line leftover =\mintinline= / =\mint= take optional =[...]=, ={lang}=, then a delimiter or ={...}= body; a flush following sentence stays on its own line leftover =\SaveVerb= takes optional =[...]=, a ={name}=, then the same delimiter body as =\Verb=; a flush following sentence stays on its own line diff --git a/src/parser/latex.rs b/src/parser/latex.rs index fb8a9916..d1c88469 100644 --- a/src/parser/latex.rs +++ b/src/parser/latex.rs @@ -3552,6 +3552,65 @@ Some text. assert_eq!(format_text(&out, &latex_cfg()).unwrap(), out); } + #[test] + fn verb_letter_delimiter_round_trips() { + use crate::format_text; + + for cmd in [r"\verb", r"\verb*", r"\Verb", r"\spverb"] { + let input = format!( + "\\begin{{document}}\nSee {cmd} zCode. Next. Morez here. Done.\n\\end{{document}}\n" + ); + let out = format_text(&input, &latex_cfg()).unwrap(); + let span = format!("{cmd} zCode. Next. Morez"); + assert!( + out.contains(&span), + "{cmd} letter body must stay intact, got:\n{out}" + ); + assert!( + !out.contains("Next.\nMore"), + "{cmd} must not split after Next., got:\n{out}" + ); + assert!( + out.contains(&format!("See {span} here.\nDone.")), + "prose after {cmd} must still split, got:\n{out}" + ); + assert_eq!(format_text(&out, &latex_cfg()).unwrap(), out); + } + } + + #[test] + fn url_char_delimiter_round_trips() { + use crate::format_text; + + for (cmd, body) in [ + (r"\url", r"http://example.com/A. B"), + (r"\path", r"Foo. Bar"), + (r"\nolinkurl", r"http://example.com/A. B"), + ] { + let span = format!("{cmd}|{body}|"); + let input = format!("\\begin{{document}}\nSee {span} here. Done.\n\\end{{document}}\n"); + let out = format_text(&input, &latex_cfg()).unwrap(); + assert!( + out.contains(&span), + "{cmd} character body must stay intact, got:\n{out}" + ); + assert!( + out.contains(&format!("See {span} here.\nDone.")), + "prose after {cmd} must still split, got:\n{out}" + ); + let braced = format!(r"{cmd}{{{body}}}"); + let braced_in = + format!("\\begin{{document}}\nSee {braced} here. Done.\n\\end{{document}}\n"); + let braced_out = format_text(&braced_in, &latex_cfg()).unwrap(); + assert!( + braced_out.contains(&format!("See {braced} here.\nDone.")), + "braced {cmd} must stay one span and the next sentence must split, got:\n{braced_out}" + ); + assert_eq!(format_text(&out, &latex_cfg()).unwrap(), out); + assert_eq!(format_text(&braced_out, &latex_cfg()).unwrap(), braced_out); + } + } + #[test] fn fancyvrb_verb_with_inner_punct_round_trips() { use crate::format_text; diff --git a/src/sentence/unicode.rs b/src/sentence/unicode.rs index 82d2834f..ad6836d3 100644 --- a/src/sentence/unicode.rs +++ b/src/sentence/unicode.rs @@ -325,9 +325,13 @@ fn protect_latex_verbatim( /// `\InsertBuffer` / `\IterateBuffer` / extra-name span /// starting at `at`. /// -/// `\verb` / `\verb*` / `\spverb` / `\spverb*` / `\Verb` / `\Verb*`: next -/// character is the -/// delimiter; content runs to the same character. Leftover walker +/// `\verb` / `\verb*` / `\spverb` / `\spverb*` / `\Verb` / `\Verb*`: the +/// delimiter is the next character after spaces TeX drops following the +/// control word (and after `*` for the star form). A letter there is +/// the delimiter (`\verb x...x`), not the space. Content runs to the +/// same character. url.sty `\url` / `\path` and hyperref `\nolinkurl` +/// use that non-brace delimiter; a `{` stays on the generic `\cmd{arg}` +/// path. Leftover walker /// (GitHub #452) classifies `\verb` / `\verb*` / `\lstinline` / /// `\mintinline` / `\mint` / `\SaveVerb` / `\spverb` / `\piton` as /// Structure so following flush prose does not join. `\lstinline` / @@ -777,6 +781,12 @@ pub(crate) fn latex_verb_span_end_with( return None; } (after_bs + "SaveVerb".len(), VerbKind::SaveVerb) + } else if let Some(name) = url_delim_cs_name(tail) { + // url.sty `\url` / `\path` and hyperref `\nolinkurl`. Longer + // name first is unnecessary (`url` is not a prefix of + // `nolinkurl`). Alphabetic leftover rejects a longer name. + // A `{` body stays on the generic `\cmd{arg}` path. + (after_bs + name.len(), VerbKind::UrlDelim) } else if let Some(stripped) = tail.strip_prefix("verb") { if stripped.starts_with(|c: char| c.is_ascii_alphabetic()) { return None; @@ -787,7 +797,14 @@ pub(crate) fn latex_verb_span_end_with( (after_bs + name.len(), VerbKind::Delim) }; - if text.get(i..)?.starts_with('*') { + // `\url` / `\path` / `\nolinkurl` have no star form. A following + // `*` is not this span, so the generic command path can still see + // `\url*{...}`. + if kind == VerbKind::UrlDelim { + if text.get(i..).is_some_and(|s| s.starts_with('*')) { + return None; + } + } else if text.get(i..)?.starts_with('*') { i += 1; } @@ -1107,10 +1124,22 @@ pub(crate) fn latex_verb_span_end_with( } } + // TeX drops spaces after a control word, and the undelimited + // delimiter argument drops spaces too. The character after that + // space is the delimiter. + if matches!(kind, VerbKind::Delim | VerbKind::UrlDelim) { + i = skip_ascii_ws(text, i); + } + let delim = text.get(i..).and_then(|s| s.chars().next())?; if delim == '\n' { return None; } + // Braced `\url{...}` / `\path{...}` / `\nolinkurl{...}` stay on the + // generic `\cmd{arg}` path. Do not retokenize them here. + if kind == VerbKind::UrlDelim && delim == '{' { + return None; + } // pythontex `\py After.` is leftover prose, not `A` as a delimiter. // Same for `\pythontexcustomc{python} After.` after the type brace // and for `\pyth After.`. @@ -1157,8 +1186,12 @@ pub(crate) fn latex_verb_span_end_with( /// Built-in verb-like command shape (GitHub #245 minted `{lang}` body). #[derive(Clone, Copy, PartialEq, Eq)] enum VerbKind { - /// `\verb` / `\spverb` / extras: next character is the delimiter. + /// `\verb` / `\spverb` / `\Verb` / extras: next character after + /// dropped spaces is the delimiter. Delim, + /// url.sty `\url` / `\path` and hyperref `\nolinkurl`: non-brace + /// delimiter. `{` is not this span. + UrlDelim, /// `\lstinline`: optional `[...]` then delimiter or `{...}`. Lstinline, /// `\lstinputlisting` / fancyvrb `\VerbatimInput` family / @@ -1370,6 +1403,22 @@ pub(crate) fn fancyvrb_shortverb_leftover_cs_name(tail: &str) -> Option<&'static None } +/// url.sty `\url` / `\path` and hyperref `\nolinkurl`. Alphabetic +/// leftover rejects a longer name (`\urlfoo`, `\pathological`). No +/// `*` form. A `{` body is not this span. +fn url_delim_cs_name(tail: &str) -> Option<&'static str> { + for name in ["nolinkurl", "path", "url"] { + let Some(after) = tail.strip_prefix(name) else { + continue; + }; + if after.starts_with(|c: char| c.is_ascii_alphabetic()) { + return None; + } + return Some(name); + } + None +} + /// Leftover inline verb-span cmds (GitHub #452). Longer names first so /// `\mintinline` is not `\mint` + leftover. `\verbatiminput` / /// `\lstinputlisting` / `\inputminted` stay their own leftovers @@ -3917,6 +3966,72 @@ mod tests { ); } + #[test] + fn latex_verb_letter_delimiter_stays_atomic() { + for cmd in [r"\verb", r"\verb*", r"\Verb", r"\spverb"] { + // `z` is the delimiter. `Next` contains `x`, so an `x` + // delimiter closes inside that word. + let text = format!("See {cmd} zCode. Next. Morez here. Done."); + let span = format!("{cmd} zCode. Next. Morez"); + assert_eq!( + latex_verb_span_end_with(&span, 0, &[]), + Some(span.len()), + "{cmd} letter delimiter must close on the letter" + ); + assert_eq!( + split(&text), + vec![format!("See {span} here."), "Done.".to_string()], + "{cmd} letter body must stay one span and the next sentence must split" + ); + } + let symbol = r"See \verb|Code. Next| here. Done."; + assert_eq!( + split(symbol), + vec![ + r"See \verb|Code. Next| here.".to_string(), + "Done.".to_string() + ] + ); + } + + #[test] + fn latex_url_char_delimiter_stays_atomic() { + for (cmd, body) in [ + (r"\url", r"http://example.com/A. B"), + (r"\path", r"Foo. Bar"), + (r"\nolinkurl", r"http://example.com/A. B"), + ] { + let span = format!("{cmd}|{body}|"); + let text = format!("See {span} here. Done."); + assert_eq!( + latex_verb_span_end_with(&span, 0, &[]), + Some(span.len()), + "{cmd} character delimiter must close on the delimiter" + ); + assert_eq!( + split(&text), + vec![format!("See {span} here."), "Done.".to_string()], + "{cmd} character body must stay one span and the next sentence must split" + ); + let braced = format!(r"See {cmd}{{{body}}} here. Done."); + assert_eq!( + latex_verb_span_end_with(&format!(r"{cmd}{{{body}}}"), 0, &[]), + None, + "{cmd} braced form must stay off the delimiter scanner" + ); + assert_eq!( + split(&braced), + vec![format!(r"See {cmd}{{{body}}} here."), "Done.".to_string(),], + "{cmd} braced form must stay one span and the next sentence must split" + ); + } + assert_eq!(latex_verb_span_end_with(r"\urlfoo|a.b|", 0, &[]), None); + assert_eq!( + latex_verb_span_end_with(r"\pathological|a.b|", 0, &[]), + None + ); + } + #[test] fn latex_spverb_inner_percent_stays_atomic() { let text = r"See \spverb|a.b%| please. Next.";