From 66c2c79ef5373bdfb783741f71c1c7a8a09e5b86 Mon Sep 17 00:00:00 2001 From: OmarAlJarrah Date: Sun, 9 Aug 2026 06:43:14 +0300 Subject: [PATCH] fix(ir): keep CanonicalWords a fixed point on caseless capitals MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit CanonicalWords is documented as a fixed point and the fuzz target asserts it, but feeding a canonical back in could still change it: CanonicalWords("ℤℤA") == "ℤℤa" CanonicalWords("ℤℤa") == "ℤ_ℤa" The acronym-tail rule splits at the last capital of a run when the next rune is lowercase. "ℤℤA" has no lowercase rune at all, so the first pass produces one word and lowercases it -- and lowercasing is what supplies the missing lowercase letter, since ℤ has no lowercase form and survives as a capital while A becomes a. The second pass sees two capitals with a lowercase following, fires the tail rule, and yields "ℤ_ℤa". This is #187's mechanism at a position #187 did not reach: there the caseless capital followed the cased one, here it precedes it. The consequence is the same -- the segmentation depends on the casing of the source spelling rather than on its words, and Canonical is ABI precisely so an emitter never has to know which spelling produced it. The tail rule now requires that lowercasing change at least one of the two capitals either side of the split. Requiring it of both would be too strong in either direction: asking it of the left rune loses ℤ_server, asking it of the right loses http_ℤerver, and canonicalCases pins both. Asking it of the pair keeps every pinned row, and makes the grammar a fixed point for a reason that can be stated: one pass lowercases every rune that carries case, so neither case rule has anything left to fire on, and the letter/digit rule is case-independent. --- ir/naming.go | 40 ++++++++++++++++++++++++++++++++------ ir/naming_grammar_test.go | 8 ++++++++ ir/naming_property_test.go | 7 ++++++- 3 files changed, 48 insertions(+), 7 deletions(-) diff --git a/ir/naming.go b/ir/naming.go index 94a9fea..dd1481a 100644 --- a/ir/naming.go +++ b/ir/naming.go @@ -94,7 +94,7 @@ func wordBoundary(prev, r rune, runes []rune, i int) bool { switch { case carriesCase(r) && (unicode.IsLower(prev) || unicode.IsDigit(prev)): return true // lower/digit -> Upper: "userID" -> user|ID - case isCapital(prev) && isCapital(r) && i+1 < len(runes) && unicode.IsLower(runes[i+1]): + case acronymTail(prev, r, runes, i): return true // acronym tail: "HTTPServer" -> HTTP|Server case unicode.IsLetter(prev) && unicode.IsDigit(r), unicode.IsDigit(prev) && unicode.IsLetter(r): return true // letter<->digit: "APIKey2" -> ...Key|2 @@ -103,6 +103,31 @@ func wordBoundary(prev, r rune, runes []rune, i int) bool { } } +// acronymTail reports whether runes[i] is the last capital of a run and opens +// the next word — the S of "HTTPServer" — which is two capitals followed by a +// lowercase letter. +// +// At least one of the two capitals must be one lowercasing changes. That is what +// keeps the rule from splitting its own output: a boundary between two runes +// lowercasing leaves alone survives into the result still looking like a +// boundary, so a second pass splits there again (GitHub #336). The lowercase +// letter the rule looks for need not have been lowercase in the source — "ℤℤA" +// has none, and lowercasing the A supplies one — so asking only about the source +// runes either side is not enough to know the pattern will not reappear. +// +// Requiring case of both would be too strong in either direction: it would lose +// ℤ_server, where only the S carries case, and http_ℤerver, where only the P +// does. Both are pinned in canonicalCases. +func acronymTail(prev, r rune, runes []rune, i int) bool { + if !isCapital(prev) || !isCapital(r) { + return false + } + if !carriesCase(prev) && !carriesCase(r) { + return false // neither is lowercased, so the split would reappear + } + return i+1 < len(runes) && unicode.IsLower(runes[i+1]) +} + // The two boundary rules above ask different questions about a rune, and the // difference is what GitHub #187 turned on. // @@ -118,12 +143,15 @@ func wordBoundary(prev, r rune, runes []rune, i int) bool { // isCapital asks whether r *belongs to a run of capitals*, which is a question // about the letter's form rather than about a transition — the acronym-tail rule // splits at the last capital of a run, and ℤ is one of those whether or not -// lowercasing would change it. Using carriesCase there instead would lose -// ℤ_server, and using IsUpper alone would lose the titlecase forms. +// lowercasing would change it. Using carriesCase alone there would lose +// ℤ_server, and using IsUpper alone would lose the titlecase forms. Which is why +// the tail rule asks both: isCapital of each rune, and carriesCase of the pair. // -// Both stay idempotent: after one pass the only capitals left are the ones -// lowercasing does not change, and the tail rule needs a lowercase letter -// following, which such a run does not produce on its own. +// The grammar is a fixed point because no rule can fire on its own output. One +// pass lowercases every rune that carries case, so the two case rules — which +// each require one at the boundary — have nothing left to fire on, and the +// letter/digit rule is case-independent and has already been applied everywhere +// it applies. func carriesCase(r rune) bool { return unicode.ToLower(r) != r } // isCapital reports whether r is a capital letter form — uppercase or titlecase. diff --git a/ir/naming_grammar_test.go b/ir/naming_grammar_test.go index a1b6073..ba79fbf 100644 --- a/ir/naming_grammar_test.go +++ b/ir/naming_grammar_test.go @@ -66,6 +66,14 @@ var canonicalCases = []struct { {"a run of capitals ends at one with no lowercase form", "\u2124Server", "\u2124_server"}, {"and such a letter can be the tail itself", "HTTP\u2124erver", "http_\u2124erver"}, {"a titlecase letter opens a run", "\u01C5Bc", "\u01C6_bc"}, + // GitHub #336. A tail needs case on one side of the split, because a split + // between two runes lowercasing leaves alone is one the next pass makes + // again. The lowercase letter the rule looks for is not always in the + // source: "\u2124\u2124A" has none, and lowercasing the A supplies one, so the pair + // below used to segment differently from the spelling one case apart. + {"a run of caseless capitals is no tail", "\u2124\u2124A", "\u2124\u2124a"}, + {"so the lowercase spelling segments the same way", "\u2124\u2124a", "\u2124\u2124a"}, + {"a caseless run still ends at a cased capital", "\u2124\u2124Ab", "\u2124\u2124_ab"}, {"a name with no word rune has no words", "***", ""}, {"the empty name is empty", "", ""}, } diff --git a/ir/naming_property_test.go b/ir/naming_property_test.go index 7d9730b..2ad7a2e 100644 --- a/ir/naming_property_test.go +++ b/ir/naming_property_test.go @@ -19,6 +19,11 @@ import ( // // - ℤ, ϒ: IsUpper reports true and ToLower returns them unchanged — a letter // with no lowercase form. +// - ℤℤA: two of those before one that does lowercase. The seed beside it, +// ℤℤa, was already here and passed; the input that broke idempotence was +// the uppercase spelling one mutation away from it, because lowercasing the +// A is what supplied the lowercase letter the tail rule looks for +// (GitHub #336). // - Dž: titlecase, which is neither IsUpper nor IsLower. // - ẞ: uppercase whose lowercase ß is a different letter. // - İ: uppercase whose lowercase is two runes, so lowercasing changes length. @@ -29,7 +34,7 @@ import ( // Hebrew are cased by nothing at all. // - the rest: the boundaries the grammar splits on, and names with no words. var adversarialRunes = []string{ - "Aℤ", "aℤ", "COUNTℤ", "ℤℤa", "aℤℤb", + "Aℤ", "aℤ", "COUNTℤ", "ℤℤa", "ℤℤA", "aℤℤb", "Aϒ", "xDžy", "aẞb", "İstanbul", "Džungla", "café_v2", "́x", "x́", "ᎠᎡ", "ꭰx", "𐐀𐐨", "Δε", "Жx", "中文", "אב",