From a74b3ecde67f219c67ee2e345cff1ff39d98f131 Mon Sep 17 00:00:00 2001 From: Junyi Yao Date: Wed, 23 Sep 2026 22:45:21 -0700 Subject: [PATCH 1/2] fix: preserve underscores when creating integers from strings TOML 1.0 allows grouping underscores in integers (1_000, 0xdead_beef), and tomlkit parses them fine, but api.integer("1_000") silently normalized the raw text to "1000" because it built the item via item(int(raw)), discarding the original string. Validate string input against the TOML v1.0.0 integer grammar and preserve it verbatim as the item's raw text; invalid strings raise ValueError. The parse path and integer(int) behavior are unchanged. Fixes #336. Signed-off-by: Junyi Yao --- tests/test_api.py | 58 +++++++++++++++++++++++++++++++++++++++++++++++ tomlkit/api.py | 28 ++++++++++++++++++++++- 2 files changed, 85 insertions(+), 1 deletion(-) diff --git a/tests/test_api.py b/tests/test_api.py index b5f31bc3..21671010 100644 --- a/tests/test_api.py +++ b/tests/test_api.py @@ -611,3 +611,61 @@ def test_parse_accepts_nesting_at_the_depth_limit() -> None: assert parse(array).as_string() == array dotted = ".".join(["a"] * depth) + " = 1" assert parse(dotted).as_string() == dotted + + +@pytest.mark.parametrize( + "raw", + [ + "0", + "42", + "1_000", + "5_349_221", + "+1_000", + "-1_000", + "0xdead_beef", + "0xDEAD_BEEF", + "0o7_7", + "0b1_0", + ], +) +def test_integer_preserves_grouping_underscores(raw: str) -> None: + assert tomlkit.integer(raw).as_string() == raw + assert int(tomlkit.integer(raw)) == int(raw, 0) + + +@pytest.mark.parametrize("raw", ["1_000", "0xdead_beef", "+1_000", "0o7_7"]) +def test_integer_emission_matches_parse(raw: str) -> None: + doc = tomlkit.document() + doc["n"] = tomlkit.integer(raw) + assert tomlkit.dumps(doc) == f"n = {raw}\n" + # the emission path agrees with the parse path on the same literal + assert tomlkit.integer(raw).as_string() == tomlkit.parse(f"n = {raw}\n")["n"].as_string() + + +@pytest.mark.parametrize( + "raw", + [ + "", + "abc", + "_100", + "100_", + "1__000", + "01_2", + "0x_1", + "0X1", # the parser only accepts a lowercase prefix + "+0x1", # hex/oct/bin take no sign + "-0o7", + " 1", + "1 ", + "1.0", + "1e3", + ], +) +def test_integer_rejects_invalid_strings(raw: str) -> None: + with pytest.raises(ValueError): + tomlkit.integer(raw) + + +def test_integer_from_int_unchanged() -> None: + assert tomlkit.integer(1000).as_string() == "1000" + assert int(tomlkit.integer(1000)) == 1000 diff --git a/tomlkit/api.py b/tomlkit/api.py index b0f8cd6d..69fdbb53 100644 --- a/tomlkit/api.py +++ b/tomlkit/api.py @@ -2,6 +2,7 @@ import contextlib import datetime as _datetime +import re from collections.abc import Iterable from collections.abc import Mapping @@ -114,8 +115,33 @@ def document() -> TOMLDocument: # Items +_INTEGER_RE = re.compile( + r"(?:" + r"0x[0-9a-fA-F](?:_?[0-9a-fA-F])*" # hex: lowercase prefix only, no sign + r"|0o[0-7](?:_?[0-7])*" # octal: lowercase prefix only, no sign + r"|0b[01](?:_?[01])*" # binary: lowercase prefix only, no sign + r"|[+-]?(?:0|[1-9](?:_?[0-9])*)" # decimal: optional sign, no leading zeros + r")" +) +"""Valid TOML v1.0.0 integer literals (https://toml.io/en/v1.0.0#integer). + +Underscores may only appear between digits; hex/octal/binary use a +lowercase prefix and take no sign, mirroring ``Parser._parse_number``. +""" + + def integer(raw: str | int) -> Integer: - """Create an integer item from a number or string.""" + """Create an integer item from a number or string. + + When ``raw`` is a string it must be a valid TOML v1.0.0 integer + literal; the original text is then preserved verbatim so grouping + underscores (e.g. ``"1_000"``) survive ``dumps()``. Anything else + raises ``ValueError``. + """ + if isinstance(raw, str): + if not _INTEGER_RE.fullmatch(raw): + raise ValueError(f"Invalid TOML integer: {raw!r}") + return Integer(int(raw, 0), Trivia(), raw) return item(int(raw)) From fea9f09ae09d78401b46615b5ef423150aa00a9f Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 05:45:44 +0000 Subject: [PATCH 2/2] [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --- tests/test_api.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/tests/test_api.py b/tests/test_api.py index 21671010..000b94c9 100644 --- a/tests/test_api.py +++ b/tests/test_api.py @@ -639,7 +639,10 @@ def test_integer_emission_matches_parse(raw: str) -> None: doc["n"] = tomlkit.integer(raw) assert tomlkit.dumps(doc) == f"n = {raw}\n" # the emission path agrees with the parse path on the same literal - assert tomlkit.integer(raw).as_string() == tomlkit.parse(f"n = {raw}\n")["n"].as_string() + assert ( + tomlkit.integer(raw).as_string() + == tomlkit.parse(f"n = {raw}\n")["n"].as_string() + ) @pytest.mark.parametrize(