diff --git a/tests/test_api.py b/tests/test_api.py index b5f31bc3..000b94c9 100644 --- a/tests/test_api.py +++ b/tests/test_api.py @@ -611,3 +611,64 @@ def test_parse_accepts_nesting_at_the_depth_limit() -> None: assert parse(array).as_string() == array dotted = ".".join(["a"] * depth) + " = 1" assert parse(dotted).as_string() == dotted + + +@pytest.mark.parametrize( + "raw", + [ + "0", + "42", + "1_000", + "5_349_221", + "+1_000", + "-1_000", + "0xdead_beef", + "0xDEAD_BEEF", + "0o7_7", + "0b1_0", + ], +) +def test_integer_preserves_grouping_underscores(raw: str) -> None: + assert tomlkit.integer(raw).as_string() == raw + assert int(tomlkit.integer(raw)) == int(raw, 0) + + +@pytest.mark.parametrize("raw", ["1_000", "0xdead_beef", "+1_000", "0o7_7"]) +def test_integer_emission_matches_parse(raw: str) -> None: + doc = tomlkit.document() + doc["n"] = tomlkit.integer(raw) + assert tomlkit.dumps(doc) == f"n = {raw}\n" + # the emission path agrees with the parse path on the same literal + assert ( + tomlkit.integer(raw).as_string() + == tomlkit.parse(f"n = {raw}\n")["n"].as_string() + ) + + +@pytest.mark.parametrize( + "raw", + [ + "", + "abc", + "_100", + "100_", + "1__000", + "01_2", + "0x_1", + "0X1", # the parser only accepts a lowercase prefix + "+0x1", # hex/oct/bin take no sign + "-0o7", + " 1", + "1 ", + "1.0", + "1e3", + ], +) +def test_integer_rejects_invalid_strings(raw: str) -> None: + with pytest.raises(ValueError): + tomlkit.integer(raw) + + +def test_integer_from_int_unchanged() -> None: + assert tomlkit.integer(1000).as_string() == "1000" + assert int(tomlkit.integer(1000)) == 1000 diff --git a/tomlkit/api.py b/tomlkit/api.py index b0f8cd6d..69fdbb53 100644 --- a/tomlkit/api.py +++ b/tomlkit/api.py @@ -2,6 +2,7 @@ import contextlib import datetime as _datetime +import re from collections.abc import Iterable from collections.abc import Mapping @@ -114,8 +115,33 @@ def document() -> TOMLDocument: # Items +_INTEGER_RE = re.compile( + r"(?:" + r"0x[0-9a-fA-F](?:_?[0-9a-fA-F])*" # hex: lowercase prefix only, no sign + r"|0o[0-7](?:_?[0-7])*" # octal: lowercase prefix only, no sign + r"|0b[01](?:_?[01])*" # binary: lowercase prefix only, no sign + r"|[+-]?(?:0|[1-9](?:_?[0-9])*)" # decimal: optional sign, no leading zeros + r")" +) +"""Valid TOML v1.0.0 integer literals (https://toml.io/en/v1.0.0#integer). + +Underscores may only appear between digits; hex/octal/binary use a +lowercase prefix and take no sign, mirroring ``Parser._parse_number``. +""" + + def integer(raw: str | int) -> Integer: - """Create an integer item from a number or string.""" + """Create an integer item from a number or string. + + When ``raw`` is a string it must be a valid TOML v1.0.0 integer + literal; the original text is then preserved verbatim so grouping + underscores (e.g. ``"1_000"``) survive ``dumps()``. Anything else + raises ``ValueError``. + """ + if isinstance(raw, str): + if not _INTEGER_RE.fullmatch(raw): + raise ValueError(f"Invalid TOML integer: {raw!r}") + return Integer(int(raw, 0), Trivia(), raw) return item(int(raw))