From 5eb2a8de211b0bbdacea2eca15bf31dc35ea1889 Mon Sep 17 00:00:00 2001 From: Sai Asish Y Date: Fri, 2 Oct 2026 13:24:49 -0700 Subject: [PATCH] Support locale-defined scales in intword --- README.md | 36 ++ .../locale/ja_JP/LC_MESSAGES/humanize.po | 8 + src/humanize/number.py | 88 +++++ tests/test_intword_scales.py | 309 ++++++++++++++++++ 4 files changed, 441 insertions(+) create mode 100644 tests/test_intword_scales.py diff --git a/README.md b/README.md index e14dec69..65b6abb7 100644 --- a/README.md +++ b/README.md @@ -249,3 +249,39 @@ Where `` is a locale abbreviation, eg. `en_GB`, `pt_BR` or just `ru etc. List the language at the top of this README. + +### Localized large-number scales + +By default, `intword()` uses the existing thousand/million/etc. translations. A locale +can instead opt into its own named powers of ten through two JSON translations. For +example, a catalog with two plural forms can contain: + +```po +msgid "intword:scales:v1" +msgstr "[4, 8]" + +msgid "intword:patterns:v1" +msgid_plural "intword:patterns:v1" +msgstr[0] "{\"exponents\": [4, 8], \"patterns\": [\"unit {number}\", \"large unit {number}\"]}" +msgstr[1] "{\"exponents\": [4, 8], \"patterns\": [\"units {number}\", \"large units {number}\"]}" +``` + +Exponents must be strictly increasing positive integers, at most 308 (the finite float +range accepted by `intword`). Each plural translation must repeat the exact exponent +array and provide one pattern for every exponent. This binding prevents a regional +catalog's scales from being combined with different scales inherited from a fallback +catalog. Each pattern must contain exactly one literal `{number}` placeholder and no +other braces. It controls the number's position and spacing; the caller's `format` and +the locale's decimal separator still apply. Plural selection uses the same +rounded-number rule as the existing unit messages. + +Values below the first named power remain integers. Larger values use the highest named +power as a multiple, without inventing further units. Japanese uses the +[modern four-digit scale](https://www-utap.phys.s.u-tokyo.ac.jp/~suto/myresearch/motomura-scafe-2018Oct26.pdf#page=29) +from 万 (10⁴) through 無量大数 (10⁶⁸), so 234909023 and 2349090 examples become `2.3億` +and `234.9万`. + +Missing, fuzzy, or malformed profile entries use the existing unit translations. All +plural forms should supply a complete matching profile. The usual translation update +script extracts these messages and preserves them during catalog merging; no changes are +needed in other locales. diff --git a/src/humanize/locale/ja_JP/LC_MESSAGES/humanize.po b/src/humanize/locale/ja_JP/LC_MESSAGES/humanize.po index 83bf7f1a..3e2aafdd 100644 --- a/src/humanize/locale/ja_JP/LC_MESSAGES/humanize.po +++ b/src/humanize/locale/ja_JP/LC_MESSAGES/humanize.po @@ -352,3 +352,11 @@ msgstr "昨日" #, python-format msgid "%s and %s" msgstr "" + +#. Named powers of ten and matching number/unit patterns for intword. +msgid "intword:scales:v1" +msgstr "[4, 8, 12, 16, 20, 24, 28, 32, 36, 40, 44, 48, 52, 56, 60, 64, 68]" + +msgid "intword:patterns:v1" +msgid_plural "intword:patterns:v1" +msgstr[0] "{\"exponents\": [4, 8, 12, 16, 20, 24, 28, 32, 36, 40, 44, 48, 52, 56, 60, 64, 68], \"patterns\": [\"{number}万\", \"{number}億\", \"{number}兆\", \"{number}京\", \"{number}垓\", \"{number}秭\", \"{number}穣\", \"{number}溝\", \"{number}澗\", \"{number}正\", \"{number}載\", \"{number}極\", \"{number}恒河沙\", \"{number}阿僧祇\", \"{number}那由他\", \"{number}不可思議\", \"{number}無量大数\"]}" diff --git a/src/humanize/number.py b/src/humanize/number.py index 52a5356a..f96c10d0 100644 --- a/src/humanize/number.py +++ b/src/humanize/number.py @@ -220,6 +220,90 @@ def intcomma(value: NumberOrString, ndigits: int | None = None) -> str: ) +# Catalogs opt in without changing the existing unit translations. +_INTWORD_SCALES = N_("intword:scales:v1") +_INTWORD_PATTERNS = NS_("intword:patterns:v1", "intword:patterns:v1") + + +def _valid_intword_exponents(exponents: object) -> bool: + """Check named powers against the finite float domain accepted by intword.""" + import sys + + return ( + isinstance(exponents, list) + and bool(exponents) + and all( + type(exponent) is int and 0 < exponent <= sys.float_info.max_10_exp + for exponent in exponents + ) + and all(left < right for left, right in zip(exponents, exponents[1:])) + ) + + +def _intword_patterns(exponents: list[int], count: int) -> list[str] | None: + """Read a plural pattern table bound to the same catalog scale vector.""" + import json + + try: + payload = json.loads(_ngettext(*_INTWORD_PATTERNS, count)) + except ValueError: + return None + if not isinstance(payload, dict): + return None + bound_exponents = payload.get("exponents") + patterns = payload.get("patterns") + if ( + not _valid_intword_exponents(bound_exponents) + or bound_exponents != exponents + or not isinstance(patterns, list) + or len(patterns) != len(exponents) + or not all( + isinstance(pattern, str) + and pattern.count("{number}") == 1 + and "{" not in pattern.replace("{number}", "") + and "}" not in pattern.replace("{number}", "") + for pattern in patterns + ) + ): + return None + return patterns + + +def _translated_intword(value: int, format: str, negative_prefix: str) -> str | None: + """Format with an optional translation-defined scale, or use the legacy path.""" + translated_scales = _(_INTWORD_SCALES) + if translated_scales == _INTWORD_SCALES: + return None + + import json + import math + + try: + exponents = json.loads(translated_scales) + except ValueError: + return None + if ( + not _valid_intword_exponents(exponents) + or _intword_patterns(exponents, 1) is None + ): + return None + localized_powers = tuple(10**exponent for exponent in exponents) + if value < localized_powers[0]: + return f"{negative_prefix}{value}" + ordinal = bisect.bisect_right(localized_powers, value) - 1 + rounded = float(format % (value / localized_powers[ordinal])) + if ordinal + 1 < len(localized_powers) and rounded == float( + localized_powers[ordinal + 1] // localized_powers[ordinal] + ): + ordinal += 1 + rounded = 1.0 + patterns = _intword_patterns(exponents, math.ceil(rounded)) + if patterns is None: + return None + number = negative_prefix + (format % rounded).replace(".", decimal_separator()) + return patterns[ordinal].replace("{number}", number) + + def intword(value: NumberOrString, format: str = "%.1f") -> str: """Converts a large integer to a friendly text representation. @@ -270,6 +354,10 @@ def intword(value: NumberOrString, format: str = "%.1f") -> str: else: negative_prefix = "" + translated = _translated_intword(value, format, negative_prefix) + if translated is not None: + return translated + if value < powers[0]: return f"{negative_prefix}{value}" diff --git a/tests/test_intword_scales.py b/tests/test_intword_scales.py new file mode 100644 index 00000000..b61cf28e --- /dev/null +++ b/tests/test_intword_scales.py @@ -0,0 +1,309 @@ +"""Translation-defined intword scales and catalog compatibility.""" + +from __future__ import annotations + +import gettext +import json +import shutil +import subprocess +from concurrent.futures import ThreadPoolExecutor +from pathlib import Path +from threading import Barrier + +import pytest + +import humanize + + +@pytest.fixture(autouse=True) +def isolated_translations(monkeypatch: pytest.MonkeyPatch): + monkeypatch.setattr( + humanize.i18n, "_TRANSLATIONS", {None: gettext.NullTranslations()} + ) + humanize.deactivate() + yield + humanize.deactivate() + + +def compile_catalog(path: Path) -> None: + subprocess.run( + ["msgfmt", "--check", "-o", str(path.with_suffix(".mo")), str(path)], + check=True, + ) + + +def write_catalog( + root: Path, + locale: str, + scales: object, + patterns: object | None, + plural_patterns: object | None = None, +) -> None: + path = root / locale / "LC_MESSAGES" / "humanize.po" + path.parent.mkdir(parents=True, exist_ok=True) + header = ( + "Project-Id-Version: humanize test\n" + "PO-Revision-Date: 2026-10-02 00:00+0000\n" + "Last-Translator: Test\n" + "Language-Team: Test\n" + "MIME-Version: 1.0\n" + "Content-Transfer-Encoding: 8bit\n" + "Content-Type: text/plain; charset=UTF-8\n" + "Plural-Forms: nplurals=2; plural=(n != 1);\n" + f"Language: {locale}\n" + ) + text = f'msgid ""\nmsgstr {json.dumps(header)}\n\n' + text += 'msgid "intword:scales:v1"\n' + text += f"msgstr {json.dumps(json.dumps(scales))}\n\n" + if patterns is not None: + plural_patterns = patterns if plural_patterns is None else plural_patterns + text += 'msgid "intword:patterns:v1"\nmsgid_plural "intword:patterns:v1"\n' + text += f"msgstr[0] {json.dumps(json.dumps(patterns))}\n" + text += f"msgstr[1] {json.dumps(json.dumps(plural_patterns))}\n" + path.write_text(text, encoding="utf-8") + compile_catalog(path) + + +@pytest.mark.parametrize( + "value,expected", + [(234909023, "2.3億"), (2349090, "234.9万"), (-2349090, "-234.9万")], +) +def test_japanese_intword(tmp_path: Path, value: int, expected: str) -> None: + source = ( + Path(humanize.i18n.__file__).parent / "locale/ja_JP/LC_MESSAGES/humanize.po" + ) + destination = tmp_path / "ja_JP/LC_MESSAGES/humanize.po" + destination.parent.mkdir(parents=True) + shutil.copyfile(source, destination) + compile_catalog(destination) + humanize.activate("ja_JP", path=tmp_path) + assert humanize.intword(value) == expected + + +def test_translated_scale_plural_and_rollover(tmp_path: Path) -> None: + write_catalog( + tmp_path, + "zz", + [4, 8], + {"exponents": [4, 8], "patterns": ["unit {number}", "large {number}"]}, + {"exponents": [4, 8], "patterns": ["units {number}", "large units {number}"]}, + ) + humanize.activate("zz", path=tmp_path) + assert humanize.intword(9999) == "9999" + assert humanize.intword(10000) == "unit 1.0" + assert humanize.intword(20000) == "units 2.0" + assert humanize.intword(99999999) == "large 1.0" + assert humanize.intword(200000000) == "large units 2.0" + assert humanize.intword(-10000) == "unit -1.0" + assert humanize.intword(-99999999) == "large -1.0" + assert humanize.intword(-9999) == "-9999" + + +def test_regional_scale_mismatch_uses_legacy(tmp_path: Path) -> None: + write_catalog( + tmp_path, + "zz", + [4, 8], + {"exponents": [4, 8], "patterns": ["{number}万", "{number}億"]}, + ) + write_catalog(tmp_path, "zz_ZZ", [3, 6], None) + translation = humanize.activate("zz_ZZ", path=tmp_path) + assert translation is not None + assert json.loads(translation.gettext("intword:scales:v1")) == [3, 6] + inherited = translation.ngettext("intword:patterns:v1", "intword:patterns:v1", 1) + assert json.loads(inherited)["exponents"] == [4, 8] + assert humanize.intword(1000) == "1.0 thousand" + assert humanize.intword(1000000) == "1.0 million" + + +def test_untranslated_scale_uses_legacy() -> None: + assert humanize.intword(234909023) == "234.9 million" + assert humanize.intword(10**36) == "1000.0 decillion" + assert humanize.intword(2 * 10**100) == "2.0 googol" + + +@pytest.mark.parametrize( + "scales,patterns", + [ + ([], {"exponents": [], "patterns": []}), + ([0], {"exponents": [0], "patterns": ["{number} unit"]}), + ([309], {"exponents": [309], "patterns": ["{number} unit"]}), + ([4, 4], {"exponents": [4, 4], "patterns": ["{number}", "{number}"]}), + ([8, 4], {"exponents": [8, 4], "patterns": ["{number}", "{number}"]}), + ([True], {"exponents": [True], "patterns": ["{number} unit"]}), + ([4.0], {"exponents": [4.0], "patterns": ["{number} unit"]}), + ([4], {"exponents": [4.0], "patterns": ["{number} unit"]}), + ([1], {"exponents": [True], "patterns": ["{number} unit"]}), + ([4], {"exponents": [4], "patterns": []}), + ([4], {"exponents": [4], "patterns": ["missing placeholder"]}), + ([4], {"exponents": [4], "patterns": ["{number}{number}"]}), + ([4], {"exponents": [4], "patterns": ["{number}{other}"]}), + ([4], {"exponents": [4], "patterns": [1]}), + ([4], ["{number} unit"]), + ("invalid", {"exponents": [4], "patterns": ["{number} unit"]}), + ], +) +def test_invalid_catalog_uses_legacy( + tmp_path: Path, scales: object, patterns: object +) -> None: + write_catalog(tmp_path, "zz", scales, patterns) + humanize.activate("zz", path=tmp_path) + assert humanize.intword(20000) == "20.0 thousand" + + +def test_plural_scale_mismatch_uses_legacy(tmp_path: Path) -> None: + write_catalog( + tmp_path, + "zz", + [4], + {"exponents": [4], "patterns": ["unit {number}"]}, + {"exponents": [3], "patterns": ["units {number}"]}, + ) + humanize.activate("zz", path=tmp_path) + assert humanize.intword(10000) == "unit 1.0" + assert humanize.intword(20000) == "20.0 thousand" + + +def test_scales_above_googol(tmp_path: Path) -> None: + write_catalog( + tmp_path, + "zz", + [104, 308], + {"exponents": [104, 308], "patterns": ["{number} high", "{number} highest"]}, + ) + humanize.activate("zz", path=tmp_path) + assert humanize.intword(10**104) == "1.0 high" + assert humanize.intword(2 * 10**104) == "2.0 high" + assert humanize.intword(10**308) == "1.0 highest" + + +def test_catalog_does_not_hide_caller_errors(tmp_path: Path) -> None: + write_catalog( + tmp_path, "zz", [4], {"exponents": [4], "patterns": ["{number} unit"]} + ) + humanize.activate("zz", path=tmp_path) + with pytest.raises(ValueError, match="unsupported format character"): + humanize.intword(20000, "%q") + with pytest.raises(TypeError): + humanize.intword(20000, None) + assert humanize.intword("not a number") == "not a number" + assert humanize.intword(float("inf")) == "+Inf" + assert humanize.intword(float("nan")) == "NaN" + assert humanize.intword(20000, "%.2f") == "2.00 unit" + + +def test_catalog_scales_are_thread_local(tmp_path: Path) -> None: + write_catalog( + tmp_path, "zz", [4], {"exponents": [4], "patterns": ["{number} unit"]} + ) + write_catalog( + tmp_path, "yy", [3], {"exponents": [3], "patterns": ["other {number}"]} + ) + original_powers = humanize.number.powers.copy() + barrier = Barrier(2) + + def render(locale: str) -> str: + humanize.activate(locale, path=tmp_path) + try: + barrier.wait(timeout=5) + return humanize.intword(20000) + finally: + humanize.deactivate() + + with ThreadPoolExecutor(max_workers=2) as executor: + assert list(executor.map(render, ["zz", "yy", "zz", "yy"])) == [ + "2.0 unit", + "other 20.0", + "2.0 unit", + "other 20.0", + ] + assert humanize.intword(20000) == "20.0 thousand" + assert humanize.number.powers == original_powers + + +def test_japanese_profile_survives_translation_update(tmp_path: Path) -> None: + root = Path(__file__).resolve().parents[1] + source = root / "src/humanize" + destination = tmp_path / "src/humanize" + messages = destination / "locale/ja_JP/LC_MESSAGES" + messages.mkdir(parents=True) + for path in source.glob("*.py"): + shutil.copyfile(path, destination / path.name) + catalog = messages / "humanize.po" + shutil.copyfile(source / "locale/ja_JP/LC_MESSAGES/humanize.po", catalog) + subprocess.run( + ["bash", str(root / "scripts/update-translations.sh")], cwd=tmp_path, check=True + ) + humanize.activate("ja_JP", path=destination / "locale") + units = [ + "万", + "億", + "兆", + "京", + "垓", + "秭", + "穣", + "溝", + "澗", + "正", + "載", + "極", + "恒河沙", + "阿僧祇", + "那由他", + "不可思議", + "無量大数", + ] + for exponent, unit in zip(range(4, 69, 4), units): + assert humanize.intword(10**exponent) == f"1.0{unit}" + assert humanize.intword(10**72) == "10000.0無量大数" + assert humanize.intword(10**104, "%.0e") == "1e+36無量大数" + assert humanize.intword(99999999) == "1.0億" + assert humanize.intword(9999) == "9999" + + +def test_fuzzy_profile_uses_legacy(tmp_path: Path) -> None: + write_catalog( + tmp_path, "zz", [4], {"exponents": [4], "patterns": ["{number} unit"]} + ) + catalog = tmp_path / "zz/LC_MESSAGES/humanize.po" + catalog.write_text( + catalog.read_text().replace( + 'msgid "intword:patterns:v1"', '#, fuzzy\nmsgid "intword:patterns:v1"' + ) + ) + compile_catalog(catalog) + humanize.activate("zz", path=tmp_path) + assert humanize.intword(20000) == "20.0 thousand" + + +def test_profile_uses_locale_decimal_separator(tmp_path: Path) -> None: + write_catalog( + tmp_path, "fr_FR", [4], {"exponents": [4], "patterns": ["{number} unités"]} + ) + humanize.activate("fr_FR", path=tmp_path) + assert humanize.intword(-2349090, "%.2f") == "-234,91 unités" + + +def test_malformed_scale_json_uses_legacy(tmp_path: Path) -> None: + write_catalog( + tmp_path, "zz", [4], {"exponents": [4], "patterns": ["{number} unit"]} + ) + catalog = tmp_path / "zz/LC_MESSAGES/humanize.po" + catalog.write_text(catalog.read_text().replace('msgstr "[4]"', 'msgstr "not JSON"')) + compile_catalog(catalog) + humanize.activate("zz", path=tmp_path) + assert humanize.intword(20000) == "20.0 thousand" + + +@pytest.mark.parametrize("exponents", [[4, 28], [104, 308]]) +def test_wide_scale_rounding_rollover(tmp_path: Path, exponents: list[int]) -> None: + write_catalog( + tmp_path, + "zz", + exponents, + {"exponents": exponents, "patterns": ["{number} lower", "{number} upper"]}, + ) + humanize.activate("zz", path=tmp_path) + assert humanize.intword(10 ** exponents[1] - 1) == "1.0 upper" + assert humanize.intword(-(10 ** exponents[1] - 1)) == "-1.0 upper"