Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
36 changes: 36 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -249,3 +249,39 @@ Where `<locale name>` is a locale abbreviation, eg. `en_GB`, `pt_BR` or just `ru
etc.

List the language at the top of this README.

### Localized large-number scales

By default, `intword()` uses the existing thousand/million/etc. translations. A locale
can instead opt into its own named powers of ten through two JSON translations. For
example, a catalog with two plural forms can contain:

```po
msgid "intword:scales:v1"
msgstr "[4, 8]"

msgid "intword:patterns:v1"
msgid_plural "intword:patterns:v1"
msgstr[0] "{\"exponents\": [4, 8], \"patterns\": [\"unit {number}\", \"large unit {number}\"]}"
msgstr[1] "{\"exponents\": [4, 8], \"patterns\": [\"units {number}\", \"large units {number}\"]}"
```

Exponents must be strictly increasing positive integers, at most 308 (the finite float
range accepted by `intword`). Each plural translation must repeat the exact exponent
array and provide one pattern for every exponent. This binding prevents a regional
catalog's scales from being combined with different scales inherited from a fallback
catalog. Each pattern must contain exactly one literal `{number}` placeholder and no
other braces. It controls the number's position and spacing; the caller's `format` and
the locale's decimal separator still apply. Plural selection uses the same
rounded-number rule as the existing unit messages.

Values below the first named power remain integers. Larger values use the highest named
power as a multiple, without inventing further units. Japanese uses the
[modern four-digit scale](https://www-utap.phys.s.u-tokyo.ac.jp/~suto/myresearch/motomura-scafe-2018Oct26.pdf#page=29)
from 万 (10⁴) through 無量大数 (10⁶⁸), so 234909023 and 2349090 examples become `2.3億`
and `234.9万`.

Missing, fuzzy, or malformed profile entries use the existing unit translations. All
plural forms should supply a complete matching profile. The usual translation update
script extracts these messages and preserves them during catalog merging; no changes are
needed in other locales.
8 changes: 8 additions & 0 deletions src/humanize/locale/ja_JP/LC_MESSAGES/humanize.po
Original file line number Diff line number Diff line change
Expand Up @@ -352,3 +352,11 @@ msgstr "昨日"
#, python-format
msgid "%s and %s"
msgstr ""

#. Named powers of ten and matching number/unit patterns for intword.
msgid "intword:scales:v1"
msgstr "[4, 8, 12, 16, 20, 24, 28, 32, 36, 40, 44, 48, 52, 56, 60, 64, 68]"

msgid "intword:patterns:v1"
msgid_plural "intword:patterns:v1"
msgstr[0] "{\"exponents\": [4, 8, 12, 16, 20, 24, 28, 32, 36, 40, 44, 48, 52, 56, 60, 64, 68], \"patterns\": [\"{number}万\", \"{number}億\", \"{number}兆\", \"{number}京\", \"{number}垓\", \"{number}秭\", \"{number}穣\", \"{number}溝\", \"{number}澗\", \"{number}正\", \"{number}載\", \"{number}極\", \"{number}恒河沙\", \"{number}阿僧祇\", \"{number}那由他\", \"{number}不可思議\", \"{number}無量大数\"]}"
88 changes: 88 additions & 0 deletions src/humanize/number.py
Original file line number Diff line number Diff line change
Expand Up @@ -220,6 +220,90 @@ def intcomma(value: NumberOrString, ndigits: int | None = None) -> str:
)


# Catalogs opt in without changing the existing unit translations.
_INTWORD_SCALES = N_("intword:scales:v1")
_INTWORD_PATTERNS = NS_("intword:patterns:v1", "intword:patterns:v1")


def _valid_intword_exponents(exponents: object) -> bool:
"""Check named powers against the finite float domain accepted by intword."""
import sys

return (
isinstance(exponents, list)
and bool(exponents)
and all(
type(exponent) is int and 0 < exponent <= sys.float_info.max_10_exp
for exponent in exponents
)
and all(left < right for left, right in zip(exponents, exponents[1:]))
)


def _intword_patterns(exponents: list[int], count: int) -> list[str] | None:
"""Read a plural pattern table bound to the same catalog scale vector."""
import json

try:
payload = json.loads(_ngettext(*_INTWORD_PATTERNS, count))
except ValueError:
return None
if not isinstance(payload, dict):
return None
bound_exponents = payload.get("exponents")
patterns = payload.get("patterns")
if (
not _valid_intword_exponents(bound_exponents)
or bound_exponents != exponents
or not isinstance(patterns, list)
or len(patterns) != len(exponents)
or not all(
isinstance(pattern, str)
and pattern.count("{number}") == 1
and "{" not in pattern.replace("{number}", "")
and "}" not in pattern.replace("{number}", "")
for pattern in patterns
)
):
return None
return patterns


def _translated_intword(value: int, format: str, negative_prefix: str) -> str | None:
"""Format with an optional translation-defined scale, or use the legacy path."""
translated_scales = _(_INTWORD_SCALES)
if translated_scales == _INTWORD_SCALES:
return None

import json
import math

try:
exponents = json.loads(translated_scales)
except ValueError:
return None
if (
not _valid_intword_exponents(exponents)
or _intword_patterns(exponents, 1) is None
):
return None
localized_powers = tuple(10**exponent for exponent in exponents)
if value < localized_powers[0]:
return f"{negative_prefix}{value}"
ordinal = bisect.bisect_right(localized_powers, value) - 1
rounded = float(format % (value / localized_powers[ordinal]))
if ordinal + 1 < len(localized_powers) and rounded == float(
localized_powers[ordinal + 1] // localized_powers[ordinal]
):
ordinal += 1
rounded = 1.0
patterns = _intword_patterns(exponents, math.ceil(rounded))
if patterns is None:
return None
number = negative_prefix + (format % rounded).replace(".", decimal_separator())
return patterns[ordinal].replace("{number}", number)


def intword(value: NumberOrString, format: str = "%.1f") -> str:
"""Converts a large integer to a friendly text representation.

Expand Down Expand Up @@ -270,6 +354,10 @@ def intword(value: NumberOrString, format: str = "%.1f") -> str:
else:
negative_prefix = ""

translated = _translated_intword(value, format, negative_prefix)
if translated is not None:
return translated

if value < powers[0]:
return f"{negative_prefix}{value}"

Expand Down
Loading