from typing import Callable, List, Optional, Tuple from .money_amount import Money, money from .money_currency_digits import currency_digits from .money_currency_digits_data import CURRENCY_DIGITS from .money_parse_data import CURRENCY_SYMBOLS from .money_parse_types import NumberStyle MAX_SAFE = 9007199254740991 # Space, no-break space, narrow no-break space: what real documents use. SPACES = (" ", " ", " ") def _style_of(style: str) -> Tuple[Tuple[str, ...], str]: if style == "comma-dot": return (",",), "." if style == "dot-comma": return (".",), "," if style == "space-comma": return SPACES, "," if style == "apostrophe-dot": return ("'", "’"), "." raise ValueError('unknown number style "%s"' % (style,)) def _is_digits(s: str) -> bool: # Not str.isdigit: that accepts other scripts' digits and superscripts. return s != "" and all(ch in "0123456789" for ch in s) def _longest_symbol(matches: Callable[[str], bool]) -> Optional[str]: """The longest known symbol or ISO code that ``matches``, or None.""" best: Optional[str] = None candidates = [row.symbol for row in CURRENCY_SYMBOLS] + [row.code for row in CURRENCY_DIGITS] for symbol in candidates: if matches(symbol) and (best is None or len(symbol) > len(best)): best = symbol return best def _split_number(s: str, groups: Tuple[str, ...], decimal: str) -> Optional[Tuple[str, str]]: """Whole digits and fraction digits, or None when the text is not a number in this style.""" parts = s.split(decimal) if len(parts) > 2: return None fraction = parts[1] if len(parts) == 2 else "" if len(parts) == 2 and not _is_digits(fraction): return None chunks: List[str] = [""] for ch in parts[0]: if ch in groups: chunks.append("") else: chunks[-1] += ch if not all(_is_digits(c) for c in chunks): return None if len(chunks) > 1: # Real thousands grouping only: 1-3 digits without a leading zero, then threes. first, rest = chunks[0], chunks[1:] if len(first) > 3 or first.startswith("0"): return None if not all(len(c) == 3 for c in rest): return None return "".join(chunks), fraction def parse_money(text: str, currency: str, style: NumberStyle) -> Money: """Parse an amount written as text into integer minor units. Strict by design: the caller names the currency and the number style, and anything ambiguous (too many decimals, odd grouping, a symbol for another currency) is an error rather than a guess. """ groups, decimal = _style_of(style) digits = currency_digits(currency) s = text.strip(" ") negative = False if s.startswith("-"): negative = True s = s[1:] symbol = _longest_symbol(lambda k: s.startswith(k)) if symbol is not None: s = s[len(symbol):] if s[:1] in SPACES and s != "": s = s[1:] if not negative and s.startswith("-"): negative = True s = s[1:] else: symbol = _longest_symbol(lambda k: s.endswith(k)) if symbol is not None: s = s[: len(s) - len(symbol)] if s != "" and s[-1] in SPACES: s = s[:-1] if symbol is not None and symbol != currency and not any( r.code == currency and r.symbol == symbol for r in CURRENCY_SYMBOLS ): raise ValueError('currency symbol "%s" does not match %s' % (symbol, currency)) number = _split_number(s, groups, decimal) if number is None: raise ValueError('"%s" is not a valid amount' % (text,)) whole, fraction = number if len(fraction) > digits: raise ValueError("too many decimal places for %s (at most %d)" % (currency, digits)) minor = int(whole + fraction.ljust(digits, "0")) if minor > MAX_SAFE: raise ValueError('amount is too large: "%s" exceeds 2^53 - 1 minor units' % (text,)) return money(-minor if negative else minor, currency)