import re from typing import List, Optional, Tuple from .validation_uk_modulus_table_types import UkModulusRow, UkModulusTable, UkSortCodeSubstitution _SORT_CODE = re.compile(r"[0-9]{6}") _WEIGHT = re.compile(r"-?[0-9]{1,4}") _EXCEPTION = re.compile(r"[0-9]{1,2}") _SEPARATOR = re.compile(r"[ \t]+") _ALGORITHMS = ("MOD10", "MOD11", "DBLAL") def _lines(text: str) -> List[Tuple[int, List[str]]]: """The fields of each non-blank line. Only ASCII spaces and tabs separate fields, and a trailing carriage return is dropped, so a file saved with Windows line endings reads the same and every language splits alike.""" out: List[Tuple[int, List[str]]] = [] body = text[1:] if text.startswith("") else text for i, raw in enumerate(body.split("\n")): line = raw[:-1] if raw.endswith("\r") else raw fields = [f for f in _SEPARATOR.split(line) if f != ""] if fields: out.append((i + 1, fields)) return out def _sort_code(file: str, line: int, value: str) -> str: if not _SORT_CODE.fullmatch(value): raise ValueError('%s line %d: "%s" is not a six-digit sort code' % (file, line, value)) return value def _parse_row(line: int, fields: List[str]) -> UkModulusRow: where = "VALACDOS.txt line %d" % line if len(fields) not in (17, 18): raise ValueError( "%s: expected 17 or 18 fields (start, end, algorithm, 14 weights, optional exception), found %d" % (where, len(fields)) ) start = _sort_code("VALACDOS.txt", line, fields[0]) end = _sort_code("VALACDOS.txt", line, fields[1]) if end < start: raise ValueError("%s: range %s to %s ends before it starts" % (where, start, end)) algorithm = fields[2] if algorithm not in _ALGORITHMS: raise ValueError('%s: unknown algorithm "%s"; expected MOD10, MOD11 or DBLAL' % (where, algorithm)) weights: List[int] = [] for w in fields[3:17]: if not _WEIGHT.fullmatch(w): raise ValueError('%s: weight "%s" is not a whole number' % (where, w)) weights.append(int(w)) # A digit sum of a negative product means different things in different # languages' remainder rules; the specification never needs one. if algorithm == "DBLAL" and any(w < 0 for w in weights): raise ValueError("%s: a DBLAL row cannot have a negative weight" % where) exception: Optional[int] = None if len(fields) == 18: e = fields[17] if not _EXCEPTION.fullmatch(e) or not 1 <= int(e) <= 14: raise ValueError('%s: exception "%s" is not a number from 1 to 14' % (where, e)) exception = int(e) return UkModulusRow(start=start, end=end, algorithm=algorithm, weights=weights, exception=exception) def _parse_substitution(line: int, fields: List[str]) -> UkSortCodeSubstitution: if len(fields) != 2: raise ValueError("SCSUBTAB.txt line %d: expected two sort codes, found %d fields" % (line, len(fields))) return UkSortCodeSubstitution( original=_sort_code("SCSUBTAB.txt", line, fields[0]), substitute=_sort_code("SCSUBTAB.txt", line, fields[1]), ) def parse_uk_modulus_table(valacdos: str, scsubtab: str) -> UkModulusTable: """Parse the text of Vocalink's VALACDOS.txt (the modulus weight table) and SCSUBTAB.txt (exception 5's sort code substitutions) into the table validate_uk_sort_code_account checks against. Rows keep their file order, which matters: where a sort code falls in two rows, the first is the first check. Blank lines are skipped; anything else malformed is an error naming the file and the line, because a silently dropped row turns a checkable sort code into an unchecked one. """ rows = [_parse_row(number, fields) for number, fields in _lines(valacdos)] if not rows: raise ValueError("VALACDOS.txt has no rows") substitutions = [_parse_substitution(number, fields) for number, fields in _lines(scsubtab)] return UkModulusTable(rows=rows, substitutions=substitutions)