Functional Weave
Code in Python

todo.import-csv@1.0.0

impl/python.py

5,135 bytes · the Python implementation · view raw

Imports name this capability’s declared dependencies, which fune builds next to it in your project; each one links to its page.

import re
from typing import Dict, List, NoReturn, Optional, Set, Tuple

from .todo_item import Recurrence, Todo, validate_todo  ← from todo.item ^1.0.0 · built alongside by fune
from .todo_normalise_tags import normalise_tags  ← from todo.normalise-tags ^1.0.0 · built alongside by fune

_COLUMNS = (
    "id", "title", "notes", "done", "priority", "due", "tags",
    "recurrenceFrequency", "recurrenceInterval", "recurrenceAnchor", "createdAt", "completedAt", "order",
)
_HEADER_RULE = "the header must have the columns " + ", ".join(_COLUMNS)
_WHOLE = re.compile(r"-?[0-9]{1,15}")


def _fail(row: int, message: str) -> NoReturn:
    raise ValueError(f"row {row}: {message}")


def _read_record(text: str, pos: int, row: int) -> Tuple[List[str], int]:
    """One RFC 4180 record from pos: its fields and where the next record starts."""
    fields: List[str] = []
    size = len(text)
    while True:
        if pos < size and text[pos] == '"':
            pos += 1
            parts: List[str] = []
            while True:
                quote = text.find('"', pos)
                if quote < 0:
                    _fail(row, "a quoted field is not closed")
                parts.append(text[pos:quote])
                if quote + 1 < size and text[quote + 1] == '"':
                    parts.append('"')
                    pos = quote + 2
                else:
                    pos = quote + 1
                    break
            if pos < size and text[pos] not in ",\r\n":
                _fail(row, "text after the closing quote of a field")
            value = "".join(parts)
        else:
            end = pos
            while end < size and text[end] not in ",\r\n":
                if text[end] == '"':
                    _fail(row, "a field with a double quote in it must be quoted")
                end += 1
            value = text[pos:end]
            pos = end
        fields.append(value)
        if pos >= size:
            return fields, pos
        if text[pos] == ",":
            pos += 1
        elif text[pos] == "\n":
            return fields, pos + 1
        elif pos + 1 < size and text[pos + 1] == "\n":
            return fields, pos + 2
        else:
            _fail(row, "a CR outside quotes must be followed by LF")


def _read_header(names: List[str]) -> List[int]:
    at: Dict[str, int] = {}
    for i, name in enumerate(names):
        if name not in _COLUMNS:
            raise ValueError(f'{_HEADER_RULE}: unknown column "{name}"')
        if name in at:
            raise ValueError(f'{_HEADER_RULE}: column "{name}" appears twice')
        at[name] = i
    for name in _COLUMNS:
        if name not in at:
            raise ValueError(f'{_HEADER_RULE}: missing column "{name}"')
    return [at[name] for name in _COLUMNS]


def _whole(row: int, column: str, text: str) -> int:
    if not _WHOLE.fullmatch(text):
        _fail(row, f'{column} must be a whole number, found "{text}"')
    return int(text)


def _read_todo(fields: List[str], at: List[int], row: int) -> Todo:
    if len(fields) != len(_COLUMNS):
        _fail(row, f"expected {len(_COLUMNS)} fields, found {len(fields)}")
    (id_, title, notes, done, priority, due, tags, frequency, interval, anchor, created_at, completed_at,
     order) = [fields[i] for i in at]
    if done not in ("true", "false"):
        _fail(row, f'done must be true or false, found "{done}"')
    filled = sum(1 for text in (frequency, interval, anchor) if text != "")
    recurrence: Optional[Recurrence] = None
    if filled == 3:
        recurrence = Recurrence(frequency=frequency, interval=_whole(row, "recurrenceInterval", interval), anchor=anchor)
    elif filled != 0:
        _fail(row, "fill in all three recurrence columns or leave them all empty")
    return Todo(
        id=id_,
        title=title,
        notes=None if notes == "" else notes,
        done=done == "true",
        priority=priority,
        due=None if due == "" else due,
        tags=normalise_tags(tags.split(" ")),
        recurrence=recurrence,
        created_at=created_at,
        completed_at=None if completed_at == "" else completed_at,
        order=_whole(row, "order", order),
    )


def import_csv(csv: str) -> List[Todo]:
    """Todos from CSV: 13 columns in any order, CRLF or LF, optional BOM; errors name the row (header = row 1)."""
    text = csv[1:] if csv.startswith("") else csv
    todos: List[Todo] = []
    ids: Set[str] = set()
    at: Optional[List[int]] = None
    row = 0
    pos = 0
    size = len(text)
    while pos < size:
        if text[pos] == "\n":
            pos += 1
            continue
        if text.startswith("\r\n", pos):
            pos += 2
            continue
        row += 1
        fields, pos = _read_record(text, pos, row)
        if at is None:
            at = _read_header(fields)
            continue
        todo = _read_todo(fields, at, row)
        errors = validate_todo(todo).errors
        for field, message in errors.items():
            _fail(row, f"{field}: {message}")
        if todo.id in ids:
            _fail(row, f'duplicate id "{todo.id}"')
        ids.add(todo.id)
        todos.append(todo)
    if at is None:
        raise ValueError("the CSV has no header row")
    return todos