Functional Weave
Code in Rust

todo.import-csv@1.1.0

impl/rust/validate_todo_csv.rs

10,649 bytes · the Rust implementation · view raw

Imports name this capability’s declared dependencies, which fune builds next to it in your project; each one links to its page.

use std::collections::HashSet;

use super::funejson::Value;  ← the fune runtime: the JSON value the test vectors use; fune build keeps it only where a signature takes one
use super::todo_item::{validate_todo, Recurrence, Todo};  ← from todo.item ^1.0.0 · built alongside by fune
use super::todo_normalise_tags::normalise_tags;  ← from todo.normalise-tags ^1.0.0 · built alongside by fune

const COLUMNS: [&str; 13] = [
    "id", "title", "notes", "done", "priority", "due", "tags",
    "recurrenceFrequency", "recurrenceInterval", "recurrenceAnchor", "createdAt", "completedAt", "order",
];

/// One problem, and the message import_csv panics with when it is the first.

pub struct TodoCsvProblem {
    pub error: CsvRowError,
    pub thrown: String,
}

/// What reading a whole file found: the good rows' todos, and every problem.
pub struct TodoCsvReading {
    pub todos: Vec<Todo>,
    pub rows: i64,
    pub problems: Vec<TodoCsvProblem>,
}

fn header_rule() -> String {
    format!("the header must have the columns {}", COLUMNS.join(", "))
}

/// One RFC 4180 record from `pos`: its fields and where the next record
/// starts, or why it cannot be read. Only ASCII bytes are looked at, so every
/// slice is valid UTF-8.
fn record_at(text: &str, mut pos: usize) -> Result<(Vec<String>, usize), String> {
    let b = text.as_bytes();
    let size = b.len();
    let mut fields: Vec<String> = Vec::new();
    loop {
        let value: String;
        if pos < size && b[pos] == b'"' {
            pos += 1;
            let mut out = String::new();
            loop {
                let quote = match text[pos..].find('"') {
                    Some(q) => pos + q,
                    None => return Err("a quoted field is not closed".to_string()),
                };
                out.push_str(&text[pos..quote]);
                if quote + 1 < size && b[quote + 1] == b'"' {
                    out.push('"');
                    pos = quote + 2;
                } else {
                    pos = quote + 1;
                    break;
                }
            }
            if pos < size && !matches!(b[pos], b',' | b'\r' | b'\n') {
                return Err("text after the closing quote of a field".to_string());
            }
            value = out;
        } else {
            let mut end = pos;
            while end < size && !matches!(b[end], b',' | b'\r' | b'\n') {
                if b[end] == b'"' {
                    return Err("a field with a double quote in it must be quoted".to_string());
                }
                end += 1;
            }
            value = text[pos..end].to_string();
            pos = end;
        }
        fields.push(value);
        if pos >= size {
            return Ok((fields, pos));
        }
        if b[pos] == b',' {
            pos += 1;
        } else if b[pos] == b'\n' {
            return Ok((fields, pos + 1));
        } else if pos + 1 < size && b[pos + 1] == b'\n' {
            return Ok((fields, pos + 2));
        } else {
            return Err("a CR outside quotes must be followed by LF".to_string());
        }
    }
}

/// Where each of COLUMNS sits in the file's header, or what is wrong with it.
fn columns_at(names: &[String]) -> Result<Vec<usize>, String> {
    let mut at: Vec<(&str, usize)> = Vec::new();
    for (i, name) in names.iter().enumerate() {
        if !COLUMNS.contains(&name.as_str()) {
            return Err(format!("{}: unknown column \"{}\"", header_rule(), name));
        }
        if at.iter().any(|(n, _)| n == name) {
            return Err(format!("{}: column \"{}\" appears twice", header_rule(), name));
        }
        at.push((name.as_str(), i));
    }
    let mut out = Vec::new();
    for column in COLUMNS.iter() {
        match at.iter().find(|(n, _)| n == column) {
            Some((_, i)) => out.push(*i),
            None => return Err(format!("{}: missing column \"{}\"", header_rule(), column)),
        }
    }
    Ok(out)
}

fn whole_number(text: &str) -> Option<i64> {
    let digits = text.strip_prefix('-').unwrap_or(text);
    if digits.is_empty() || digits.len() > 15 || !digits.bytes().all(|c| c.is_ascii_digit()) {
        return None;
    }
    Some(text.parse::<i64>().unwrap())
}

fn opt(text: &str) -> Option<String> {
    if text.is_empty() { None } else { Some(text.to_string()) }
}

fn problem(row: i64, field: Option<&str>, message: &str, thrown: String) -> TodoCsvProblem {
    TodoCsvProblem {
        error: CsvRowError { row, field: field.map(|f| f.to_string()), message: message.to_string() },
        thrown,
    }
}

/// Reads the whole file, collecting every problem; the first is the one
/// import_csv 1.0.0 panicked with. A header or a record that cannot be split
/// into fields ends the reading.
pub fn read_todo_csv(csv: &str) -> TodoCsvReading {
    let text = csv.strip_prefix('\u{feff}').unwrap_or(csv);
    let b = text.as_bytes();
    let mut todos: Vec<Todo> = Vec::new();
    let mut problems: Vec<TodoCsvProblem> = Vec::new();
    let mut ids: HashSet<String> = HashSet::new();
    let mut at: Option<Vec<usize>> = None;
    let mut row: i64 = 0;
    let mut rows: i64 = 0;
    let mut pos = 0;
    while pos < b.len() {
        if b[pos] == b'\n' {
            pos += 1;
            continue;
        }
        if b[pos] == b'\r' && pos + 1 < b.len() && b[pos + 1] == b'\n' {
            pos += 2;
            continue;
        }
        row += 1;
        let fields = match record_at(text, pos) {
            Err(message) => {
                problems.push(problem(row, None, &message, format!("row {}: {}", row, message)));
                break;
            }
            Ok((fields, next)) => {
                pos = next;
                fields
            }
        };
        let columns = match &at {
            None => match columns_at(&fields) {
                Err(message) => {
                    problems.push(problem(row, None, &message, message.clone()));
                    break;
                }
                Ok(columns) => {
                    at = Some(columns);
                    continue;
                }
            },
            Some(columns) => columns.clone(),
        };
        rows += 1;
        if fields.len() != COLUMNS.len() {
            let message = format!("expected {} fields, found {}", COLUMNS.len(), fields.len());
            problems.push(problem(row, None, &message, format!("row {}: {}", row, message)));
            continue;
        }
        let before = problems.len();
        let plain = |problems: &mut Vec<TodoCsvProblem>, field: &str, message: String| {
            problems.push(problem(row, Some(field), &message, format!("row {}: {}", row, message)));
        };
        let get = |i: usize| fields[columns[i]].as_str();
        let done = get(3);
        let done_read = done == "true" || done == "false";
        if !done_read {
            plain(&mut problems, "done", format!("done must be true or false, found \"{}\"", done));
        }
        let (frequency, interval, anchor) = (get(7), get(8), get(9));
        let filled = [frequency, interval, anchor].iter().filter(|t| !t.is_empty()).count();
        let mut recurrence: Option<Recurrence> = None;
        let mut recurrence_read = true;
        if filled == 3 {
            match whole_number(interval) {
                None => {
                    plain(&mut problems, "recurrenceInterval", format!("recurrenceInterval must be a whole number, found \"{}\"", interval));
                    recurrence_read = false;
                }
                Some(every) => {
                    recurrence = Some(Recurrence { frequency: frequency.to_string(), interval: every, anchor: anchor.to_string() });
                }
            }
        } else if filled != 0 {
            plain(&mut problems, "recurrence", "fill in all three recurrence columns or leave them all empty".to_string());
            recurrence_read = false;
        }
        let order = get(12);
        let position = whole_number(order);
        if position.is_none() {
            plain(&mut problems, "order", format!("order must be a whole number, found \"{}\"", order));
        }
        let tags: Vec<String> = get(6).split(' ').map(|t| t.to_string()).collect();
        let todo = Todo {
            id: get(0).to_string(),
            title: get(1).to_string(),
            notes: opt(get(2)),
            done: done == "true",
            priority: get(4).to_string(),
            due: opt(get(5)),
            tags: normalise_tags(&tags),
            recurrence,
            created_at: get(10).to_string(),
            completed_at: opt(get(11)),
            order: position.unwrap_or(0),
        };
        // A column that could not be read has already been reported; the
        // stand-in value must not raise a second, misleading message.
        let errors = validate_todo(&todo).errors;
        for (field, message) in errors.iter() {
            if (field == "completedAt" && !done_read) || (field == "recurrence" && !recurrence_read) || (field == "order" && position.is_none()) {
                continue;
            }
            problems.push(problem(row, Some(field.as_str()), message, format!("row {}: {}: {}", row, field, message)));
        }
        if !errors.iter().any(|(field, _)| field == "id") {
            if ids.contains(&todo.id) {
                plain(&mut problems, "id", format!("duplicate id \"{}\"", todo.id));
            }
            ids.insert(todo.id.clone());
        }
        if problems.len() == before {
            todos.push(todo);
        }
    }
    if at.is_none() && problems.is_empty() {
        let message = "the CSV has no header row";
        problems.push(problem(1, None, message, message.to_string()));
    }
    TodoCsvReading { todos, rows, problems }
}

/// Every problem in a todo CSV file, row by row: the row (header = 1), the
/// field or column, and the message. Valid exactly when import_csv accepts it.
pub fn validate_todo_csv(csv: &str) -> CsvValidation {
    let reading = read_todo_csv(csv);
    CsvValidation {
        valid: reading.problems.is_empty(),
        rows: reading.rows,
        errors: reading.problems.into_iter().map(|p| p.error).collect(),
    }
}

pub fn csv_validation_to_value(result: &CsvValidation) -> Value {
    let errors = result
        .errors
        .iter()
        .map(|e| {
            Value::obj(vec![
                ("row", Value::Int(e.row)),
                ("field", e.field.as_deref().map(Value::str).unwrap_or(Value::Null)),
                ("message", Value::str(&e.message)),
            ])
        })
        .collect();
    Value::obj(vec![("valid", Value::Bool(result.valid)), ("rows", Value::Int(result.rows)), ("errors", Value::Arr(errors))])
}

pub fn fune_vector(args: &[Value]) -> Value {
    csv_validation_to_value(&validate_todo_csv(args[0].as_str()))
}