impl/rust/validate_todo_csv.rs
10,649 bytes · the Rust implementation · view raw
Imports name this capability’s declared dependencies, which fune builds next to it in your project; each one links to its page.
use std::collections::HashSet;
use super::funejson::Value; ← the fune runtime: the JSON value the test vectors use; fune build keeps it only where a signature takes one
use super::todo_item::{validate_todo, Recurrence, Todo}; ← from todo.item ^1.0.0 · built alongside by fune
use super::todo_normalise_tags::normalise_tags; ← from todo.normalise-tags ^1.0.0 · built alongside by fune
const COLUMNS: [&str; 13] = [
"id", "title", "notes", "done", "priority", "due", "tags",
"recurrenceFrequency", "recurrenceInterval", "recurrenceAnchor", "createdAt", "completedAt", "order",
];
/// One problem, and the message import_csv panics with when it is the first.
pub struct TodoCsvProblem {
pub error: CsvRowError,
pub thrown: String,
}
/// What reading a whole file found: the good rows' todos, and every problem.
pub struct TodoCsvReading {
pub todos: Vec<Todo>,
pub rows: i64,
pub problems: Vec<TodoCsvProblem>,
}
fn header_rule() -> String {
format!("the header must have the columns {}", COLUMNS.join(", "))
}
/// One RFC 4180 record from `pos`: its fields and where the next record
/// starts, or why it cannot be read. Only ASCII bytes are looked at, so every
/// slice is valid UTF-8.
fn record_at(text: &str, mut pos: usize) -> Result<(Vec<String>, usize), String> {
let b = text.as_bytes();
let size = b.len();
let mut fields: Vec<String> = Vec::new();
loop {
let value: String;
if pos < size && b[pos] == b'"' {
pos += 1;
let mut out = String::new();
loop {
let quote = match text[pos..].find('"') {
Some(q) => pos + q,
None => return Err("a quoted field is not closed".to_string()),
};
out.push_str(&text[pos..quote]);
if quote + 1 < size && b[quote + 1] == b'"' {
out.push('"');
pos = quote + 2;
} else {
pos = quote + 1;
break;
}
}
if pos < size && !matches!(b[pos], b',' | b'\r' | b'\n') {
return Err("text after the closing quote of a field".to_string());
}
value = out;
} else {
let mut end = pos;
while end < size && !matches!(b[end], b',' | b'\r' | b'\n') {
if b[end] == b'"' {
return Err("a field with a double quote in it must be quoted".to_string());
}
end += 1;
}
value = text[pos..end].to_string();
pos = end;
}
fields.push(value);
if pos >= size {
return Ok((fields, pos));
}
if b[pos] == b',' {
pos += 1;
} else if b[pos] == b'\n' {
return Ok((fields, pos + 1));
} else if pos + 1 < size && b[pos + 1] == b'\n' {
return Ok((fields, pos + 2));
} else {
return Err("a CR outside quotes must be followed by LF".to_string());
}
}
}
/// Where each of COLUMNS sits in the file's header, or what is wrong with it.
fn columns_at(names: &[String]) -> Result<Vec<usize>, String> {
let mut at: Vec<(&str, usize)> = Vec::new();
for (i, name) in names.iter().enumerate() {
if !COLUMNS.contains(&name.as_str()) {
return Err(format!("{}: unknown column \"{}\"", header_rule(), name));
}
if at.iter().any(|(n, _)| n == name) {
return Err(format!("{}: column \"{}\" appears twice", header_rule(), name));
}
at.push((name.as_str(), i));
}
let mut out = Vec::new();
for column in COLUMNS.iter() {
match at.iter().find(|(n, _)| n == column) {
Some((_, i)) => out.push(*i),
None => return Err(format!("{}: missing column \"{}\"", header_rule(), column)),
}
}
Ok(out)
}
fn whole_number(text: &str) -> Option<i64> {
let digits = text.strip_prefix('-').unwrap_or(text);
if digits.is_empty() || digits.len() > 15 || !digits.bytes().all(|c| c.is_ascii_digit()) {
return None;
}
Some(text.parse::<i64>().unwrap())
}
fn opt(text: &str) -> Option<String> {
if text.is_empty() { None } else { Some(text.to_string()) }
}
fn problem(row: i64, field: Option<&str>, message: &str, thrown: String) -> TodoCsvProblem {
TodoCsvProblem {
error: CsvRowError { row, field: field.map(|f| f.to_string()), message: message.to_string() },
thrown,
}
}
/// Reads the whole file, collecting every problem; the first is the one
/// import_csv 1.0.0 panicked with. A header or a record that cannot be split
/// into fields ends the reading.
pub fn read_todo_csv(csv: &str) -> TodoCsvReading {
let text = csv.strip_prefix('\u{feff}').unwrap_or(csv);
let b = text.as_bytes();
let mut todos: Vec<Todo> = Vec::new();
let mut problems: Vec<TodoCsvProblem> = Vec::new();
let mut ids: HashSet<String> = HashSet::new();
let mut at: Option<Vec<usize>> = None;
let mut row: i64 = 0;
let mut rows: i64 = 0;
let mut pos = 0;
while pos < b.len() {
if b[pos] == b'\n' {
pos += 1;
continue;
}
if b[pos] == b'\r' && pos + 1 < b.len() && b[pos + 1] == b'\n' {
pos += 2;
continue;
}
row += 1;
let fields = match record_at(text, pos) {
Err(message) => {
problems.push(problem(row, None, &message, format!("row {}: {}", row, message)));
break;
}
Ok((fields, next)) => {
pos = next;
fields
}
};
let columns = match &at {
None => match columns_at(&fields) {
Err(message) => {
problems.push(problem(row, None, &message, message.clone()));
break;
}
Ok(columns) => {
at = Some(columns);
continue;
}
},
Some(columns) => columns.clone(),
};
rows += 1;
if fields.len() != COLUMNS.len() {
let message = format!("expected {} fields, found {}", COLUMNS.len(), fields.len());
problems.push(problem(row, None, &message, format!("row {}: {}", row, message)));
continue;
}
let before = problems.len();
let plain = |problems: &mut Vec<TodoCsvProblem>, field: &str, message: String| {
problems.push(problem(row, Some(field), &message, format!("row {}: {}", row, message)));
};
let get = |i: usize| fields[columns[i]].as_str();
let done = get(3);
let done_read = done == "true" || done == "false";
if !done_read {
plain(&mut problems, "done", format!("done must be true or false, found \"{}\"", done));
}
let (frequency, interval, anchor) = (get(7), get(8), get(9));
let filled = [frequency, interval, anchor].iter().filter(|t| !t.is_empty()).count();
let mut recurrence: Option<Recurrence> = None;
let mut recurrence_read = true;
if filled == 3 {
match whole_number(interval) {
None => {
plain(&mut problems, "recurrenceInterval", format!("recurrenceInterval must be a whole number, found \"{}\"", interval));
recurrence_read = false;
}
Some(every) => {
recurrence = Some(Recurrence { frequency: frequency.to_string(), interval: every, anchor: anchor.to_string() });
}
}
} else if filled != 0 {
plain(&mut problems, "recurrence", "fill in all three recurrence columns or leave them all empty".to_string());
recurrence_read = false;
}
let order = get(12);
let position = whole_number(order);
if position.is_none() {
plain(&mut problems, "order", format!("order must be a whole number, found \"{}\"", order));
}
let tags: Vec<String> = get(6).split(' ').map(|t| t.to_string()).collect();
let todo = Todo {
id: get(0).to_string(),
title: get(1).to_string(),
notes: opt(get(2)),
done: done == "true",
priority: get(4).to_string(),
due: opt(get(5)),
tags: normalise_tags(&tags),
recurrence,
created_at: get(10).to_string(),
completed_at: opt(get(11)),
order: position.unwrap_or(0),
};
// A column that could not be read has already been reported; the
// stand-in value must not raise a second, misleading message.
let errors = validate_todo(&todo).errors;
for (field, message) in errors.iter() {
if (field == "completedAt" && !done_read) || (field == "recurrence" && !recurrence_read) || (field == "order" && position.is_none()) {
continue;
}
problems.push(problem(row, Some(field.as_str()), message, format!("row {}: {}: {}", row, field, message)));
}
if !errors.iter().any(|(field, _)| field == "id") {
if ids.contains(&todo.id) {
plain(&mut problems, "id", format!("duplicate id \"{}\"", todo.id));
}
ids.insert(todo.id.clone());
}
if problems.len() == before {
todos.push(todo);
}
}
if at.is_none() && problems.is_empty() {
let message = "the CSV has no header row";
problems.push(problem(1, None, message, message.to_string()));
}
TodoCsvReading { todos, rows, problems }
}
/// Every problem in a todo CSV file, row by row: the row (header = 1), the
/// field or column, and the message. Valid exactly when import_csv accepts it.
pub fn validate_todo_csv(csv: &str) -> CsvValidation {
let reading = read_todo_csv(csv);
CsvValidation {
valid: reading.problems.is_empty(),
rows: reading.rows,
errors: reading.problems.into_iter().map(|p| p.error).collect(),
}
}
pub fn csv_validation_to_value(result: &CsvValidation) -> Value {
let errors = result
.errors
.iter()
.map(|e| {
Value::obj(vec![
("row", Value::Int(e.row)),
("field", e.field.as_deref().map(Value::str).unwrap_or(Value::Null)),
("message", Value::str(&e.message)),
])
})
.collect();
Value::obj(vec![("valid", Value::Bool(result.valid)), ("rows", Value::Int(result.rows)), ("errors", Value::Arr(errors))])
}
pub fn fune_vector(args: &[Value]) -> Value {
csv_validation_to_value(&validate_todo_csv(args[0].as_str()))
}