use std::collections::HashSet; use super::funejson::Value; use super::todo_item::{todos_to_value, validate_todo, Recurrence, Todo}; use super::todo_normalise_tags::normalise_tags; const COLUMNS: [&str; 13] = [ "id", "title", "notes", "done", "priority", "due", "tags", "recurrenceFrequency", "recurrenceInterval", "recurrenceAnchor", "createdAt", "completedAt", "order", ]; fn header_rule() -> String { format!("the header must have the columns {}", COLUMNS.join(", ")) } fn fail(row: usize, message: &str) -> ! { panic!("row {}: {}", row, message) } /// One RFC 4180 record from `pos`: its fields and where the next record /// starts. Only ASCII bytes are looked at, so every slice is valid UTF-8. fn read_record(text: &str, mut pos: usize, row: usize) -> (Vec, usize) { let b = text.as_bytes(); let size = b.len(); let mut fields: Vec = Vec::new(); loop { let value: String; if pos < size && b[pos] == b'"' { pos += 1; let mut out = String::new(); loop { let quote = match text[pos..].find('"') { Some(q) => pos + q, None => fail(row, "a quoted field is not closed"), }; out.push_str(&text[pos..quote]); if quote + 1 < size && b[quote + 1] == b'"' { out.push('"'); pos = quote + 2; } else { pos = quote + 1; break; } } if pos < size && !matches!(b[pos], b',' | b'\r' | b'\n') { fail(row, "text after the closing quote of a field"); } value = out; } else { let mut end = pos; while end < size && !matches!(b[end], b',' | b'\r' | b'\n') { if b[end] == b'"' { fail(row, "a field with a double quote in it must be quoted"); } end += 1; } value = text[pos..end].to_string(); pos = end; } fields.push(value); if pos >= size { return (fields, pos); } if b[pos] == b',' { pos += 1; } else if b[pos] == b'\n' { return (fields, pos + 1); } else if pos + 1 < size && b[pos + 1] == b'\n' { return (fields, pos + 2); } else { fail(row, "a CR outside quotes must be followed by LF"); } } } /// Where each of COLUMNS sits in the file's header. fn read_header(names: &[String]) -> Vec { let mut at: Vec<(&str, usize)> = Vec::new(); for (i, name) in names.iter().enumerate() { if !COLUMNS.contains(&name.as_str()) { panic!("{}: unknown column \"{}\"", header_rule(), name); } if at.iter().any(|(n, _)| n == name) { panic!("{}: column \"{}\" appears twice", header_rule(), name); } at.push((name.as_str(), i)); } COLUMNS .iter() .map(|column| match at.iter().find(|(n, _)| n == column) { Some((_, i)) => *i, None => panic!("{}: missing column \"{}\"", header_rule(), column), }) .collect() } fn whole(row: usize, column: &str, text: &str) -> i64 { let digits = text.strip_prefix('-').unwrap_or(text); if digits.is_empty() || digits.len() > 15 || !digits.bytes().all(|c| c.is_ascii_digit()) { fail(row, &format!("{} must be a whole number, found \"{}\"", column, text)); } text.parse::().unwrap() } fn opt(text: &str) -> Option { if text.is_empty() { None } else { Some(text.to_string()) } } fn read_todo(fields: &[String], at: &[usize], row: usize) -> Todo { if fields.len() != COLUMNS.len() { fail(row, &format!("expected {} fields, found {}", COLUMNS.len(), fields.len())); } let get = |i: usize| fields[at[i]].as_str(); let done = get(3); if done != "true" && done != "false" { fail(row, &format!("done must be true or false, found \"{}\"", done)); } let (frequency, interval, anchor) = (get(7), get(8), get(9)); let filled = [frequency, interval, anchor].iter().filter(|t| !t.is_empty()).count(); let recurrence = match filled { 0 => None, 3 => Some(Recurrence { frequency: frequency.to_string(), interval: whole(row, "recurrenceInterval", interval), anchor: anchor.to_string(), }), _ => fail(row, "fill in all three recurrence columns or leave them all empty"), }; let tags: Vec = get(6).split(' ').map(|t| t.to_string()).collect(); Todo { id: get(0).to_string(), title: get(1).to_string(), notes: opt(get(2)), done: done == "true", priority: get(4).to_string(), due: opt(get(5)), tags: normalise_tags(&tags), recurrence, created_at: get(10).to_string(), completed_at: opt(get(11)), order: whole(row, "order", get(12)), } } /// Todos from CSV: the 13 columns in any order, CRLF or LF, an optional /// byte order mark. Every row is checked with validate_todo. /// /// # Panics /// With "row N: ..." (the header is row 1) on the first malformed or invalid /// row, or on a header without exactly the 13 columns. pub fn import_csv(csv: &str) -> Vec { let text = csv.strip_prefix('\u{feff}').unwrap_or(csv); let b = text.as_bytes(); let mut todos: Vec = Vec::new(); let mut ids: HashSet = HashSet::new(); let mut at: Option> = None; let mut row = 0; let mut pos = 0; while pos < b.len() { if b[pos] == b'\n' { pos += 1; continue; } if b[pos] == b'\r' && pos + 1 < b.len() && b[pos + 1] == b'\n' { pos += 2; continue; } row += 1; let (fields, next) = read_record(text, pos, row); pos = next; let columns = match &at { None => { at = Some(read_header(&fields)); continue; } Some(columns) => columns, }; let todo = read_todo(&fields, columns, row); if let Some((field, message)) = validate_todo(&todo).errors.first() { fail(row, &format!("{}: {}", field, message)); } if !ids.insert(todo.id.clone()) { fail(row, &format!("duplicate id \"{}\"", todo.id)); } todos.push(todo); } if at.is_none() { panic!("the CSV has no header row"); } todos } pub fn fune_vector(args: &[Value]) -> Value { todos_to_value(&import_csv(args[0].as_str())) }