// SPDX-License-Identifier: Prosperity-3.0.0 // Copyright Scientific Computing Studio // Source: https://git.scient.ing/education/coursebank //! Where response data lives on disk. //! //! One file per administration, under `data/`, named after the administration id. //! Not one big file, because an exam's responses are written once and then only //! read: separate files mean re-ingesting Exam 4 cannot corrupt Exam 3, and a //! term's data can be archived or excluded by moving files rather than filtering //! rows. //! //! Parquet is the default format. It is columnar, typed, compressed, and readable //! by pandas, polars, R, and DuckDB without an export step, which matters because //! the point of storing this data is to still be able to analyze it in five years //! with whatever tool exists then. //! //! CSV is the fallback, and it is a real fallback rather than a degraded mode: //! `--no-default-features` builds the entire tool with CSV storage and loses //! nothing but file size and read speed. Committing to a format you cannot open //! without the right library version is how course data gets lost. use std::path::{Path, PathBuf}; use crate::error::{Error, Result}; use crate::responses::{FlatResponse, Response, ResponseSet}; /// A storage format. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum Format { /// Apache Parquet, the default. Parquet, /// Comma-separated values. Csv, } impl Format { /// The file extension, without a dot. pub fn extension(self) -> &'static str { match self { Format::Parquet => "parquet", Format::Csv => "csv", } } /// The format compiled in as the default. /// /// # Returns /// /// Parquet when the `parquet` feature is on, CSV otherwise. pub fn preferred() -> Format { Format::Parquet } /// Whether this format can be written by the current build. pub fn is_available(self) -> bool { match self { Format::Csv => true, Format::Parquet => true, } } /// The format implied by a file extension. /// /// # Arguments /// /// * `path` - the file path. /// /// # Returns /// /// The format, or `None` when the extension is not one we write. pub fn from_path(path: &Path) -> Option { match path.extension().and_then(|e| e.to_str()) { Some("parquet") => Some(Format::Parquet), Some("csv") => Some(Format::Csv), _ => None, } } } /// The response store rooted at a course's `data/` directory. #[derive(Debug, Clone)] pub struct Store { /// The data directory. pub dir: PathBuf, /// The format to write. pub format: Format, } impl Store { /// Opens a store, creating the directory if needed. /// /// # Arguments /// /// * `dir` - the data directory. /// /// # Returns /// /// The store, writing the preferred available format. /// /// # Errors /// /// Returns [`Error::Io`] when the directory cannot be created. pub fn open(dir: impl Into) -> Result { let dir = dir.into(); std::fs::create_dir_all(&dir).map_err(|e| Error::io(&dir, e))?; Ok(Store { dir, format: Format::preferred(), }) } /// Sets the format to write. /// /// # Arguments /// /// * `format` - the format. /// /// # Returns /// /// The store, for chaining. /// /// # Errors /// /// Returns [`Error::FeatureDisabled`] when the format was compiled out. pub fn with_format(mut self, format: Format) -> Result { if !format.is_available() { return Err(Error::FeatureDisabled("Parquet", "parquet")); } self.format = format; Ok(self) } /// The path for an administration's responses. /// /// # Arguments /// /// * `administration_id` - the administration id. /// /// # Returns /// /// The path, with slashes in the id replaced so it is one file. pub fn path_for(&self, administration_id: &str) -> PathBuf { self.dir.join(format!( "{}.{}", sanitize(administration_id), self.format.extension() )) } /// Writes one administration's responses, replacing any existing file. /// /// Replacing rather than appending is deliberate. Re-ingesting after a /// regrade should produce the corrected data, not two contradictory copies of /// the same student's response, and there is no way to tell those apart later. /// /// # Arguments /// /// * `set` - the responses, which must all share one administration id. /// /// # Returns /// /// The paths written, one per administration found in the set. /// /// # Errors /// /// Returns [`Error::Io`] on a write failure and [`Error::FeatureDisabled`] /// when writing Parquet without the feature. pub fn write(&self, set: &ResponseSet) -> Result> { let mut written = Vec::new(); for admin in set.administrations() { let rows: Vec = set .rows .iter() .filter(|r| r.administration_id == admin) .map(FlatResponse::from_response) .collect(); let path = self.path_for(&admin); match self.format { Format::Csv => write_csv(&path, &rows)?, Format::Parquet => write_parquet(&path, &rows)?, } written.push(path); } Ok(written) } /// Reads one administration's responses. /// /// # Arguments /// /// * `administration_id` - the administration id. /// /// # Returns /// /// The responses. /// /// # Errors /// /// Returns [`Error::Io`] when the file is missing. pub fn read(&self, administration_id: &str) -> Result { // Accept either format regardless of what this build prefers, so a repo // written by a Parquet build is still readable by a lean one where the // CSV happens to exist, and vice versa. let stem = sanitize(administration_id); for format in [self.format, Format::Parquet, Format::Csv] { let path = self.dir.join(format!("{stem}.{}", format.extension())); if path.exists() { return read_path(&path); } } Err(Error::Other(format!( "no stored responses for `{administration_id}` in {}", self.dir.display() ))) } /// Every stored data file, sorted. /// /// # Returns /// /// The paths. /// /// # Errors /// /// Returns [`Error::Io`] when the directory cannot be listed. pub fn files(&self) -> Result> { if !self.dir.exists() { return Ok(Vec::new()); } let mut out: Vec = std::fs::read_dir(&self.dir) .map_err(|e| Error::io(&self.dir, e))? .filter_map(|e| e.ok()) .map(|e| e.path()) .filter(|p| Format::from_path(p).is_some()) .collect(); out.sort(); Ok(out) } /// Reads every stored administration. /// /// This is what pooled item statistics run on: several administrations of the /// same item, which is the only way the numbers become trustworthy for a class /// of twenty-five. /// /// # Returns /// /// All responses, with one file's failure recorded as a warning rather than /// aborting the rest. /// /// # Errors /// /// Returns [`Error::Io`] when the directory cannot be listed. pub fn read_all(&self) -> Result { let mut set = ResponseSet::new(); for path in self.files()? { match read_path(&path) { Ok(part) => set.absorb(part), Err(e) => set .warnings .push(format!("skipping {}: {e}", path.display())), } } Ok(set) } /// Reads every administration of one assessment, across terms. /// /// # Arguments /// /// * `assessment_id` - the assessment id to match. /// /// # Returns /// /// The matching responses. /// /// # Errors /// /// Returns [`Error::Io`] when the directory cannot be listed. pub fn read_assessment(&self, assessment_id: &str) -> Result { let all = self.read_all()?; let mut set = ResponseSet::new(); set.warnings = all.warnings; set.rows = all .rows .into_iter() .filter(|r| r.assessment_id == assessment_id) .collect(); Ok(set) } } /// Reads a data file, choosing the reader by extension. /// /// # Arguments /// /// * `path` - the file. /// /// # Returns /// /// The responses. /// /// # Errors /// /// Returns [`Error::Other`] for an unrecognized extension and /// [`Error::FeatureDisabled`] for Parquet without the feature. pub fn read_path(path: &Path) -> Result { match Format::from_path(path) { Some(Format::Csv) => read_csv(path), Some(Format::Parquet) => read_parquet(path), None => Err(Error::Other(format!( "{} is not a response file; expected a .parquet or .csv", path.display() ))), } } /// Writes flat responses as CSV. /// /// # Arguments /// /// * `path` - the destination. /// * `rows` - the rows. /// /// # Errors /// /// Returns [`Error::Csv`] on a serialization failure. fn write_csv(path: &Path, rows: &[FlatResponse]) -> Result<()> { if let Some(parent) = path.parent() { std::fs::create_dir_all(parent).map_err(|e| Error::io(parent, e))?; } let mut w = csv::Writer::from_path(path).map_err(|e| Error::Csv { path: path.to_path_buf(), source: e, })?; for r in rows { w.serialize(r).map_err(|e| Error::Csv { path: path.to_path_buf(), source: e, })?; } w.flush().map_err(|e| Error::io(path, e))?; Ok(()) } /// Reads flat responses from CSV. /// /// # Arguments /// /// * `path` - the file. /// /// # Returns /// /// The responses. /// /// # Errors /// /// Returns [`Error::Csv`] on a parse failure. fn read_csv(path: &Path) -> Result { let mut r = csv::Reader::from_path(path).map_err(|e| Error::Csv { path: path.to_path_buf(), source: e, })?; let mut set = ResponseSet::new(); for rec in r.deserialize::() { let flat = rec.map_err(|e| Error::Csv { path: path.to_path_buf(), source: e, })?; set.rows.push(flat.to_response()); } Ok(set) } /// Writes flat responses as Parquet. /// /// # Arguments /// /// * `path` - the destination. /// * `rows` - the rows. fn write_parquet(path: &Path, rows: &[FlatResponse]) -> Result<()> { crate::store_parquet::write(path, rows) } /// Reads flat responses from Parquet. /// /// # Arguments /// /// * `path` - the file. /// /// # Returns /// /// The responses. /// /// # Errors /// /// Returns [`Error::FeatureDisabled`] when the feature is off. fn read_parquet(path: &Path) -> Result { let rows = crate::store_parquet::read(path)?; let mut set = ResponseSet::new(); set.rows = rows.iter().map(|f| f.to_response()).collect(); Ok(set) } /// Makes an administration id usable as a file name. /// /// # Arguments /// /// * `s` - the id. /// /// # Returns /// /// The sanitized stem. pub fn sanitize(s: &str) -> String { let mut out = String::with_capacity(s.len()); let mut last_sep = false; for ch in s.chars() { if ch.is_ascii_alphanumeric() || ch == '-' || ch == '.' { out.push(ch.to_ascii_lowercase()); last_sep = false; } else if !last_sep { out.push('_'); last_sep = true; } } let trimmed = out.trim_matches('_').to_string(); if trimmed.is_empty() { "responses".to_string() } else { trimmed } } /// Exports responses to an arbitrary path, in the format its extension implies. /// /// This exists so `export` is a separate verb from `ingest`: the store is the /// system of record, and handing a colleague a CSV should not change it. /// /// # Arguments /// /// * `path` - the destination. /// * `set` - the responses. /// /// # Errors /// /// Returns [`Error::Other`] for an unrecognized extension. pub fn export(path: &Path, set: &ResponseSet) -> Result<()> { let rows: Vec = set.rows.iter().map(FlatResponse::from_response).collect(); match Format::from_path(path) { Some(Format::Csv) => write_csv(path, &rows), Some(Format::Parquet) => write_parquet(path, &rows), None => Err(Error::Other(format!( "cannot tell what format {} should be; use a .csv or .parquet extension", path.display() ))), } } /// Summarizes what is in the store, for `coursebank data list`. #[derive(Debug, Clone)] pub struct StoredSummary { /// The administration id. pub administration_id: String, /// The file. pub path: PathBuf, /// How many response rows. pub rows: usize, /// How many distinct students. pub students: usize, /// How many distinct items. pub items: usize, } /// Summarizes every stored administration. /// /// # Arguments /// /// * `store` - the store. /// /// # Returns /// /// One summary per file, sorted by path. /// /// # Errors /// /// Returns [`Error::Io`] when the directory cannot be listed. pub fn summarize(store: &Store) -> Result> { let mut out = Vec::new(); for path in store.files()? { let set = match read_path(&path) { Ok(s) => s, Err(_) => continue, }; let admin = set.administrations().first().cloned().unwrap_or_else(|| { path.file_stem() .unwrap_or_default() .to_string_lossy() .into() }); out.push(StoredSummary { administration_id: admin, path, rows: set.rows.len(), students: set.students().len(), items: set.all_items().len(), }); } Ok(out) } /// Groups responses by administration. /// /// # Arguments /// /// * `set` - the responses. /// /// # Returns /// /// One set per administration, keyed by id. pub fn split_by_administration( set: &ResponseSet, ) -> std::collections::BTreeMap> { let mut out: std::collections::BTreeMap> = Default::default(); for r in &set.rows { out.entry(r.administration_id.clone()).or_default().push(r); } out } #[cfg(test)] mod tests { use super::*; use crate::responses::Response; fn row(student: &str, number: u32) -> Response { Response { administration_id: "BIOSC1540/2026s/exam-4".into(), course: "BIOSC1540".into(), term: "2026s".into(), assessment_id: "exam-4".into(), date: None, form: None, form_position: None, student_key: student.into(), sid: None, name: None, email: None, section: None, item_number: number, item_ref: Some("bank::q-x-001".into()), item_version: Some(2), selected: vec!["C".into()], selected_source: vec![], eliminated: vec![], eliminated_source: vec![], correct: Some(true), credit: 1.0, points_possible: 1.5, score: 1.5, response_time_seconds: None, level: None, learning_targets: vec!["lo-a".into()], topics: vec![], bonus: false, dropped: false, dropped_full_credit: false, } } #[test] fn sanitizes_administration_ids() { assert_eq!(sanitize("BIOSC1540/2026s/exam-4"), "biosc1540_2026s_exam-4"); assert_eq!(sanitize("///"), "responses"); // Runs of separators collapse rather than stacking underscores. assert_eq!(sanitize("a // b"), "a_b"); } #[test] fn csv_round_trips_through_the_store() { let dir = std::env::temp_dir().join(format!("cb-store-{}", std::process::id())); std::fs::remove_dir_all(&dir).ok(); let store = Store::open(&dir).unwrap().with_format(Format::Csv).unwrap(); let mut set = ResponseSet::new(); set.rows.push(row("s1", 1)); set.rows.push(row("s2", 1)); let written = store.write(&set).unwrap(); assert_eq!(written.len(), 1); assert!(written[0].exists()); let back = store.read("BIOSC1540/2026s/exam-4").unwrap(); assert_eq!(back.rows.len(), 2); assert_eq!(back.rows[0].item_ref.as_deref(), Some("bank::q-x-001")); assert_eq!(back.rows[0].selected, vec!["C".to_string()]); assert_eq!(back.rows[0].learning_targets, vec!["lo-a".to_string()]); let all = store.read_all().unwrap(); assert_eq!(all.rows.len(), 2); let summaries = summarize(&store).unwrap(); assert_eq!(summaries.len(), 1); assert_eq!(summaries[0].students, 2); assert_eq!(summaries[0].items, 1); std::fs::remove_dir_all(&dir).ok(); } #[test] fn rewriting_replaces_rather_than_duplicates() { let dir = std::env::temp_dir().join(format!("cb-store-rw-{}", std::process::id())); std::fs::remove_dir_all(&dir).ok(); let store = Store::open(&dir).unwrap().with_format(Format::Csv).unwrap(); let mut set = ResponseSet::new(); set.rows.push(row("s1", 1)); store.write(&set).unwrap(); store.write(&set).unwrap(); assert_eq!(store.read_all().unwrap().rows.len(), 1, "no duplicates"); std::fs::remove_dir_all(&dir).ok(); } #[test] fn missing_administrations_report_clearly() { let dir = std::env::temp_dir().join(format!("cb-store-miss-{}", std::process::id())); std::fs::remove_dir_all(&dir).ok(); let store = Store::open(&dir).unwrap().with_format(Format::Csv).unwrap(); let err = store.read("nope").unwrap_err(); assert!(err.to_string().contains("no stored responses")); std::fs::remove_dir_all(&dir).ok(); } #[test] fn format_extensions_round_trip() { assert_eq!( Format::from_path(Path::new("a/b.parquet")), Some(Format::Parquet) ); assert_eq!(Format::from_path(Path::new("a/b.csv")), Some(Format::Csv)); assert_eq!(Format::from_path(Path::new("a/b.yaml")), None); assert!(Format::Csv.is_available()); } }