645 lines
19 KiB
Rust
645 lines
19 KiB
Rust
// SPDX-License-Identifier: Prosperity-3.0.0
|
|
// Copyright Scientific Computing Studio
|
|
// Source: https://git.scient.ing/education/coursebank
|
|
|
|
//! Where response data lives on disk.
|
|
//!
|
|
//! One file per administration, under `data/`, named after the administration id.
|
|
//! Not one big file, because an exam's responses are written once and then only
|
|
//! read: separate files mean re-ingesting Exam 4 cannot corrupt Exam 3, and a
|
|
//! term's data can be archived or excluded by moving files rather than filtering
|
|
//! rows.
|
|
//!
|
|
//! Parquet is the default format. It is columnar, typed, compressed, and readable
|
|
//! by pandas, polars, R, and DuckDB without an export step, which matters because
|
|
//! the point of storing this data is to still be able to analyze it in five years
|
|
//! with whatever tool exists then.
|
|
//!
|
|
//! CSV is the fallback, and it is a real fallback rather than a degraded mode:
|
|
//! `--no-default-features` builds the entire tool with CSV storage and loses
|
|
//! nothing but file size and read speed. Committing to a format you cannot open
|
|
//! without the right library version is how course data gets lost.
|
|
|
|
use std::path::{Path, PathBuf};
|
|
|
|
use crate::error::{Error, Result};
|
|
use crate::responses::{FlatResponse, Response, ResponseSet};
|
|
|
|
/// A storage format.
|
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
|
pub enum Format {
|
|
/// Apache Parquet, the default.
|
|
Parquet,
|
|
/// Comma-separated values.
|
|
Csv,
|
|
}
|
|
|
|
impl Format {
|
|
/// The file extension, without a dot.
|
|
pub fn extension(self) -> &'static str {
|
|
match self {
|
|
Format::Parquet => "parquet",
|
|
Format::Csv => "csv",
|
|
}
|
|
}
|
|
|
|
/// The format compiled in as the default.
|
|
///
|
|
/// # Returns
|
|
///
|
|
/// Parquet when the `parquet` feature is on, CSV otherwise.
|
|
pub fn preferred() -> Format {
|
|
Format::Parquet
|
|
}
|
|
|
|
/// Whether this format can be written by the current build.
|
|
pub fn is_available(self) -> bool {
|
|
match self {
|
|
Format::Csv => true,
|
|
Format::Parquet => true,
|
|
}
|
|
}
|
|
|
|
/// The format implied by a file extension.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `path` - the file path.
|
|
///
|
|
/// # Returns
|
|
///
|
|
/// The format, or `None` when the extension is not one we write.
|
|
pub fn from_path(path: &Path) -> Option<Format> {
|
|
match path.extension().and_then(|e| e.to_str()) {
|
|
Some("parquet") => Some(Format::Parquet),
|
|
Some("csv") => Some(Format::Csv),
|
|
_ => None,
|
|
}
|
|
}
|
|
}
|
|
|
|
/// The response store rooted at a course's `data/` directory.
|
|
#[derive(Debug, Clone)]
|
|
pub struct Store {
|
|
/// The data directory.
|
|
pub dir: PathBuf,
|
|
/// The format to write.
|
|
pub format: Format,
|
|
}
|
|
|
|
impl Store {
|
|
/// Opens a store, creating the directory if needed.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `dir` - the data directory.
|
|
///
|
|
/// # Returns
|
|
///
|
|
/// The store, writing the preferred available format.
|
|
///
|
|
/// # Errors
|
|
///
|
|
/// Returns [`Error::Io`] when the directory cannot be created.
|
|
pub fn open(dir: impl Into<PathBuf>) -> Result<Store> {
|
|
let dir = dir.into();
|
|
std::fs::create_dir_all(&dir).map_err(|e| Error::io(&dir, e))?;
|
|
Ok(Store {
|
|
dir,
|
|
format: Format::preferred(),
|
|
})
|
|
}
|
|
|
|
/// Sets the format to write.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `format` - the format.
|
|
///
|
|
/// # Returns
|
|
///
|
|
/// The store, for chaining.
|
|
///
|
|
/// # Errors
|
|
///
|
|
/// Returns [`Error::FeatureDisabled`] when the format was compiled out.
|
|
pub fn with_format(mut self, format: Format) -> Result<Store> {
|
|
if !format.is_available() {
|
|
return Err(Error::FeatureDisabled("Parquet", "parquet"));
|
|
}
|
|
self.format = format;
|
|
Ok(self)
|
|
}
|
|
|
|
/// The path for an administration's responses.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `administration_id` - the administration id.
|
|
///
|
|
/// # Returns
|
|
///
|
|
/// The path, with slashes in the id replaced so it is one file.
|
|
pub fn path_for(&self, administration_id: &str) -> PathBuf {
|
|
self.dir.join(format!(
|
|
"{}.{}",
|
|
sanitize(administration_id),
|
|
self.format.extension()
|
|
))
|
|
}
|
|
|
|
/// Writes one administration's responses, replacing any existing file.
|
|
///
|
|
/// Replacing rather than appending is deliberate. Re-ingesting after a
|
|
/// regrade should produce the corrected data, not two contradictory copies of
|
|
/// the same student's response, and there is no way to tell those apart later.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `set` - the responses, which must all share one administration id.
|
|
///
|
|
/// # Returns
|
|
///
|
|
/// The paths written, one per administration found in the set.
|
|
///
|
|
/// # Errors
|
|
///
|
|
/// Returns [`Error::Io`] on a write failure and [`Error::FeatureDisabled`]
|
|
/// when writing Parquet without the feature.
|
|
pub fn write(&self, set: &ResponseSet) -> Result<Vec<PathBuf>> {
|
|
let mut written = Vec::new();
|
|
for admin in set.administrations() {
|
|
let rows: Vec<FlatResponse> = set
|
|
.rows
|
|
.iter()
|
|
.filter(|r| r.administration_id == admin)
|
|
.map(FlatResponse::from_response)
|
|
.collect();
|
|
let path = self.path_for(&admin);
|
|
match self.format {
|
|
Format::Csv => write_csv(&path, &rows)?,
|
|
Format::Parquet => write_parquet(&path, &rows)?,
|
|
}
|
|
written.push(path);
|
|
}
|
|
Ok(written)
|
|
}
|
|
|
|
/// Reads one administration's responses.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `administration_id` - the administration id.
|
|
///
|
|
/// # Returns
|
|
///
|
|
/// The responses.
|
|
///
|
|
/// # Errors
|
|
///
|
|
/// Returns [`Error::Io`] when the file is missing.
|
|
pub fn read(&self, administration_id: &str) -> Result<ResponseSet> {
|
|
// Accept either format regardless of what this build prefers, so a repo
|
|
// written by a Parquet build is still readable by a lean one where the
|
|
// CSV happens to exist, and vice versa.
|
|
let stem = sanitize(administration_id);
|
|
for format in [self.format, Format::Parquet, Format::Csv] {
|
|
let path = self.dir.join(format!("{stem}.{}", format.extension()));
|
|
if path.exists() {
|
|
return read_path(&path);
|
|
}
|
|
}
|
|
Err(Error::Other(format!(
|
|
"no stored responses for `{administration_id}` in {}",
|
|
self.dir.display()
|
|
)))
|
|
}
|
|
|
|
/// Every stored data file, sorted.
|
|
///
|
|
/// # Returns
|
|
///
|
|
/// The paths.
|
|
///
|
|
/// # Errors
|
|
///
|
|
/// Returns [`Error::Io`] when the directory cannot be listed.
|
|
pub fn files(&self) -> Result<Vec<PathBuf>> {
|
|
if !self.dir.exists() {
|
|
return Ok(Vec::new());
|
|
}
|
|
let mut out: Vec<PathBuf> = std::fs::read_dir(&self.dir)
|
|
.map_err(|e| Error::io(&self.dir, e))?
|
|
.filter_map(|e| e.ok())
|
|
.map(|e| e.path())
|
|
.filter(|p| Format::from_path(p).is_some())
|
|
.collect();
|
|
out.sort();
|
|
Ok(out)
|
|
}
|
|
|
|
/// Reads every stored administration.
|
|
///
|
|
/// This is what pooled item statistics run on: several administrations of the
|
|
/// same item, which is the only way the numbers become trustworthy for a class
|
|
/// of twenty-five.
|
|
///
|
|
/// # Returns
|
|
///
|
|
/// All responses, with one file's failure recorded as a warning rather than
|
|
/// aborting the rest.
|
|
///
|
|
/// # Errors
|
|
///
|
|
/// Returns [`Error::Io`] when the directory cannot be listed.
|
|
pub fn read_all(&self) -> Result<ResponseSet> {
|
|
let mut set = ResponseSet::new();
|
|
for path in self.files()? {
|
|
match read_path(&path) {
|
|
Ok(part) => set.absorb(part),
|
|
Err(e) => set
|
|
.warnings
|
|
.push(format!("skipping {}: {e}", path.display())),
|
|
}
|
|
}
|
|
Ok(set)
|
|
}
|
|
|
|
/// Reads every administration of one assessment, across terms.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `assessment_id` - the assessment id to match.
|
|
///
|
|
/// # Returns
|
|
///
|
|
/// The matching responses.
|
|
///
|
|
/// # Errors
|
|
///
|
|
/// Returns [`Error::Io`] when the directory cannot be listed.
|
|
pub fn read_assessment(&self, assessment_id: &str) -> Result<ResponseSet> {
|
|
let all = self.read_all()?;
|
|
let mut set = ResponseSet::new();
|
|
set.warnings = all.warnings;
|
|
set.rows = all
|
|
.rows
|
|
.into_iter()
|
|
.filter(|r| r.assessment_id == assessment_id)
|
|
.collect();
|
|
Ok(set)
|
|
}
|
|
}
|
|
|
|
/// Reads a data file, choosing the reader by extension.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `path` - the file.
|
|
///
|
|
/// # Returns
|
|
///
|
|
/// The responses.
|
|
///
|
|
/// # Errors
|
|
///
|
|
/// Returns [`Error::Other`] for an unrecognized extension and
|
|
/// [`Error::FeatureDisabled`] for Parquet without the feature.
|
|
pub fn read_path(path: &Path) -> Result<ResponseSet> {
|
|
match Format::from_path(path) {
|
|
Some(Format::Csv) => read_csv(path),
|
|
Some(Format::Parquet) => read_parquet(path),
|
|
None => Err(Error::Other(format!(
|
|
"{} is not a response file; expected a .parquet or .csv",
|
|
path.display()
|
|
))),
|
|
}
|
|
}
|
|
|
|
/// Writes flat responses as CSV.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `path` - the destination.
|
|
/// * `rows` - the rows.
|
|
///
|
|
/// # Errors
|
|
///
|
|
/// Returns [`Error::Csv`] on a serialization failure.
|
|
fn write_csv(path: &Path, rows: &[FlatResponse]) -> Result<()> {
|
|
if let Some(parent) = path.parent() {
|
|
std::fs::create_dir_all(parent).map_err(|e| Error::io(parent, e))?;
|
|
}
|
|
let mut w = csv::Writer::from_path(path).map_err(|e| Error::Csv {
|
|
path: path.to_path_buf(),
|
|
source: e,
|
|
})?;
|
|
for r in rows {
|
|
w.serialize(r).map_err(|e| Error::Csv {
|
|
path: path.to_path_buf(),
|
|
source: e,
|
|
})?;
|
|
}
|
|
w.flush().map_err(|e| Error::io(path, e))?;
|
|
Ok(())
|
|
}
|
|
|
|
/// Reads flat responses from CSV.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `path` - the file.
|
|
///
|
|
/// # Returns
|
|
///
|
|
/// The responses.
|
|
///
|
|
/// # Errors
|
|
///
|
|
/// Returns [`Error::Csv`] on a parse failure.
|
|
fn read_csv(path: &Path) -> Result<ResponseSet> {
|
|
let mut r = csv::Reader::from_path(path).map_err(|e| Error::Csv {
|
|
path: path.to_path_buf(),
|
|
source: e,
|
|
})?;
|
|
let mut set = ResponseSet::new();
|
|
for rec in r.deserialize::<FlatResponse>() {
|
|
let flat = rec.map_err(|e| Error::Csv {
|
|
path: path.to_path_buf(),
|
|
source: e,
|
|
})?;
|
|
set.rows.push(flat.to_response());
|
|
}
|
|
Ok(set)
|
|
}
|
|
|
|
/// Writes flat responses as Parquet.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `path` - the destination.
|
|
/// * `rows` - the rows.
|
|
fn write_parquet(path: &Path, rows: &[FlatResponse]) -> Result<()> {
|
|
crate::store_parquet::write(path, rows)
|
|
}
|
|
|
|
/// Reads flat responses from Parquet.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `path` - the file.
|
|
///
|
|
/// # Returns
|
|
///
|
|
/// The responses.
|
|
///
|
|
/// # Errors
|
|
///
|
|
/// Returns [`Error::FeatureDisabled`] when the feature is off.
|
|
fn read_parquet(path: &Path) -> Result<ResponseSet> {
|
|
let rows = crate::store_parquet::read(path)?;
|
|
let mut set = ResponseSet::new();
|
|
set.rows = rows.iter().map(|f| f.to_response()).collect();
|
|
Ok(set)
|
|
}
|
|
|
|
/// Makes an administration id usable as a file name.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `s` - the id.
|
|
///
|
|
/// # Returns
|
|
///
|
|
/// The sanitized stem.
|
|
pub fn sanitize(s: &str) -> String {
|
|
let mut out = String::with_capacity(s.len());
|
|
let mut last_sep = false;
|
|
for ch in s.chars() {
|
|
if ch.is_ascii_alphanumeric() || ch == '-' || ch == '.' {
|
|
out.push(ch.to_ascii_lowercase());
|
|
last_sep = false;
|
|
} else if !last_sep {
|
|
out.push('_');
|
|
last_sep = true;
|
|
}
|
|
}
|
|
let trimmed = out.trim_matches('_').to_string();
|
|
if trimmed.is_empty() {
|
|
"responses".to_string()
|
|
} else {
|
|
trimmed
|
|
}
|
|
}
|
|
|
|
/// Exports responses to an arbitrary path, in the format its extension implies.
|
|
///
|
|
/// This exists so `export` is a separate verb from `ingest`: the store is the
|
|
/// system of record, and handing a colleague a CSV should not change it.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `path` - the destination.
|
|
/// * `set` - the responses.
|
|
///
|
|
/// # Errors
|
|
///
|
|
/// Returns [`Error::Other`] for an unrecognized extension.
|
|
pub fn export(path: &Path, set: &ResponseSet) -> Result<()> {
|
|
let rows: Vec<FlatResponse> = set.rows.iter().map(FlatResponse::from_response).collect();
|
|
match Format::from_path(path) {
|
|
Some(Format::Csv) => write_csv(path, &rows),
|
|
Some(Format::Parquet) => write_parquet(path, &rows),
|
|
None => Err(Error::Other(format!(
|
|
"cannot tell what format {} should be; use a .csv or .parquet extension",
|
|
path.display()
|
|
))),
|
|
}
|
|
}
|
|
|
|
/// Summarizes what is in the store, for `coursebank data list`.
|
|
#[derive(Debug, Clone)]
|
|
pub struct StoredSummary {
|
|
/// The administration id.
|
|
pub administration_id: String,
|
|
/// The file.
|
|
pub path: PathBuf,
|
|
/// How many response rows.
|
|
pub rows: usize,
|
|
/// How many distinct students.
|
|
pub students: usize,
|
|
/// How many distinct items.
|
|
pub items: usize,
|
|
}
|
|
|
|
/// Summarizes every stored administration.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `store` - the store.
|
|
///
|
|
/// # Returns
|
|
///
|
|
/// One summary per file, sorted by path.
|
|
///
|
|
/// # Errors
|
|
///
|
|
/// Returns [`Error::Io`] when the directory cannot be listed.
|
|
pub fn summarize(store: &Store) -> Result<Vec<StoredSummary>> {
|
|
let mut out = Vec::new();
|
|
for path in store.files()? {
|
|
let set = match read_path(&path) {
|
|
Ok(s) => s,
|
|
Err(_) => continue,
|
|
};
|
|
let admin = set.administrations().first().cloned().unwrap_or_else(|| {
|
|
path.file_stem()
|
|
.unwrap_or_default()
|
|
.to_string_lossy()
|
|
.into()
|
|
});
|
|
out.push(StoredSummary {
|
|
administration_id: admin,
|
|
path,
|
|
rows: set.rows.len(),
|
|
students: set.students().len(),
|
|
items: set.all_items().len(),
|
|
});
|
|
}
|
|
Ok(out)
|
|
}
|
|
|
|
/// Groups responses by administration.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `set` - the responses.
|
|
///
|
|
/// # Returns
|
|
///
|
|
/// One set per administration, keyed by id.
|
|
pub fn split_by_administration(
|
|
set: &ResponseSet,
|
|
) -> std::collections::BTreeMap<String, Vec<&Response>> {
|
|
let mut out: std::collections::BTreeMap<String, Vec<&Response>> = Default::default();
|
|
for r in &set.rows {
|
|
out.entry(r.administration_id.clone()).or_default().push(r);
|
|
}
|
|
out
|
|
}
|
|
|
|
#[cfg(test)]
|
|
mod tests {
|
|
use super::*;
|
|
use crate::responses::Response;
|
|
|
|
fn row(student: &str, number: u32) -> Response {
|
|
Response {
|
|
administration_id: "BIOSC1540/2026s/exam-4".into(),
|
|
course: "BIOSC1540".into(),
|
|
term: "2026s".into(),
|
|
assessment_id: "exam-4".into(),
|
|
date: None,
|
|
form: None,
|
|
form_position: None,
|
|
student_key: student.into(),
|
|
sid: None,
|
|
name: None,
|
|
email: None,
|
|
section: None,
|
|
item_number: number,
|
|
item_ref: Some("bank::q-x-001".into()),
|
|
item_version: Some(2),
|
|
selected: vec!["C".into()],
|
|
selected_source: vec![],
|
|
eliminated: vec![],
|
|
eliminated_source: vec![],
|
|
correct: Some(true),
|
|
credit: 1.0,
|
|
points_possible: 1.5,
|
|
score: 1.5,
|
|
response_time_seconds: None,
|
|
level: None,
|
|
learning_targets: vec!["lo-a".into()],
|
|
topics: vec![],
|
|
bonus: false,
|
|
dropped: false,
|
|
dropped_full_credit: false,
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn sanitizes_administration_ids() {
|
|
assert_eq!(sanitize("BIOSC1540/2026s/exam-4"), "biosc1540_2026s_exam-4");
|
|
assert_eq!(sanitize("///"), "responses");
|
|
// Runs of separators collapse rather than stacking underscores.
|
|
assert_eq!(sanitize("a // b"), "a_b");
|
|
}
|
|
|
|
#[test]
|
|
fn csv_round_trips_through_the_store() {
|
|
let dir = std::env::temp_dir().join(format!("cb-store-{}", std::process::id()));
|
|
std::fs::remove_dir_all(&dir).ok();
|
|
let store = Store::open(&dir).unwrap().with_format(Format::Csv).unwrap();
|
|
|
|
let mut set = ResponseSet::new();
|
|
set.rows.push(row("s1", 1));
|
|
set.rows.push(row("s2", 1));
|
|
let written = store.write(&set).unwrap();
|
|
assert_eq!(written.len(), 1);
|
|
assert!(written[0].exists());
|
|
|
|
let back = store.read("BIOSC1540/2026s/exam-4").unwrap();
|
|
assert_eq!(back.rows.len(), 2);
|
|
assert_eq!(back.rows[0].item_ref.as_deref(), Some("bank::q-x-001"));
|
|
assert_eq!(back.rows[0].selected, vec!["C".to_string()]);
|
|
assert_eq!(back.rows[0].learning_targets, vec!["lo-a".to_string()]);
|
|
|
|
let all = store.read_all().unwrap();
|
|
assert_eq!(all.rows.len(), 2);
|
|
|
|
let summaries = summarize(&store).unwrap();
|
|
assert_eq!(summaries.len(), 1);
|
|
assert_eq!(summaries[0].students, 2);
|
|
assert_eq!(summaries[0].items, 1);
|
|
|
|
std::fs::remove_dir_all(&dir).ok();
|
|
}
|
|
|
|
#[test]
|
|
fn rewriting_replaces_rather_than_duplicates() {
|
|
let dir = std::env::temp_dir().join(format!("cb-store-rw-{}", std::process::id()));
|
|
std::fs::remove_dir_all(&dir).ok();
|
|
let store = Store::open(&dir).unwrap().with_format(Format::Csv).unwrap();
|
|
|
|
let mut set = ResponseSet::new();
|
|
set.rows.push(row("s1", 1));
|
|
store.write(&set).unwrap();
|
|
store.write(&set).unwrap();
|
|
|
|
assert_eq!(store.read_all().unwrap().rows.len(), 1, "no duplicates");
|
|
std::fs::remove_dir_all(&dir).ok();
|
|
}
|
|
|
|
#[test]
|
|
fn missing_administrations_report_clearly() {
|
|
let dir = std::env::temp_dir().join(format!("cb-store-miss-{}", std::process::id()));
|
|
std::fs::remove_dir_all(&dir).ok();
|
|
let store = Store::open(&dir).unwrap().with_format(Format::Csv).unwrap();
|
|
let err = store.read("nope").unwrap_err();
|
|
assert!(err.to_string().contains("no stored responses"));
|
|
std::fs::remove_dir_all(&dir).ok();
|
|
}
|
|
|
|
#[test]
|
|
fn format_extensions_round_trip() {
|
|
assert_eq!(
|
|
Format::from_path(Path::new("a/b.parquet")),
|
|
Some(Format::Parquet)
|
|
);
|
|
assert_eq!(Format::from_path(Path::new("a/b.csv")), Some(Format::Csv));
|
|
assert_eq!(Format::from_path(Path::new("a/b.yaml")), None);
|
|
assert!(Format::Csv.is_available());
|
|
}
|
|
}
|