Files
coursebank/src/data/store.rs
T
alexm 8ec48fb185
Pipeline / check (pull_request) Failing after 3m25s
Pipeline / docs (pull_request) Skipped
Pipeline / nightly (pull_request) Skipped
Pipeline / release (pull_request) Skipped
feat: add learning objective targets
2026-09-21 14:02:34 -04:00

645 lines
19 KiB
Rust

// SPDX-License-Identifier: Prosperity-3.0.0
// Copyright Scientific Computing Studio
// Source: https://git.scient.ing/education/coursebank
//! Where response data lives on disk.
//!
//! One file per administration, under `data/`, named after the administration id.
//! Not one big file, because an exam's responses are written once and then only
//! read: separate files mean re-ingesting Exam 4 cannot corrupt Exam 3, and a
//! term's data can be archived or excluded by moving files rather than filtering
//! rows.
//!
//! Parquet is the default format. It is columnar, typed, compressed, and readable
//! by pandas, polars, R, and DuckDB without an export step, which matters because
//! the point of storing this data is to still be able to analyze it in five years
//! with whatever tool exists then.
//!
//! CSV is the fallback, and it is a real fallback rather than a degraded mode:
//! `--no-default-features` builds the entire tool with CSV storage and loses
//! nothing but file size and read speed. Committing to a format you cannot open
//! without the right library version is how course data gets lost.
use std::path::{Path, PathBuf};
use crate::error::{Error, Result};
use crate::responses::{FlatResponse, Response, ResponseSet};
/// A storage format.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Format {
/// Apache Parquet, the default.
Parquet,
/// Comma-separated values.
Csv,
}
impl Format {
/// The file extension, without a dot.
pub fn extension(self) -> &'static str {
match self {
Format::Parquet => "parquet",
Format::Csv => "csv",
}
}
/// The format compiled in as the default.
///
/// # Returns
///
/// Parquet when the `parquet` feature is on, CSV otherwise.
pub fn preferred() -> Format {
Format::Parquet
}
/// Whether this format can be written by the current build.
pub fn is_available(self) -> bool {
match self {
Format::Csv => true,
Format::Parquet => true,
}
}
/// The format implied by a file extension.
///
/// # Arguments
///
/// * `path` - the file path.
///
/// # Returns
///
/// The format, or `None` when the extension is not one we write.
pub fn from_path(path: &Path) -> Option<Format> {
match path.extension().and_then(|e| e.to_str()) {
Some("parquet") => Some(Format::Parquet),
Some("csv") => Some(Format::Csv),
_ => None,
}
}
}
/// The response store rooted at a course's `data/` directory.
#[derive(Debug, Clone)]
pub struct Store {
/// The data directory.
pub dir: PathBuf,
/// The format to write.
pub format: Format,
}
impl Store {
/// Opens a store, creating the directory if needed.
///
/// # Arguments
///
/// * `dir` - the data directory.
///
/// # Returns
///
/// The store, writing the preferred available format.
///
/// # Errors
///
/// Returns [`Error::Io`] when the directory cannot be created.
pub fn open(dir: impl Into<PathBuf>) -> Result<Store> {
let dir = dir.into();
std::fs::create_dir_all(&dir).map_err(|e| Error::io(&dir, e))?;
Ok(Store {
dir,
format: Format::preferred(),
})
}
/// Sets the format to write.
///
/// # Arguments
///
/// * `format` - the format.
///
/// # Returns
///
/// The store, for chaining.
///
/// # Errors
///
/// Returns [`Error::FeatureDisabled`] when the format was compiled out.
pub fn with_format(mut self, format: Format) -> Result<Store> {
if !format.is_available() {
return Err(Error::FeatureDisabled("Parquet", "parquet"));
}
self.format = format;
Ok(self)
}
/// The path for an administration's responses.
///
/// # Arguments
///
/// * `administration_id` - the administration id.
///
/// # Returns
///
/// The path, with slashes in the id replaced so it is one file.
pub fn path_for(&self, administration_id: &str) -> PathBuf {
self.dir.join(format!(
"{}.{}",
sanitize(administration_id),
self.format.extension()
))
}
/// Writes one administration's responses, replacing any existing file.
///
/// Replacing rather than appending is deliberate. Re-ingesting after a
/// regrade should produce the corrected data, not two contradictory copies of
/// the same student's response, and there is no way to tell those apart later.
///
/// # Arguments
///
/// * `set` - the responses, which must all share one administration id.
///
/// # Returns
///
/// The paths written, one per administration found in the set.
///
/// # Errors
///
/// Returns [`Error::Io`] on a write failure and [`Error::FeatureDisabled`]
/// when writing Parquet without the feature.
pub fn write(&self, set: &ResponseSet) -> Result<Vec<PathBuf>> {
let mut written = Vec::new();
for admin in set.administrations() {
let rows: Vec<FlatResponse> = set
.rows
.iter()
.filter(|r| r.administration_id == admin)
.map(FlatResponse::from_response)
.collect();
let path = self.path_for(&admin);
match self.format {
Format::Csv => write_csv(&path, &rows)?,
Format::Parquet => write_parquet(&path, &rows)?,
}
written.push(path);
}
Ok(written)
}
/// Reads one administration's responses.
///
/// # Arguments
///
/// * `administration_id` - the administration id.
///
/// # Returns
///
/// The responses.
///
/// # Errors
///
/// Returns [`Error::Io`] when the file is missing.
pub fn read(&self, administration_id: &str) -> Result<ResponseSet> {
// Accept either format regardless of what this build prefers, so a repo
// written by a Parquet build is still readable by a lean one where the
// CSV happens to exist, and vice versa.
let stem = sanitize(administration_id);
for format in [self.format, Format::Parquet, Format::Csv] {
let path = self.dir.join(format!("{stem}.{}", format.extension()));
if path.exists() {
return read_path(&path);
}
}
Err(Error::Other(format!(
"no stored responses for `{administration_id}` in {}",
self.dir.display()
)))
}
/// Every stored data file, sorted.
///
/// # Returns
///
/// The paths.
///
/// # Errors
///
/// Returns [`Error::Io`] when the directory cannot be listed.
pub fn files(&self) -> Result<Vec<PathBuf>> {
if !self.dir.exists() {
return Ok(Vec::new());
}
let mut out: Vec<PathBuf> = std::fs::read_dir(&self.dir)
.map_err(|e| Error::io(&self.dir, e))?
.filter_map(|e| e.ok())
.map(|e| e.path())
.filter(|p| Format::from_path(p).is_some())
.collect();
out.sort();
Ok(out)
}
/// Reads every stored administration.
///
/// This is what pooled item statistics run on: several administrations of the
/// same item, which is the only way the numbers become trustworthy for a class
/// of twenty-five.
///
/// # Returns
///
/// All responses, with one file's failure recorded as a warning rather than
/// aborting the rest.
///
/// # Errors
///
/// Returns [`Error::Io`] when the directory cannot be listed.
pub fn read_all(&self) -> Result<ResponseSet> {
let mut set = ResponseSet::new();
for path in self.files()? {
match read_path(&path) {
Ok(part) => set.absorb(part),
Err(e) => set
.warnings
.push(format!("skipping {}: {e}", path.display())),
}
}
Ok(set)
}
/// Reads every administration of one assessment, across terms.
///
/// # Arguments
///
/// * `assessment_id` - the assessment id to match.
///
/// # Returns
///
/// The matching responses.
///
/// # Errors
///
/// Returns [`Error::Io`] when the directory cannot be listed.
pub fn read_assessment(&self, assessment_id: &str) -> Result<ResponseSet> {
let all = self.read_all()?;
let mut set = ResponseSet::new();
set.warnings = all.warnings;
set.rows = all
.rows
.into_iter()
.filter(|r| r.assessment_id == assessment_id)
.collect();
Ok(set)
}
}
/// Reads a data file, choosing the reader by extension.
///
/// # Arguments
///
/// * `path` - the file.
///
/// # Returns
///
/// The responses.
///
/// # Errors
///
/// Returns [`Error::Other`] for an unrecognized extension and
/// [`Error::FeatureDisabled`] for Parquet without the feature.
pub fn read_path(path: &Path) -> Result<ResponseSet> {
match Format::from_path(path) {
Some(Format::Csv) => read_csv(path),
Some(Format::Parquet) => read_parquet(path),
None => Err(Error::Other(format!(
"{} is not a response file; expected a .parquet or .csv",
path.display()
))),
}
}
/// Writes flat responses as CSV.
///
/// # Arguments
///
/// * `path` - the destination.
/// * `rows` - the rows.
///
/// # Errors
///
/// Returns [`Error::Csv`] on a serialization failure.
fn write_csv(path: &Path, rows: &[FlatResponse]) -> Result<()> {
if let Some(parent) = path.parent() {
std::fs::create_dir_all(parent).map_err(|e| Error::io(parent, e))?;
}
let mut w = csv::Writer::from_path(path).map_err(|e| Error::Csv {
path: path.to_path_buf(),
source: e,
})?;
for r in rows {
w.serialize(r).map_err(|e| Error::Csv {
path: path.to_path_buf(),
source: e,
})?;
}
w.flush().map_err(|e| Error::io(path, e))?;
Ok(())
}
/// Reads flat responses from CSV.
///
/// # Arguments
///
/// * `path` - the file.
///
/// # Returns
///
/// The responses.
///
/// # Errors
///
/// Returns [`Error::Csv`] on a parse failure.
fn read_csv(path: &Path) -> Result<ResponseSet> {
let mut r = csv::Reader::from_path(path).map_err(|e| Error::Csv {
path: path.to_path_buf(),
source: e,
})?;
let mut set = ResponseSet::new();
for rec in r.deserialize::<FlatResponse>() {
let flat = rec.map_err(|e| Error::Csv {
path: path.to_path_buf(),
source: e,
})?;
set.rows.push(flat.to_response());
}
Ok(set)
}
/// Writes flat responses as Parquet.
///
/// # Arguments
///
/// * `path` - the destination.
/// * `rows` - the rows.
fn write_parquet(path: &Path, rows: &[FlatResponse]) -> Result<()> {
crate::store_parquet::write(path, rows)
}
/// Reads flat responses from Parquet.
///
/// # Arguments
///
/// * `path` - the file.
///
/// # Returns
///
/// The responses.
///
/// # Errors
///
/// Returns [`Error::FeatureDisabled`] when the feature is off.
fn read_parquet(path: &Path) -> Result<ResponseSet> {
let rows = crate::store_parquet::read(path)?;
let mut set = ResponseSet::new();
set.rows = rows.iter().map(|f| f.to_response()).collect();
Ok(set)
}
/// Makes an administration id usable as a file name.
///
/// # Arguments
///
/// * `s` - the id.
///
/// # Returns
///
/// The sanitized stem.
pub fn sanitize(s: &str) -> String {
let mut out = String::with_capacity(s.len());
let mut last_sep = false;
for ch in s.chars() {
if ch.is_ascii_alphanumeric() || ch == '-' || ch == '.' {
out.push(ch.to_ascii_lowercase());
last_sep = false;
} else if !last_sep {
out.push('_');
last_sep = true;
}
}
let trimmed = out.trim_matches('_').to_string();
if trimmed.is_empty() {
"responses".to_string()
} else {
trimmed
}
}
/// Exports responses to an arbitrary path, in the format its extension implies.
///
/// This exists so `export` is a separate verb from `ingest`: the store is the
/// system of record, and handing a colleague a CSV should not change it.
///
/// # Arguments
///
/// * `path` - the destination.
/// * `set` - the responses.
///
/// # Errors
///
/// Returns [`Error::Other`] for an unrecognized extension.
pub fn export(path: &Path, set: &ResponseSet) -> Result<()> {
let rows: Vec<FlatResponse> = set.rows.iter().map(FlatResponse::from_response).collect();
match Format::from_path(path) {
Some(Format::Csv) => write_csv(path, &rows),
Some(Format::Parquet) => write_parquet(path, &rows),
None => Err(Error::Other(format!(
"cannot tell what format {} should be; use a .csv or .parquet extension",
path.display()
))),
}
}
/// Summarizes what is in the store, for `coursebank data list`.
#[derive(Debug, Clone)]
pub struct StoredSummary {
/// The administration id.
pub administration_id: String,
/// The file.
pub path: PathBuf,
/// How many response rows.
pub rows: usize,
/// How many distinct students.
pub students: usize,
/// How many distinct items.
pub items: usize,
}
/// Summarizes every stored administration.
///
/// # Arguments
///
/// * `store` - the store.
///
/// # Returns
///
/// One summary per file, sorted by path.
///
/// # Errors
///
/// Returns [`Error::Io`] when the directory cannot be listed.
pub fn summarize(store: &Store) -> Result<Vec<StoredSummary>> {
let mut out = Vec::new();
for path in store.files()? {
let set = match read_path(&path) {
Ok(s) => s,
Err(_) => continue,
};
let admin = set.administrations().first().cloned().unwrap_or_else(|| {
path.file_stem()
.unwrap_or_default()
.to_string_lossy()
.into()
});
out.push(StoredSummary {
administration_id: admin,
path,
rows: set.rows.len(),
students: set.students().len(),
items: set.all_items().len(),
});
}
Ok(out)
}
/// Groups responses by administration.
///
/// # Arguments
///
/// * `set` - the responses.
///
/// # Returns
///
/// One set per administration, keyed by id.
pub fn split_by_administration(
set: &ResponseSet,
) -> std::collections::BTreeMap<String, Vec<&Response>> {
let mut out: std::collections::BTreeMap<String, Vec<&Response>> = Default::default();
for r in &set.rows {
out.entry(r.administration_id.clone()).or_default().push(r);
}
out
}
#[cfg(test)]
mod tests {
use super::*;
use crate::responses::Response;
fn row(student: &str, number: u32) -> Response {
Response {
administration_id: "BIOSC1540/2026s/exam-4".into(),
course: "BIOSC1540".into(),
term: "2026s".into(),
assessment_id: "exam-4".into(),
date: None,
form: None,
form_position: None,
student_key: student.into(),
sid: None,
name: None,
email: None,
section: None,
item_number: number,
item_ref: Some("bank::q-x-001".into()),
item_version: Some(2),
selected: vec!["C".into()],
selected_source: vec![],
eliminated: vec![],
eliminated_source: vec![],
correct: Some(true),
credit: 1.0,
points_possible: 1.5,
score: 1.5,
response_time_seconds: None,
level: None,
learning_targets: vec!["lo-a".into()],
topics: vec![],
bonus: false,
dropped: false,
dropped_full_credit: false,
}
}
#[test]
fn sanitizes_administration_ids() {
assert_eq!(sanitize("BIOSC1540/2026s/exam-4"), "biosc1540_2026s_exam-4");
assert_eq!(sanitize("///"), "responses");
// Runs of separators collapse rather than stacking underscores.
assert_eq!(sanitize("a // b"), "a_b");
}
#[test]
fn csv_round_trips_through_the_store() {
let dir = std::env::temp_dir().join(format!("cb-store-{}", std::process::id()));
std::fs::remove_dir_all(&dir).ok();
let store = Store::open(&dir).unwrap().with_format(Format::Csv).unwrap();
let mut set = ResponseSet::new();
set.rows.push(row("s1", 1));
set.rows.push(row("s2", 1));
let written = store.write(&set).unwrap();
assert_eq!(written.len(), 1);
assert!(written[0].exists());
let back = store.read("BIOSC1540/2026s/exam-4").unwrap();
assert_eq!(back.rows.len(), 2);
assert_eq!(back.rows[0].item_ref.as_deref(), Some("bank::q-x-001"));
assert_eq!(back.rows[0].selected, vec!["C".to_string()]);
assert_eq!(back.rows[0].learning_targets, vec!["lo-a".to_string()]);
let all = store.read_all().unwrap();
assert_eq!(all.rows.len(), 2);
let summaries = summarize(&store).unwrap();
assert_eq!(summaries.len(), 1);
assert_eq!(summaries[0].students, 2);
assert_eq!(summaries[0].items, 1);
std::fs::remove_dir_all(&dir).ok();
}
#[test]
fn rewriting_replaces_rather_than_duplicates() {
let dir = std::env::temp_dir().join(format!("cb-store-rw-{}", std::process::id()));
std::fs::remove_dir_all(&dir).ok();
let store = Store::open(&dir).unwrap().with_format(Format::Csv).unwrap();
let mut set = ResponseSet::new();
set.rows.push(row("s1", 1));
store.write(&set).unwrap();
store.write(&set).unwrap();
assert_eq!(store.read_all().unwrap().rows.len(), 1, "no duplicates");
std::fs::remove_dir_all(&dir).ok();
}
#[test]
fn missing_administrations_report_clearly() {
let dir = std::env::temp_dir().join(format!("cb-store-miss-{}", std::process::id()));
std::fs::remove_dir_all(&dir).ok();
let store = Store::open(&dir).unwrap().with_format(Format::Csv).unwrap();
let err = store.read("nope").unwrap_err();
assert!(err.to_string().contains("no stored responses"));
std::fs::remove_dir_all(&dir).ok();
}
#[test]
fn format_extensions_round_trip() {
assert_eq!(
Format::from_path(Path::new("a/b.parquet")),
Some(Format::Parquet)
);
assert_eq!(Format::from_path(Path::new("a/b.csv")), Some(Format::Csv));
assert_eq!(Format::from_path(Path::new("a/b.yaml")), None);
assert!(Format::Csv.is_available());
}
}