3 Commits
Author SHA1 Message Date
alexm c6d6ee10b6 feat: improvement
Pipeline / check (pull_request) Successful in 5m0s
Pipeline / docs (pull_request) Skipped
Pipeline / nightly (pull_request) Skipped
Pipeline / release (pull_request) Skipped
2026-09-27 01:11:09 -04:00
alexm 5ac1e317c0 feat: cooked up something fierce
Pipeline / check (pull_request) Successful in 3m9s
Pipeline / docs (pull_request) Skipped
Pipeline / nightly (pull_request) Skipped
Pipeline / release (pull_request) Skipped
2026-09-26 18:22:25 -04:00
alexm eabc98ad31 feat: splitting 2026-09-26 01:15:12 -04:00
41 changed files with 8857 additions and 377 deletions
+384 -42
View File
@@ -28,18 +28,19 @@
use std::collections::{BTreeMap, BTreeSet};
use std::path::PathBuf;
use crate::SCHEMA_VERSION;
use crate::assessment::AssessmentFile;
use crate::bank::BankFile;
use crate::calibration::{CalibrationFile, Measurement, MeasurementFile, MeasurementMeta};
use crate::catalog::Catalog;
use crate::classical::{self, Analysis, ItemAnalysis, Thresholds};
use crate::date::Date;
use crate::error::{Error, Result};
use crate::irt::{self, Fit};
use crate::item::{Calibration, IrtParams, OptionStat};
use crate::item::{Calibration, IrtParams, Item, OptionStat, VariantCalibration};
use crate::layout::Layout;
use crate::responses::ResponseSet;
use crate::store::Store;
use crate::taxonomy::Flag;
use crate::yaml;
/// What calibration would change about one item.
#[derive(Debug, Clone)]
@@ -267,6 +268,35 @@ pub fn plan(catalog: &Catalog, store: &Store, opts: &Options) -> Result<Plan> {
));
}
// Splitting by option set means an item administered three times with three
// different sets has three cells of 24 rather than one of 72. That is the
// honest picture, and it is worth saying out loud rather than leaving
// someone to read an IRT fit that was never possible.
for (uid, appearances) in &by_item {
let mut sizes: Vec<usize> = Vec::new();
for (_, analysis) in appearances {
if analysis.variant.is_some() {
sizes.push(analysis.n);
}
}
if sizes.len() > 1 {
let distinct: BTreeSet<&str> = appearances
.iter()
.filter_map(|(_, a)| a.variant.as_deref())
.collect();
if distinct.len() > 1 {
warnings.push(format!(
"{uid}: {} option sets across {} administrations, largest n = {}. \
Statistics are kept per set, because a stem shown with different \
distractors is a different item.",
distinct.len(),
appearances.len(),
sizes.iter().copied().max().unwrap_or(0)
));
}
}
}
let mut changes = Vec::new();
for (uid, appearances) in &by_item {
@@ -304,6 +334,12 @@ pub fn plan(catalog: &Catalog, store: &Store, opts: &Options) -> Result<Plan> {
option_stats: pooled.option_stats.clone(),
irt: irt_params,
flags: pooled.flags.clone(),
variants: variant_records(&entry.item, appearances, previous_variants(entry)),
options: BTreeMap::new(),
};
let calibration = Calibration {
options: option_histories(&calibration.variants),
..calibration
};
let previous = entry.item.calibration.as_ref();
@@ -580,60 +616,123 @@ fn diff_calibration(previous: Option<&Calibration>, next: &Calibration) -> Vec<S
out
}
/// Applies a plan, rewriting the affected bank files.
/// Applies a plan, rewriting the calibration store.
///
/// Files are rewritten one at a time and each is re-read before editing, so a plan
/// built against a bank that has since changed on disk fails loudly rather than
/// clobbering the newer version.
/// One file, `analysis/calibration.yaml`, and never a bank. A bank is reviewed
/// for what it asks; its history should be a record of wording decisions, not
/// of every grading run. The store is re-read immediately before editing, so a
/// plan built against a store that has since changed on disk fails loudly
/// rather than clobbering the newer version.
///
/// # Arguments
///
/// * `layout` - the course layout, for where the store lives.
/// * `plan` - the plan to apply.
///
/// # Returns
///
/// The bank files rewritten.
/// The file written.
///
/// # Errors
///
/// Returns [`Error::Unresolved`] when an item in the plan is no longer in its bank,
/// and [`Error::Io`] on a write failure.
pub fn apply(plan: &Plan) -> Result<Vec<PathBuf>> {
// Group by file so each is read and written once.
let mut by_file: BTreeMap<&PathBuf, Vec<&Change>> = BTreeMap::new();
/// Returns [`Error::Io`] on a write failure and [`Error::Yaml`] if the existing
/// store does not parse.
pub fn apply(layout: &Layout, plan: &Plan) -> Result<PathBuf> {
let path = layout.calibration_file();
let mut store = CalibrationFile::load(&path)?;
for change in &plan.changes {
by_file.entry(&change.path).or_default().push(change);
store
.items
.insert(change.uid.clone(), change.calibration.clone());
}
let mut written = Vec::new();
for (path, changes) in by_file {
let mut bank: BankFile = yaml::read(path)?;
for change in changes {
// The uid is `bank::item`; match on the item part.
let item_id = change
.uid
.split_once("::")
.map(|(_, id)| id)
.unwrap_or(&change.uid);
let target = bank.items.iter_mut().find(|i| i.id == item_id);
match target {
Some(item) => item.calibration = Some(change.calibration.clone()),
None => {
return Err(Error::Unresolved {
kind: "item",
id: change.uid.clone(),
context: Some(format!(
"{} — the bank changed since the plan was built; re-run calibration",
path.display()
)),
});
}
}
}
yaml::write(path, &bank)?;
written.push(path.clone());
if let Some(parent) = path.parent() {
std::fs::create_dir_all(parent).map_err(|e| Error::io(parent, e))?;
}
Ok(written)
store.save(&path)?;
Ok(path)
}
/// Records what one administration measured, as a file that is never rewritten.
///
/// The audit trail under the pooled store: these are the numbers one exam
/// produced, on a day, under a named model. Keeping them means the history
/// survives losing `data/`, which is ignored by git precisely because every row
/// of it carries a student.
///
/// # Arguments
///
/// * `layout` - the course layout.
/// * `administration` - the administration id, which names the file.
/// * `analysis` - the classical analysis of that administration.
/// * `catalog` - the loaded course, for the digests each item was measured
/// against.
/// * `record` - the assessment record, for the term and date.
///
/// # Returns
///
/// The two files written: one row per question, one row per question and
/// option.
///
/// # Errors
///
/// Returns [`Error::Usage`] when a record for this administration already
/// exists, since an administration happened once.
pub fn record_measurements(
layout: &Layout,
administration: &str,
analysis: &Analysis,
catalog: &Catalog,
record: Option<&AssessmentFile>,
) -> Result<Vec<PathBuf>> {
let mut items = Vec::new();
for item in &analysis.items {
let Some(uid) = &item.item_ref else { continue };
let entry = catalog.get(uid);
items.push(Measurement {
item: uid.clone(),
number: item.number,
variant: item.variant.clone(),
stem_digest: entry.map(|e| e.item.stem_digest()),
n: item.n,
p_value: Some(round4(item.p_value)),
point_biserial: item.point_biserial.map(round4),
discrimination_index: item.discrimination_index.map(round4),
key: item.key.clone(),
option_stats: item
.options
.iter()
.map(|(id, o)| (id.clone(), o.to_option_stat()))
.collect(),
irt: None,
flags: item.flags.clone(),
});
}
let file = MeasurementFile {
schema_version: SCHEMA_VERSION.to_string(),
administration: MeasurementMeta {
id: administration.to_string(),
assessment: record
.map(|r| r.assessment.id.clone())
.unwrap_or_else(|| administration.to_string()),
term: record.and_then(|r| r.assessment.term.clone()),
date: record.and_then(|r| r.assessment.date),
forms: record
.map(|r| r.forms.iter().map(|f| f.id.clone()).collect())
.unwrap_or_default(),
// The cohort, taken as the largest per-item n: a student who
// skipped question 7 still sat the exam.
n_examinees: analysis.items.iter().map(|i| i.n).max().unwrap_or(0),
model: None,
generated: Some(Date::today()),
coursebank: Some(crate::VERSION.to_string()),
},
items,
};
file.write_csv(&layout.measurements())
}
/// Builds a plan for a single administration, from an in-memory analysis.
@@ -703,6 +802,16 @@ pub fn plan_from_analysis(
.collect(),
irt: irt_params,
flags: item.flags.clone(),
variants: variant_records(
&entry.item,
&[(admin.clone(), item.clone())],
previous_variants(entry),
),
options: BTreeMap::new(),
};
let calibration = Calibration {
options: option_histories(&calibration.variants),
..calibration
};
let previous = entry.item.calibration.as_ref();
@@ -730,6 +839,171 @@ pub fn plan_from_analysis(
}
}
/// The variant records an item already has, to be merged with the new ones.
fn previous_variants(entry: &crate::catalog::Entry) -> Vec<VariantCalibration> {
entry
.item
.calibration
.as_ref()
.map(|c| c.variants.clone())
.unwrap_or_default()
}
/// Builds one calibration record per option set the item was administered in.
///
/// The records this pass computes replace the stored ones for the same variant
/// and leave the rest alone. That is what makes a partial recalibration safe:
/// a pass given only this term's data must not silently discard the numbers for
/// an option set that was retired two terms ago.
///
/// Statistics come from the same [`pool`] used for the flat summary, so the two
/// agree for an item whose pool is its form — the case every pre-2.0 item is in.
///
/// # Arguments
///
/// * `item` - the bank item.
/// * `appearances` - the administration id and analysis of each appearance.
/// * `previous` - the records already stored.
///
/// # Returns
///
/// The merged records, in variant order.
fn variant_records(
item: &Item,
appearances: &[(String, ItemAnalysis)],
previous: Vec<VariantCalibration>,
) -> Vec<VariantCalibration> {
// Appearances whose administration mixed two option sets under one question
// number carry no variant, and there is no set for them to describe.
let mut by_variant: BTreeMap<String, Vec<(String, ItemAnalysis)>> = BTreeMap::new();
for (admin, analysis) in appearances {
if let Some(variant) = &analysis.variant {
by_variant
.entry(variant.clone())
.or_default()
.push((admin.clone(), analysis.clone()));
}
}
let mut merged: BTreeMap<String, VariantCalibration> = previous
.into_iter()
.map(|v| (v.variant.clone(), v))
.collect();
for (variant, group) in by_variant {
let pooled = pool(&group);
// The option set is recovered from the digest's own record when the
// stored one has it, and from the options that were actually chosen
// otherwise, so a record written from data alone still says what it
// describes.
let (key, distractors) = describe(item, &variant, &group, merged.get(&variant));
merged.insert(
variant.clone(),
VariantCalibration {
variant,
key,
distractors,
administrations: group.iter().map(|(a, _)| a.clone()).collect(),
n_examinees: Some(pooled.n),
p_value: Some(round4(pooled.p_value)),
point_biserial: pooled.point_biserial.map(round4),
discrimination_index: pooled.discrimination_index.map(round4),
option_stats: pooled.option_stats.clone(),
irt: None,
flags: pooled.flags.clone(),
},
);
}
merged.into_values().collect()
}
/// Which options a variant administered.
///
/// Prefers what a stored record already says. Failing that, the options that
/// appear in the statistics are the ones students saw, and the item says which
/// of those are keyed.
fn describe(
item: &Item,
variant: &str,
group: &[(String, ItemAnalysis)],
stored: Option<&VariantCalibration>,
) -> (Vec<String>, Vec<String>) {
if let Some(stored) = stored {
if !stored.key.is_empty()
&& item.variant_digest(&stored.key, &stored.distractors) == variant
{
return (stored.key.clone(), stored.distractors.clone());
}
}
let mut seen: BTreeSet<String> = BTreeSet::new();
for (_, analysis) in group {
seen.extend(analysis.options.keys().cloned());
}
let keyed: BTreeSet<String> = group
.iter()
.flat_map(|(_, a)| a.key.iter().cloned())
.collect();
let key: Vec<String> = seen
.iter()
.filter(|o| keyed.contains(*o))
.cloned()
.collect();
let distractors: Vec<String> = seen
.iter()
.filter(|o| !keyed.contains(*o))
.cloned()
.collect();
(key, distractors)
}
/// Summarizes what each option has done across every set it appeared in.
///
/// Deliberately coarse, because selection rates are shares of a fixed set and
/// averaging them across different sets is not a statistic. The one claim this
/// view supports is the one worth having: an option that draws nobody in any
/// set it has appeared in is not doing anything, and that is the evidence for
/// retiring it — evidence a single administration cannot provide.
///
/// # Arguments
///
/// * `variants` - the per-variant records.
///
/// # Returns
///
/// One history per option id.
fn option_histories(
variants: &[VariantCalibration],
) -> BTreeMap<String, crate::item::OptionHistory> {
let mut out: BTreeMap<String, crate::item::OptionHistory> = BTreeMap::new();
let mut rates: BTreeMap<String, Vec<f64>> = BTreeMap::new();
for variant in variants {
let n = variant.n_examinees.unwrap_or(0);
for (option, stat) in &variant.option_stats {
let entry = out.entry(option.clone()).or_default();
entry.appearances += 1;
entry.n_examinees += n;
if let Some(rate) = stat.selection_rate {
rates.entry(option.clone()).or_default().push(rate);
}
}
}
for (option, entry) in &mut out {
if let Some(seen) = rates.get(option) {
if !seen.is_empty() {
entry.mean_selection_rate =
Some(round4(seen.iter().sum::<f64>() / seen.len() as f64));
entry.never_chosen = seen.iter().all(|r| *r <= f64::EPSILON);
}
}
}
out
}
/// Rounds to four decimals.
fn round4(x: f64) -> f64 {
(x * 1e4).round() / 1e4
@@ -754,9 +1028,77 @@ mod tests {
option_stats: BTreeMap::new(),
irt: None,
flags: Vec::new(),
variants: Vec::new(),
options: BTreeMap::new(),
}
}
fn stat(rate: f64) -> OptionStat {
OptionStat {
selection_rate: Some(rate),
point_biserial: None,
upper_group_rate: None,
lower_group_rate: None,
}
}
fn variant(id: &str, n: usize, dead_rate: f64) -> VariantCalibration {
VariantCalibration {
variant: id.into(),
n_examinees: Some(n),
option_stats: [
("o-key".to_string(), stat(0.8)),
("o-dead".to_string(), stat(dead_rate)),
]
.into_iter()
.collect(),
..VariantCalibration::default()
}
}
#[test]
fn an_option_that_draws_nobody_is_only_visible_across_sets() {
let histories = option_histories(&[variant("v1", 50, 0.0), variant("v2", 46, 0.0)]);
let dead = &histories["o-dead"];
assert_eq!(dead.appearances, 2);
assert_eq!(dead.n_examinees, 96);
// The claim the cross-variant view exists to support, and the one a
// single administration cannot make.
assert!(dead.never_chosen);
assert!(!histories["o-key"].never_chosen);
assert_eq!(histories["o-key"].mean_selection_rate, Some(0.8));
// One set where it drew is enough to stop the claim.
let mixed = option_histories(&[variant("v1", 50, 0.0), variant("v2", 46, 0.04)]);
assert!(!mixed["o-dead"].never_chosen);
}
#[test]
fn a_pass_with_no_data_for_an_item_keeps_the_records_it_has() {
let item: Item = serde_yaml_ng::from_str(
r#"id: q-x
status: approved
level: 1
stem: s
options:
- { id: o-key, text: right, correct: true }
- { id: o-one, text: wrong }
"#,
)
.expect("item parses");
let kept = variant_records(&item, &[], vec![variant("v-old", 96, 0.0)]);
assert_eq!(
kept.len(),
1,
"a pass given nothing must not delete history"
);
assert_eq!(kept[0].variant, "v-old");
assert_eq!(kept[0].n_examinees, Some(96));
}
#[test]
fn a_first_calibration_is_all_new() {
let next = calibration(0.7, Some(0.3), 24, "abc");
+29
View File
@@ -134,6 +134,14 @@ pub struct ItemAnalysis {
pub number: u32,
/// The item's global id, when known.
pub item_ref: Option<String>,
/// The option set administered, when the rows agree on one.
///
/// Within one administration an item has one variant, because a placement's
/// distractors are drawn once and shared by every form — only the printed
/// order differs. `None` means the rows disagreed, which happens when a
/// course opts into drawing distractors per form; the statistics below then
/// describe a mixture and cannot be pooled by option set.
pub variant: Option<String>,
/// How many students the item was administered to.
pub n: usize,
/// How many gave a non-blank response.
@@ -553,6 +561,7 @@ pub fn analyze(
item_ref: record
.and_then(|r| r.placement(*number))
.map(|p| p.item.clone()),
variant: one_variant(set, *number),
n,
n_answered,
blank_rate: blank as f64 / responded as f64,
@@ -906,6 +915,25 @@ fn reliability(coded: &[Vec<f64>], totals: &[f64], p_values: &[f64], rpbs: &[f64
/// # Returns
///
/// The letters that appear on full-credit responses.
/// The single variant every row for one question names, if they agree.
///
/// Disagreement is not an error, it is a fact about the administration: a course
/// that draws distractors per form has two option sets under one question
/// number, and no pooled statistic describes both. Returning `None` is what
/// keeps the per-variant records from claiming otherwise.
fn one_variant(set: &ResponseSet, number: u32) -> Option<String> {
let mut seen: Option<&str> = None;
for row in set.rows.iter().filter(|r| r.item_number == number) {
let variant = row.variant.as_deref()?;
match seen {
None => seen = Some(variant),
Some(first) if first == variant => {}
Some(_) => return None,
}
}
seen.map(str::to_string)
}
fn infer_key(rows: &[&crate::responses::Response]) -> Vec<String> {
let mut out: BTreeSet<String> = BTreeSet::new();
for r in rows {
@@ -1001,6 +1029,7 @@ mod tests {
item_number: number,
item_ref: None,
item_version: None,
variant: None,
selected: if letter.is_empty() {
vec![]
} else {
+4 -18
View File
@@ -71,7 +71,7 @@ use serde::Serialize;
use crate::assessment::AssessmentFile;
use crate::catalog::Catalog;
use crate::classical::Analysis;
use crate::course::{CourseFile, ReadingRole, Reference};
use crate::course::{CourseFile, ReadingRole};
use crate::irt::Fit;
use crate::item::Citation;
use crate::responses::{Response, ResponseSet};
@@ -797,9 +797,9 @@ fn item_readings(course: &CourseFile, citations: &[Citation]) -> Vec<ItemReading
let (label, title, url) = match reference {
Some((key, reference)) => (
reference.label.as_deref().unwrap_or(key).to_string(),
reference.label_or(key).to_string(),
Some(reference.title.clone()),
resolve_citation_url(citation, reference),
citation.href(reference),
),
None => (citation.display(), None, citation.url.clone()),
};
@@ -821,20 +821,6 @@ fn item_readings(course: &CourseFile, citations: &[Citation]) -> Vec<ItemReading
out
}
/// A citation's own URL, else the reference's `base_url` joined with its `path`.
fn resolve_citation_url(citation: &Citation, reference: &Reference) -> Option<String> {
if let Some(url) = &citation.url {
return Some(url.clone());
}
let path = citation.path.as_deref()?;
let base = reference.base_url.as_deref()?;
Some(match (base.ends_with('/'), path.starts_with('/')) {
(true, true) => format!("{base}{}", &path[1..]),
(false, false) => format!("{base}/{path}"),
_ => format!("{base}{path}"),
})
}
/// Ranks the lectures behind a student's missed questions.
///
/// A lecture earns its place by how many distinct objectives went wrong in it,
@@ -1694,7 +1680,7 @@ pub fn cohort(
.map(|option| {
let count = responses
.iter()
.filter(|r| r.chosen().iter().any(|l| *l == option.id))
.filter(|r| r.chosen().contains(&option.id))
.count();
OptionRow {
text: Some(option.text.clone()),
+1
View File
@@ -1363,6 +1363,7 @@ learning_objectives:
item_number: number,
item_ref: None,
item_version: None,
variant: None,
selected: vec!["A".into()],
selected_source: vec![],
eliminated: vec![],
+357 -20
View File
@@ -31,6 +31,16 @@ const BASE: &str = "https://coursebank.dev/schema";
pub enum Kind {
/// `course.yaml`.
Course,
/// `references.yaml`.
References,
/// `lectures/*.yaml`.
Lecture,
/// `objectives/*.yaml`.
Objective,
/// `analysis/calibration.yaml`.
Calibration,
/// `analysis/administrations/*.yaml`.
Measurements,
/// `banks/*.yaml`.
Bank,
/// `assessments/*.yaml`.
@@ -38,13 +48,27 @@ pub enum Kind {
}
impl Kind {
/// All three kinds.
pub const ALL: [Kind; 3] = [Kind::Course, Kind::Bank, Kind::Assessment];
/// Every kind.
pub const ALL: [Kind; 8] = [
Kind::Course,
Kind::References,
Kind::Lecture,
Kind::Objective,
Kind::Bank,
Kind::Assessment,
Kind::Calibration,
Kind::Measurements,
];
/// The file name a schema is written to.
pub fn filename(self) -> &'static str {
match self {
Kind::Course => "course.schema.json",
Kind::References => "references.schema.json",
Kind::Lecture => "lecture.schema.json",
Kind::Objective => "objective.schema.json",
Kind::Calibration => "calibration.schema.json",
Kind::Measurements => "measurements.schema.json",
Kind::Bank => "bank.schema.json",
Kind::Assessment => "assessment.schema.json",
}
@@ -80,12 +104,17 @@ impl Kind {
pub fn schema(kind: Kind) -> Value {
match kind {
Kind::Course => course_schema(),
Kind::References => references_schema(),
Kind::Lecture => lecture_fragment_schema(),
Kind::Objective => objective_fragment_schema(),
Kind::Calibration => calibration_file_schema(),
Kind::Measurements => measurements_file_schema(),
Kind::Bank => bank_schema(),
Kind::Assessment => assessment_schema(),
}
}
/// Writes all three schemas to a directory.
/// Writes every schema to a directory.
///
/// # Arguments
///
@@ -328,6 +357,13 @@ fn lecture_schema() -> Value {
"date": date("Date delivered."),
"unit": { "type": "string", "description": "Unit id." },
"slides_url": { "type": "string" },
"teaches": {
"type": "array",
"description": "The objectives this session develops. Each named objective gains \
this lecture in its `lectures` list when the course is loaded, \
so the pair is declared once, here, while planning the lecture.",
"items": { "type": "string" }
},
"readings": {
"type": "array",
"description": "Readings assigned with this lecture, in the order you assign \
@@ -355,9 +391,12 @@ fn target_schema() -> Value {
"order": {
"type": "integer",
"minimum": 1,
"description": "Position among the other targets of the same objective, low \
first. Ordered within its objective rather than across the \
course, so inserting one renumbers nothing outside its group."
"description": "Ignored since 2.0. An objective's targets are a set of question \
templates, not steps in a sequence — they are not taught in \
order and an exam samples from them — so a position asserts an \
order that does not exist. Where one target depends on another, \
say so with `prerequisites`. `coursebank migrate order` removes \
this."
},
"level_ceiling": level(),
"prerequisites": string_array(
@@ -386,8 +425,10 @@ fn objective_schema() -> Value {
"order": {
"type": "integer",
"minimum": 1,
"description": "Position in teaching order, low first. Without it objectives \
sort by id, which puts one before its own prerequisite."
"description": "Position in teaching order, low first. Derived since 2.0 from \
the position of this objective in a lecture's `teaches` list; \
an authored value still wins, and `coursebank migrate order` \
removes them."
},
"level_ceiling": level(),
"prerequisites": string_array(
@@ -434,7 +475,23 @@ fn reference_schema() -> Value {
"volume": { "type": "string" },
"issue": { "type": "string" },
"pages": { "type": "string", "description": "Pages of the work, not of a reading." },
"doi": { "type": "string", "description": "Bare DOI: 10.1038/nature12373." },
"doi": {
"type": "string",
"pattern": "^(doi:|https?://(dx\\.)?doi\\.org/)?10\\.",
"description": "Bare DOI: 10.1038/nature12373. For a manuscript this is \
usually the only link worth storing, since a reading list \
resolves it to doi.org."
},
"arxiv": { "type": "string", "description": "Bare arXiv id: 2301.00001." },
"pmcid": {
"type": "string",
"description": "PubMed Central id, which hosts the full text: PMC3084216."
},
"pmid": {
"type": "string",
"pattern": "^[0-9]+$",
"description": "PubMed id, which hosts a record about the work: 21471563."
},
"isbn": { "type": "string" },
"url": { "type": "string", "description": "Canonical URL for the whole work." },
"base_url": {
@@ -568,6 +625,94 @@ fn course_schema() -> Value {
})
}
/// The schema for `references.yaml`.
fn references_schema() -> Value {
json!({
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": format!("{BASE}/references.schema.json"),
"title": "coursebank references file",
"description": "The works the course cites, by citation key. One fragment of the \
course file; see course.schema.json for the whole.",
"type": "object",
"additionalProperties": false,
"properties": {
"schema_version": {
"type": ["string", "number"],
"description": format!("Format version; currently {SCHEMA_VERSION}. Declared in \
course.yaml; fragments inherit it.")
},
"references": {
"type": "object",
"description": "Works by citation key.",
"additionalProperties": reference_schema()
}
}
})
}
/// The schema for one file under `lectures/`.
fn lecture_fragment_schema() -> Value {
json!({
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": format!("{BASE}/lecture.schema.json"),
"title": "coursebank lecture file",
"description": "One session: its readings, and the objectives it develops. A fragment \
of the course file, merged on load.",
"type": "object",
"required": ["lectures"],
"additionalProperties": false,
"properties": {
"schema_version": {
"type": ["string", "number"],
"description": "Declared in course.yaml; fragments inherit it."
},
"lectures": {
"type": "object",
"description": "Keyed by lecture id, conventionally one entry per file.",
"additionalProperties": lecture_schema()
}
}
})
}
/// The schema for one file under `objectives/`.
fn objective_fragment_schema() -> Value {
json!({
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": format!("{BASE}/objective.schema.json"),
"title": "coursebank objective file",
"description": "One learning objective and the learning targets it decomposes into. A \
fragment of the course file, merged on load.",
"type": "object",
"required": ["learning_objectives"],
"additionalProperties": false,
"properties": {
"schema_version": {
"type": ["string", "number"],
"description": "Declared in course.yaml; fragments inherit it."
},
"learning_objectives": {
"type": "object",
"description": "Keyed by objective id, conventionally one entry per file. Its \
`lectures` list is derived from each lecture's `teaches`, so \
leave it out unless you prefer to declare it here.",
"additionalProperties": objective_schema()
},
"learning_targets": {
"type": "object",
"description": "The targets of this file's objective, by id. A target with no \
`lectures` of its own inherits its objective's.",
"additionalProperties": target_schema()
},
"stimuli": {
"type": "object",
"description": "Shared passages, figures, or data that several items refer to.",
"additionalProperties": stimulus_schema()
}
}
})
}
/// One option's schema.
///
/// Split out from [`item_schema`] rather than inlined, because `serde_json`'s
@@ -583,9 +728,11 @@ fn option_schema() -> Value {
"properties": {
"id": {
"type": "string",
"pattern": "^[A-H]$",
"description": "Option letter. Identity, not print position — shuffled forms \
relabel on the way out."
"pattern": "^(o-[a-z0-9]+(-[a-z0-9]+)*|[A-H])$",
"description": "Option id, unique within the item: `o-fourth-line`. An \
identity, not a print position — shuffled forms relabel on the \
way out. A single letter A-H is the pre-2.0 form; \
`coursebank migrate options` renames it."
},
"text": text("The option as a student reads it."),
"correct": { "type": "boolean" },
@@ -617,7 +764,8 @@ fn option_schema() -> Value {
},
"selection_rate_expected": proportion(
"How often you expect this to be chosen. Compared against reality."
)
),
"retired": retirement_schema()
}
})
}
@@ -812,6 +960,21 @@ fn calibration_schema() -> Value {
"additionalProperties": option_stat_schema()
},
"irt": irt_schema(),
"variants": {
"type": "array",
"description": "One record per option set ever administered. A stem shown with \
different distractors is a different item, so a p-value pooled \
across both would average two questions.",
"items": variant_calibration_schema()
},
"options": {
"type": "object",
"description": "One record per option, pooled across every set it appeared in. \
Supports one claim — this option draws nobody, anywhere — which \
is what retires a distractor and what one administration cannot \
show.",
"additionalProperties": option_history_schema()
},
"flags": {
"type": "array",
"items": { "type": "string", "enum": strings(&flags) }
@@ -853,6 +1016,163 @@ fn history_schema() -> Value {
})
}
/// The schema for one option set's statistics.
fn variant_calibration_schema() -> Value {
json!({
"type": "object",
"required": ["variant"],
"additionalProperties": false,
"properties": {
"variant": {
"type": "string",
"description": "Digest of the stem, the administered options, and which was keyed."
},
"key": string_array("The option ids keyed correct in this set."),
"distractors": string_array("The option ids offered alongside them."),
"administrations": string_array("The administrations pooled into these numbers."),
"n_examinees": { "type": "integer", "minimum": 0 },
"p_value": proportion("Proportion correct, for this option set only."),
"point_biserial": { "type": "number", "minimum": -1.0, "maximum": 1.0 },
"discrimination_index": { "type": "number", "minimum": -1.0, "maximum": 1.0 },
"option_stats": {
"type": "object",
"description": "Per-option behaviour within this set, by option id.",
"additionalProperties": option_stat_schema()
},
"irt": irt_schema(),
"flags": {
"type": "array",
"items": {
"type": "string",
"enum": strings(
&Flag::ALL.iter().map(|f| f.as_str()).collect::<Vec<&str>>()
)
}
}
}
})
}
/// The schema for one option's cross-variant history.
fn option_history_schema() -> Value {
json!({
"type": "object",
"additionalProperties": false,
"properties": {
"appearances": { "type": "integer", "minimum": 0 },
"n_examinees": { "type": "integer", "minimum": 0 },
"mean_selection_rate": proportion(
"Mean of the within-variant rates. For reading, not for acting on: each rate is \
a share of a different set."
),
"never_chosen": {
"type": "boolean",
"description": "Never chosen, anywhere. The claim that justifies retiring it."
}
}
})
}
/// The schema for `analysis/calibration.yaml`.
fn calibration_file_schema() -> Value {
json!({
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": format!("{BASE}/calibration.schema.json"),
"title": "coursebank calibration store",
"description": "The pooled statistics, by item id. Kept out of the banks: a bank's diff \
should be a change of intent, not the output of a grading run. Committed \
— every number here is a cohort aggregate, and there is no field for a \
student.",
"type": "object",
"additionalProperties": false,
"properties": {
"schema_version": { "type": ["string", "number"] },
"items": {
"type": "object",
"description": "Keyed by item id, which since 2.0 names the item course-wide and \
carries no file name, so moving a question between banks does \
not orphan its statistics.",
"additionalProperties": calibration_schema()
}
}
})
}
/// The schema for one file under `analysis/administrations/`.
fn measurements_file_schema() -> Value {
json!({
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": format!("{BASE}/measurements.schema.json"),
"title": "coursebank administration record",
"description": "What one administration measured. Written once and not rewritten, like \
a seal: it records a thing that happened on a day. Cohort aggregates \
only — no per-section or per-student breakdown, because a small cell \
crossed with anything else stops being an aggregate.",
"type": "object",
"required": ["administration"],
"additionalProperties": false,
"properties": {
"schema_version": { "type": ["string", "number"] },
"administration": {
"type": "object",
"required": ["id", "assessment", "n_examinees"],
"additionalProperties": false,
"properties": {
"id": { "type": "string" },
"assessment": { "type": "string" },
"term": { "type": "string" },
"date": date("When it was given."),
"forms": string_array("The forms in play."),
"n_examinees": {
"type": "integer",
"minimum": 0,
"description": "The number to read before any of the others: a \
point-biserial on twenty-seven students is a different \
kind of claim than one on three hundred."
},
"model": { "type": "string" },
"generated": date("When the analysis was run."),
"coursebank": { "type": "string" }
}
},
"items": {
"type": "array",
"items": {
"type": "object",
"required": ["item", "number", "n"],
"additionalProperties": false,
"properties": {
"item": { "type": "string" },
"number": { "type": "integer", "minimum": 1 },
"variant": { "type": "string" },
"stem_digest": { "type": "string" },
"n": { "type": "integer", "minimum": 0 },
"p_value": proportion("Proportion correct on this administration."),
"point_biserial": { "type": "number", "minimum": -1.0, "maximum": 1.0 },
"discrimination_index": {
"type": "number", "minimum": -1.0, "maximum": 1.0
},
"option_stats": {
"type": "object",
"additionalProperties": option_stat_schema()
},
"irt": irt_schema(),
"flags": {
"type": "array",
"items": {
"type": "string",
"enum": strings(
&Flag::ALL.iter().map(|f| f.as_str()).collect::<Vec<&str>>()
)
}
}
}
}
}
}
})
}
/// The schema for a retirement record.
fn retirement_schema() -> Value {
json!({
@@ -875,9 +1195,12 @@ fn item_identity_properties() -> Value {
json!({
"id": {
"type": "string",
"pattern": "^q-[a-z0-9]+(-[a-z0-9]+)*-[0-9]{3}$",
"description": "Item id, e.g. q-glycolysis-014. Stable forever: assessment records \
and stored responses refer to it."
"pattern": "^q-[a-z0-9]+(-[a-z0-9]+)*$",
"description": "Item id, e.g. q-glycolysis-rate-limiting-step. Stable forever: \
assessment records and stored responses refer to it, so renaming one \
is a migration rather than an edit. A trailing counter is no longer \
expected — it recorded when the item was written, which git knows — \
but an id that still has one stays valid."
},
"version": {
"type": "integer",
@@ -944,7 +1267,6 @@ fn item_content_properties() -> Value {
"prerequisites": string_array("Objective ids a student needs before this item."),
"assets": { "type": "array", "items": asset_schema() },
"design": design_schema(),
"calibration": calibration_schema(),
"review": review_schema(),
"history": {
"type": "array",
@@ -1157,7 +1479,21 @@ fn placement_schema() -> Value {
},
"points": { "type": "number", "minimum": 0.0 },
"bonus": { "type": "boolean" },
"key": string_array("Keyed option letters as administered."),
"key": string_array(
"The option ids keyed correct for this administration. One for a \
single_best_answer, chosen from the item's pool of defensible keys."
),
"distractors": string_array(
"The option ids offered alongside the key. Resolved when the assessment is \
assembled and written out explicitly, so a later bank edit cannot change the \
paper. Empty means the whole pool."
),
"variant": {
"type": "string",
"description": "Digest of the item as this administration showed it: stem, \
administered options, and which was keyed. The key statistics \
pool on."
},
"level": level(),
"learning_targets": string_array("Targets as administered."),
"credit_overrides": {
@@ -1260,7 +1596,8 @@ mod tests {
assert_eq!(props["options"]["maxItems"], 8);
assert_eq!(
props["options"]["items"]["properties"]["id"]["pattern"],
"^[A-H]$"
// Either form: the 2.0 name, or the letter it replaces.
"^(o-[a-z0-9]+(-[a-z0-9]+)*|[A-H])$"
);
}
@@ -1283,7 +1620,7 @@ mod tests {
let dir = std::env::temp_dir().join(format!("cb-schema-{}", std::process::id()));
std::fs::remove_dir_all(&dir).ok();
let written = write_all(&dir).unwrap();
assert_eq!(written.len(), 3);
assert_eq!(written.len(), Kind::ALL.len());
for path in &written {
assert!(path.exists());
let text = std::fs::read_to_string(path).unwrap();
+162 -2
View File
@@ -88,6 +88,16 @@ pub enum Rule {
/// An expectation of low discrimination on a higher-level item.
ContradictoryDesign,
// --- the option pool ---
/// Fewer usable distractors than a form shows.
ThinOptionPool,
/// An option that has never been administered and has not been retired.
UnusedOption,
/// An option retired without saying what it did.
UnjustifiedRetirement,
/// A distractor that has never been chosen, in any set it appeared in.
NonfunctioningDistractor,
// --- evidence ---
/// Statistics describe an older version of the item.
StaleCalibration,
@@ -104,7 +114,7 @@ impl Rule {
///
/// Used by `--list-rules`, and by the test that keeps this list in step with
/// the enum.
pub const ALL: [Rule; 26] = [
pub const ALL: [Rule; 30] = [
Rule::KeyIsLongest,
Rule::UnevenOptionLength,
Rule::WordRepeatCue,
@@ -127,6 +137,10 @@ impl Rule {
Rule::WeakFormatForLevel,
Rule::ScoredBonusLevel,
Rule::ContradictoryDesign,
Rule::ThinOptionPool,
Rule::UnusedOption,
Rule::UnjustifiedRetirement,
Rule::NonfunctioningDistractor,
Rule::StaleCalibration,
Rule::DifficultyMissed,
Rule::DiscriminationMissed,
@@ -152,6 +166,9 @@ impl Rule {
// Statistics attached to text that has since changed are actively
// misleading, which is worse than absent.
R::StaleCalibration => Severity::High,
// An item that cannot fill a form is an item `assemble` will put on
// a paper short an option.
R::ThinOptionPool => Severity::High,
// An unanswerable question for a screen-reader user.
R::AssetWithoutAltText => Severity::High,
@@ -178,6 +195,13 @@ impl Rule {
| R::NoStudentFeedback
| R::DifficultyMissed
| R::DiscriminationMissed => Severity::Low,
// A distractor that draws nobody across several administrations is
// evidence to act on, not a style note.
R::NonfunctioningDistractor => Severity::Medium,
// Both are tidiness: the item still works, but its pool is
// carrying something nobody has accounted for.
R::UnusedOption | R::UnjustifiedRetirement => Severity::Low,
}
}
@@ -206,6 +230,10 @@ impl Rule {
Rule::WeakFormatForLevel => "complete-format-level",
Rule::ScoredBonusLevel => "complete-bonus-policy",
Rule::ContradictoryDesign => "complete-design-conflict",
Rule::ThinOptionPool => "pool-thin",
Rule::UnusedOption => "pool-unused",
Rule::UnjustifiedRetirement => "pool-unjustified-retirement",
Rule::NonfunctioningDistractor => "evidence-nonfunctioning",
Rule::StaleCalibration => "evidence-stale",
Rule::DifficultyMissed => "evidence-difficulty",
Rule::DiscriminationMissed => "evidence-discrimination",
@@ -238,9 +266,11 @@ impl Rule {
| Rule::WeakFormatForLevel
| Rule::ScoredBonusLevel
| Rule::ContradictoryDesign => "completeness",
Rule::ThinOptionPool | Rule::UnusedOption | Rule::UnjustifiedRetirement => "pool",
Rule::StaleCalibration
| Rule::DifficultyMissed
| Rule::DiscriminationMissed
| Rule::NonfunctioningDistractor
| Rule::DuplicateStem => "evidence",
}
}
@@ -270,6 +300,10 @@ impl Rule {
Rule::WeakFormatForLevel,
Rule::ScoredBonusLevel,
Rule::ContradictoryDesign,
Rule::ThinOptionPool,
Rule::UnusedOption,
Rule::UnjustifiedRetirement,
Rule::NonfunctioningDistractor,
Rule::StaleCalibration,
Rule::DifficultyMissed,
Rule::DiscriminationMissed,
@@ -302,6 +336,12 @@ impl Rule {
Rule::WeakFormatForLevel => "true/false at an analytic level",
Rule::ScoredBonusLevel => "a level the policy reserves for bonus is scored",
Rule::ContradictoryDesign => "low expected discrimination on a higher-level item",
Rule::ThinOptionPool => "fewer usable distractors than a form shows",
Rule::UnusedOption => "an option has never been administered and is not retired",
Rule::UnjustifiedRetirement => "an option was retired without saying what it did",
Rule::NonfunctioningDistractor => {
"a distractor has never been chosen in any set it appeared in"
}
Rule::StaleCalibration => "statistics describe an older version of the item",
Rule::DifficultyMissed => "observed difficulty was far from predicted",
Rule::DiscriminationMissed => "observed discrimination contradicted the prediction",
@@ -749,6 +789,87 @@ pub fn lint_item(entry: &Entry, course: &CourseFile, t: &Thresholds) -> Vec<Find
}
}
// --- the option pool
// These only make sense once options are a pool, and the pool is where an
// item's spare parts sit. A bank that never draws from it will not trip any
// of them.
if it.format.has_options() {
let (keys, distractors) = it.pool();
let wanted = course.policy.options_per_item.saturating_sub(1);
if distractors.len() < wanted {
push(
Rule::ThinOptionPool,
Severity::High,
format!(
"has {} usable distractor(s) but a form shows {}, so `assemble` will put \
this on a paper an option short",
distractors.len(),
course.policy.options_per_item
),
);
}
for option in &it.options {
match &option.retired {
Some(retirement) => {
if retirement.reason.trim().len() < 12 {
push(
Rule::UnjustifiedRetirement,
Severity::Low,
format!(
"option `{}` is retired with no real reason. The reason is the \
finding — what it drew, or failed to draw — and it is the only \
part of a retirement worth anything in two years",
option.id
),
);
}
}
None => {
// An option nobody has been shown is a draft, and a draft
// sitting in an approved item's pool will eventually be
// drawn onto a paper without ever having been reviewed
// against data.
if let Some(history) = it
.calibration
.as_ref()
.filter(|c| !c.options.is_empty())
.and_then(|c| c.options.get(&option.id))
{
if history.never_chosen && history.appearances > 1 {
push(
Rule::NonfunctioningDistractor,
Severity::Medium,
format!(
"option `{}` has appeared in {} option set(s) across {} \
examinees and has never been chosen. One administration \
would not show this; several do.",
option.id, history.appearances, history.n_examinees
),
);
}
} else if it
.calibration
.as_ref()
.is_some_and(|c| !c.options.is_empty())
{
push(
Rule::UnusedOption,
Severity::Low,
format!(
"option `{}` has never been administered. Either it is waiting \
its turn, or it was drafted and forgotten — retire it and say \
which.",
option.id
),
);
}
}
}
}
let _ = keys;
}
// --- evidence
if !it.calibration_is_current() {
push(
@@ -1196,7 +1317,7 @@ mod tests {
fn entry(yaml: &str) -> Entry {
let item: Item = serde_yaml_ng::from_str(yaml).expect("item parses");
Entry {
uid: format!("b::{}", item.id),
uid: item.id.clone(),
bank: "b".into(),
path: PathBuf::from("b.yaml"),
index: 0,
@@ -1215,6 +1336,45 @@ mod tests {
c
}
#[test]
fn a_pool_too_thin_to_fill_a_form_is_flagged() {
let c = codes(
r#"
id: q-a-001
status: draft
level: 2
stem: Which mechanism best explains the sigmoidal binding curve?
options:
- { id: o-shift, text: Ligand binding shifts the tetramer to a higher-affinity state, correct: true }
- { id: o-fixed, text: Each subunit binds with the same fixed affinity throughout }
"#,
);
// Policy shows four options and the pool can supply two, so `assemble`
// would put this on a paper two short.
assert!(c.contains(&"pool-thin"), "{c:?}");
}
#[test]
fn a_retirement_with_no_finding_is_flagged() {
let c = codes(
r#"
id: q-a-001
status: draft
level: 2
stem: Which mechanism best explains the sigmoidal binding curve?
options:
- { id: o-shift, text: Ligand binding shifts the tetramer to a higher-affinity state, correct: true }
- { id: o-fixed, text: Each subunit binds with the same fixed affinity throughout }
- { id: o-consumed, text: "Ligand is consumed as it binds, depleting the available pool" }
- { id: o-oxidation, text: The heme iron changes oxidation state upon binding }
- { id: o-cooperative, text: Subunits bind independently of one another, retired: { 'on': 2026-09-20, reason: bad } }
"#,
);
assert!(c.contains(&"pool-unjustified-retirement"), "{c:?}");
// Four live distractors is enough for a four-option form.
assert!(!c.contains(&"pool-thin"), "{c:?}");
}
#[test]
fn clean_item_passes() {
let c = codes(
+154 -3
View File
@@ -33,8 +33,9 @@ use crate::course::{CourseFile, SCHEMA_VERSION};
use crate::date::Date;
use crate::error::{Error, Result};
use crate::history::History;
use crate::item::{Choice, Item};
use crate::rng::Rng;
use crate::taxonomy::Level;
use crate::taxonomy::{Format, Level};
/// The result of a draw.
#[derive(Debug, Clone)]
@@ -466,14 +467,23 @@ pub fn to_record(
.chain(selection.bonus.iter().map(|u| (u, true))),
) {
let e = catalog.require(uid)?;
let (key, distractors) = draw_options(
&e.item,
catalog.course.policy.options_per_item,
blueprint.seed.unwrap_or(0),
uid,
);
items.push(Placement {
number,
item: uid.clone(),
version: Some(e.item.version),
version: None,
stem_digest: Some(e.item.stem_digest()),
variant: Some(e.item.variant_digest(&key, &distractors)),
fingerprint: Some(e.item.fingerprint()),
points: Some(e.item.points(default_points)),
bonus: is_bonus || e.item.bonus,
key: e.item.key_letters(),
distractors,
key,
level: Some(e.item.level),
learning_targets: e.item.learning_targets.clone(),
credit_overrides: BTreeMap::new(),
@@ -570,6 +580,85 @@ pub fn layout(record: &AssessmentFile, form: &Form) -> Vec<Placement> {
scored.into_iter().chain(bonus).collect()
}
/// Draws the key and the distractors one placement administers.
///
/// Resolved here, at assembly, and written into the record as explicit lists.
/// Nothing downstream samples: an export that drew its own options would print
/// a different paper every time the bank was touched.
///
/// The draw is seeded on the blueprint and the item, so re-running `assemble`
/// with the same seed produces the same paper, and two items in one assessment
/// draw independently.
///
/// # Arguments
///
/// * `item` - the item, whose options are a pool.
/// * `per_item` - how many options a form shows, from course policy.
/// * `seed` - the blueprint seed.
/// * `uid` - the item id, salting the draw.
///
/// # Returns
///
/// The keyed ids and the distractor ids, each sorted, naming options of `item`.
/// Both empty for an item with no options, which is an open response.
pub fn draw_options(
item: &Item,
per_item: usize,
seed: u64,
uid: &str,
) -> (Vec<String>, Vec<String>) {
let (keys, distractors) = item.pool();
if keys.is_empty() && distractors.is_empty() {
return (Vec::new(), Vec::new());
}
// Multiple response keys every correct option; anything else keys one, and
// when the pool offers several defensible keys the draw picks one so that
// the record says which.
let wanted_keys = match item.format {
Format::MultipleResponse => keys.len(),
_ => 1.min(keys.len()),
};
let mut rng = Rng::from_label(&format!("{seed}/{uid}/options"));
let mut key_ids = pick(&keys, wanted_keys, &mut rng);
key_ids.sort();
// A pool with fewer usable distractors than the policy asks for is a
// finding, not a failure: the form comes out short and `lint` says so,
// rather than `assemble` refusing to build the assessment at all.
let wanted = per_item.saturating_sub(key_ids.len());
let mut distractor_ids = pick(&distractors, wanted.min(distractors.len()), &mut rng);
distractor_ids.sort();
(key_ids, distractor_ids)
}
/// Takes `n` options, preferring the ones that were designed rather than merely
/// written.
///
/// A distractor carrying a misconception and an error type is one you thought
/// about; one carrying neither is filler. When the pool is larger than the form,
/// the thought-about ones go on the paper. The shuffle comes first so that
/// options of equal standing are drawn by seed rather than by declaration
/// order.
fn pick(options: &[&Choice], n: usize, rng: &mut Rng) -> Vec<String> {
if n >= options.len() {
return options.iter().map(|o| o.id.clone()).collect();
}
let mut order: Vec<usize> = (0..options.len()).collect();
rng.shuffle(&mut order);
order.sort_by_key(|&i| {
let o = options[i];
u8::from(o.misconception.is_none()) + u8::from(o.error_type.is_none())
});
order
.into_iter()
.take(n)
.map(|i| options[i].id.clone())
.collect()
}
/// The option order for one item on one form.
///
/// # Arguments
@@ -692,6 +781,68 @@ mod tests {
assert_eq!(form_label(27), "AB");
}
/// An item whose options are given as YAML, so the test needs no literal.
fn pool_item(options: &str) -> Item {
let src = format!(
r#"id: q-x
status: approved
level: 1
cognitive_process: recall
stem: Which line holds the quality scores?
learning_targets: [t-x]
sources: [{{ lecture: L1 }}]
options:
{options}"#
);
serde_yaml_ng::from_str(&src).expect("item parses")
}
const DESIGNED: &str = r#" - { id: o-key, text: right, correct: true }
- { id: o-designed-a, text: a, misconception: mistakes the separator, error_type: recall_confusion }
- { id: o-designed-b, text: b, misconception: confuses the two, error_type: recall_confusion }
- { id: o-filler-a, text: c }
- { id: o-filler-b, text: d }
"#;
#[test]
fn a_draw_prefers_designed_distractors_and_is_reproducible() {
let item = pool_item(DESIGNED);
let (key, distractors) = draw_options(&item, 3, 1103, "q-x");
assert_eq!(key, vec!["o-key".to_string()]);
assert_eq!(distractors.len(), 2);
// Thought-about distractors go on the paper before filler does.
assert!(
distractors.iter().all(|d| d.starts_with("o-designed")),
"{distractors:?}"
);
// Same seed, same paper.
assert_eq!(draw_options(&item, 3, 1103, "q-x"), (key, distractors));
// A retired option is not drawn, and the form comes out of the rest.
let retired = pool_item(&DESIGNED.replace(
"{ id: o-designed-a, text: a,",
"{ id: o-designed-a, text: a, retired: { 'on': 2026-09-20, reason: nonfunctioning },",
));
let (_, after) = draw_options(&retired, 3, 1103, "q-x");
assert!(!after.iter().any(|d| d == "o-designed-a"), "{after:?}");
}
#[test]
fn a_thin_pool_comes_out_short_rather_than_refusing_to_build() {
let item = pool_item(
" - { id: o-key, text: right, correct: true }\n - { id: o-one, text: wrong }\n",
);
let (key, distractors) = draw_options(&item, 4, 7, "q-y");
assert_eq!(key.len(), 1);
assert_eq!(
distractors.len(),
1,
"one usable distractor, so one is drawn"
);
}
#[test]
fn option_order_is_a_reproducible_permutation() {
let form = Form {
+170
View File
@@ -24,6 +24,7 @@ use coursebank::assessment::{Kind as AssessmentKind, Platform};
use coursebank::catalog::Severity;
use coursebank::item::IrtModel;
use coursebank::lecture::Style as PageStyle;
use coursebank::references;
use coursebank::store;
/// Manage course item banks, assessments, and the analysis that comes back.
@@ -48,6 +49,15 @@ pub(crate) struct Cli {
pub(crate) enum Command {
/// Create a new course directory.
Init(InitArgs),
/// Inspect the course file.
#[command(subcommand)]
Course(CourseCommand),
/// One-time conversions from an older layout.
#[command(subcommand)]
Migrate(MigrateCommand),
/// Work with the bibliography.
#[command(subcommand)]
References(ReferencesCommand),
/// Write JSON Schemas so your editor can validate the YAML as you type.
Schema,
/// Check every file for problems that must be fixed.
@@ -93,6 +103,158 @@ pub(crate) enum Command {
Data,
}
/// `course`: the course file itself, which may be one file or many.
#[derive(Debug, Subcommand)]
pub(crate) enum CourseCommand {
/// List the files the course is assembled from.
Files,
/// Print the merged course, or write it to a file.
///
/// Nothing reads what this writes. It exists so you can see what the
/// fragments add up to, and diff two revisions of a course that no longer
/// lives in one file.
Build {
/// Output path; prints to stdout when omitted.
#[arg(long)]
out: Option<PathBuf>,
},
/// Say which file defines an id.
Where {
/// A unit, lecture, objective, target, reference, or stimulus id.
id: String,
},
}
/// `migrate`: the one-time conversions, grouped so they are findable together.
#[derive(Debug, Subcommand)]
pub(crate) enum MigrateCommand {
/// Split one course.yaml into references.yaml, lectures/, and objectives/.
///
/// The original is kept as course.yaml.bak, and the result is reassembled
/// and compared against it before the command reports success.
Split {
/// Show what would be written, and write nothing.
#[arg(long)]
dry_run: bool,
},
/// Drop the trailing counter from every item id.
///
/// A rename, not a normalization: `q-x-001` and `q-x` are unrelated
/// strings, so banks, records, seals, and the response store are rewritten
/// in one pass or not at all. Two ids that would collide abort it.
Counters {
/// Show the renames, and write nothing.
#[arg(long)]
dry_run: bool,
},
/// Replace the `order:` integers with ordered declarations.
///
/// Objective order comes from each lecture's `teaches` list, target order
/// from a `targets:` list this writes onto each objective. Run
/// `migrate split` first — without `teaches`, objective order has no
/// source.
Order {
/// Show what would change, and write nothing.
#[arg(long)]
dry_run: bool,
},
/// Turn citations written into `note:` fields into real fields.
///
/// Journal, volume, pages, and DOI parsed out of the prose, with whatever
/// the note still says left in it. A field the entry already declares is
/// never overwritten; a disagreement is reported instead.
References {
/// Show what would change, and write nothing.
#[arg(long)]
dry_run: bool,
},
/// Fill in the stored `variant` column from the assessment records.
///
/// Nothing in the tool needs it — a variant is derived from the placement
/// when a row has none. It is for pandas, DuckDB, and R, which see only
/// what is in the column and will otherwise average two option sets of one
/// stem into an item that never existed.
Variants {
/// Show what would change, and write nothing.
#[arg(long)]
dry_run: bool,
},
/// Drop `version:` and `history:`, which 2.0 ignores.
///
/// A stem's text is its identity: reword it and it is a new item with a new
/// id and `supersedes:` pointing back. `validate` enforces that against
/// every seal, so what a version number used to hint at is now checked.
Stems {
/// Show what would change, and write nothing.
#[arg(long)]
dry_run: bool,
},
/// Rewrite option letters as names derived from the option text.
///
/// A letter is a position, and a position in a field that pooled
/// statistics and `credit_overrides` join on is a bug waiting for someone
/// to reorder a YAML block. Run with --dry-run first: the names land in
/// the response store, so they are as permanent as an item id.
Options {
/// Show the derived names, and write nothing.
#[arg(long)]
dry_run: bool,
},
/// Rewrite pre-2.0 `bank::item` ids as the item ids they name.
///
/// Touches assessment records, seals, and the response store. Everything
/// keeps working unmigrated — an old id still resolves — but a store
/// holding both forms groups one question into two for anything reading the
/// Parquet without this tool.
Ids {
/// Show what would change, and write nothing.
#[arg(long)]
dry_run: bool,
},
}
/// `references`: the bibliography, and the formats other tools read it in.
#[derive(Debug, Subcommand)]
pub(crate) enum ReferencesCommand {
/// List every work, with the link a reading list would use.
///
/// The column that matters is the last one: a work with no link is one a
/// student cannot reach from a report, which for a manuscript usually means
/// its DOI is missing.
List,
/// Write the bibliography in a citation format.
Export {
/// Which format to write.
#[arg(long, value_enum, default_value = "hayagriva")]
format: ReferenceFormat,
/// Output path; prints to stdout when omitted.
#[arg(long)]
out: Option<PathBuf>,
},
}
/// The citation formats `references export` can write.
#[derive(Debug, Clone, Copy, ValueEnum)]
pub(crate) enum ReferenceFormat {
/// Hayagriva YAML, which Typst reads natively.
Hayagriva,
/// CSL-JSON, for Zotero, Pandoc, and CSL processors.
CslJson,
/// BibTeX.
Bibtex,
}
impl ReferenceFormat {
/// The library-side format.
pub(crate) fn as_format(self) -> references::Format {
match self {
ReferenceFormat::Hayagriva => references::Format::Hayagriva,
ReferenceFormat::CslJson => references::Format::CslJson,
ReferenceFormat::Bibtex => references::Format::Bibtex,
}
}
}
#[derive(Debug, Args)]
pub(crate) struct InitArgs {
/// Course code, e.g. "BIOSC 1540".
@@ -593,6 +755,14 @@ pub(crate) enum AnalyzeCommand {
/// Pool every stored administration of this assessment.
#[arg(long)]
pooled: bool,
/// Also write the administration record under analysis/.
///
/// Two CSVs, one row per question and one per question-and-option,
/// written once and never rewritten. Cohort aggregates only, so unlike
/// the response data they are meant to be committed — which is what
/// keeps the history when `data/` rotates.
#[arg(long = "record")]
record: bool,
},
/// Fit an IRT model.
Irt {
+5 -2
View File
@@ -8,8 +8,8 @@
//! calls the matching handler. The handlers themselves live in submodules that
//! follow the workflow described in the crate documentation:
//!
//! - [`project`] — set up and check a course: `init`, `schema`, `validate`,
//! `lint`, `catalog`.
//! - [`project`] — set up and check a course: `init`, `course`, `schema`,
//! `validate`, `lint`, `catalog`.
//! - [`lectures`] — render a lecture's reading list and check what backs each
//! objective: `lecture`.
//! - [`banks`] — manage items and build assessments: `bank`, `assessment`,
@@ -55,6 +55,9 @@ pub(crate) enum Outcome {
pub(crate) fn run(cli: &Cli) -> Result<Outcome> {
match &cli.command {
Command::Init(args) => project::init(cli, args),
Command::Course(sub) => project::course(cli, sub),
Command::References(sub) => project::references(cli, sub),
Command::Migrate(sub) => project::migrate(cli, sub),
Command::Schema => project::schema(cli),
Command::Validate => project::validate(cli),
Command::Lint(args) => project::lint(cli, args),
+44 -7
View File
@@ -294,12 +294,46 @@ pub(crate) fn analyze(cli: &Cli, sub: &AnalyzeCommand) -> Result<Outcome> {
let store = Store::open(catalog.layout.data())?;
match sub {
AnalyzeCommand::Items { id, pooled } => {
AnalyzeCommand::Items {
id,
pooled,
record: write_record,
} => {
let record = load_record(&catalog, id)?;
let set = responses_for(&store, &catalog, &record, *pooled)?;
let analysis =
classical::analyze(&set, &Thresholds::default(), Some(&record), Some(&catalog));
if *write_record {
// The full administration id, not the assessment id: `e1` is
// given again next year, and a record named after it would
// collide with this cohort's — which `write_csv` would report
// as "already recorded" when it is a different exam entirely.
let administration = coursebank::responses::administration_id(
&catalog.course.course.code,
record
.assessment
.term
.as_deref()
.unwrap_or(&catalog.course.course.term),
&record.assessment.id,
);
let written = calibrate::record_measurements(
&catalog.layout,
&administration,
&analysis,
&catalog,
Some(&record),
)?;
for path in &written {
println!("wrote {}", path.display());
}
println!(
"\nCohort aggregates only, so these are committed. Review with\n git diff \
analysis/\n"
);
}
for w in &analysis.warnings {
println!("! {w}\n");
}
@@ -463,17 +497,20 @@ pub(crate) fn calibrate(cli: &Cli, args: &CalibrateArgs) -> Result<Outcome> {
}
if !args.apply {
println!(
"Nothing written. Re-run with --apply to write these {} change(s) into the bank \
files, then review the git diff.",
"Nothing written. Re-run with --apply to write these {} change(s) into \
analysis/calibration.yaml, then review the git diff.",
plan.changes.len()
);
return Ok(Outcome::Ok);
}
for path in calibrate::apply(&plan)? {
println!("updated {}", path.display());
}
println!("\nReview the diff before committing: git diff banks/");
let path = calibrate::apply(&catalog.layout, &plan)?;
println!("updated {}", path.display());
println!(
"\nReview the diff before committing: git diff analysis/\n\nThe banks are untouched. \
Statistics are cohort aggregates with no student in\n them, which is why analysis/ is \
committed and data/ is not."
);
Ok(Outcome::Ok)
}
+443 -8
View File
@@ -5,9 +5,10 @@
//! Setting up a course and checking it stays well-formed.
//!
//! These are the commands you reach for before and around authoring: create the
//! directory (`init`), write editor schemas (`schema`), and run the two kinds of
//! checking — [`validate`] for problems that must be fixed and [`lint`] for
//! item-writing guidance. [`catalog`] summarizes the pool that results.
//! directory (`init`), see and split the course file (`course`), export the
//! bibliography (`references`), write editor schemas (`schema`), and run the two
//! kinds of checking — [`validate`] for problems that must be fixed and [`lint`]
//! for item-writing guidance. [`catalog`] summarizes the pool that results.
use std::collections::{BTreeMap, BTreeSet};
use std::fs;
@@ -15,24 +16,34 @@ use std::path::Path;
use coursebank::assessment::AssessmentFile;
use coursebank::bank::BankFile;
use coursebank::course::fragment::{self, Section};
use coursebank::course::{COURSE_FILE, CourseFile};
use coursebank::error::{Error, Result};
use coursebank::jsonschema;
use coursebank::layout::Layout;
use coursebank::lint::{self, Rule};
use coursebank::migrate;
use coursebank::references;
use coursebank::taxonomy::{Level, Tier};
use coursebank::yaml;
use crate::cli::{CatalogArgs, Cli, InitArgs, LintArgs};
use crate::cli::{
CatalogArgs, Cli, CourseCommand, InitArgs, LintArgs, MigrateCommand, ReferencesCommand,
};
use crate::commands::Outcome;
use crate::helpers::{load, truncate};
/// The `.gitignore` written by `init`.
pub(crate) const GITIGNORE: &str = "\
# Generated output: exports, rendered exams, reports.
# Generated output: exports, rendered exams, reports. A student report carries
# names, so it belongs here rather than in the repository.
build/
reports/
# Response data: every row carries a student. The statistics derived from it are
# cohort aggregates and live in analysis/, which is committed on purpose.
data/
# Typst and PDF artifacts.
*.pdf
@@ -77,13 +88,430 @@ pub(crate) fn init(cli: &Cli, args: &InitArgs) -> Result<Outcome> {
write_gitignore(&cli.course.join(".gitignore"))?;
println!(
"\nNext: edit {} to add your learning objectives, their targets, and your\n lectures, then\n \
coursebank bank new unit-1 --title \"Unit 1\"\n coursebank validate",
COURSE_FILE
"\nNext: edit {COURSE_FILE} to add your learning objectives, their targets, and\n \
your lectures, then\n coursebank bank new unit-1 --title \"Unit 1\"\n \
coursebank validate\n\nOnce {COURSE_FILE} is more than you want to scroll, \
`coursebank migrate split`\n moves each lecture and objective into its own file under \
lectures/ and\n objectives/, and every command goes on reading the course as one."
);
Ok(Outcome::Ok)
}
/// `course`: inspect the course file, or split it into fragments.
pub(crate) fn course(cli: &Cli, sub: &CourseCommand) -> Result<Outcome> {
match sub {
CourseCommand::Files => course_files(cli),
CourseCommand::Build { out } => course_build(cli, out.as_deref()),
CourseCommand::Where { id } => course_where(cli, id),
}
}
/// `migrate`: the one-time layout conversions.
pub(crate) fn migrate(cli: &Cli, sub: &MigrateCommand) -> Result<Outcome> {
match sub {
MigrateCommand::Split { dry_run } => course_split(cli, *dry_run),
MigrateCommand::Ids { dry_run } => migrate_ids(cli, *dry_run),
MigrateCommand::Options { dry_run } => migrate_options(cli, *dry_run),
MigrateCommand::Stems { dry_run } => migrate_stems(cli, *dry_run),
MigrateCommand::Variants { dry_run } => migrate_variants(cli, *dry_run),
MigrateCommand::References { dry_run } => migrate_references(cli, *dry_run),
MigrateCommand::Order { dry_run } => migrate_order(cli, *dry_run),
MigrateCommand::Counters { dry_run } => migrate_counters(cli, *dry_run),
}
}
/// Drops the trailing counter from every item id.
fn migrate_counters(cli: &Cli, dry_run: bool) -> Result<Outcome> {
let (rename, touched) = migrate::counters(&cli.course, !dry_run)?;
if rename.is_empty() {
println!("nothing to do: no item id ends in a counter");
return Ok(Outcome::Ok);
}
for (old, new) in &rename {
println!(" {old} -> {new}");
}
println!();
for (path, n) in &touched {
println!(" {:<44} {n:>5} reference(s)", path.display());
}
if dry_run {
println!("\nnothing written");
return Ok(Outcome::Ok);
}
let sealed = touched
.iter()
.filter(|(p, _)| p.starts_with("seals"))
.count();
println!(
"\nrenamed {} item(s) across {} file(s)",
rename.len(),
touched.len()
);
if sealed > 0 {
println!(
"\n{sealed} seal(s) were rewritten. The ids are inside the digest, so each one was \
recomputed\n and the previous digest recorded under `superseded_digests`. A seal \
that has been\n rewritten says so rather than looking untouched."
);
}
println!(
"\nNext:\n coursebank validate\n coursebank seal verify\n coursebank analyze items \
--all # the join key moved; check the data still lands"
);
Ok(Outcome::Ok)
}
/// Replaces the order integers with ordered declarations.
fn migrate_order(cli: &Cli, dry_run: bool) -> Result<Outcome> {
let touched = migrate::order(&cli.course, !dry_run)?;
if touched.is_empty() {
println!("nothing to do: no `order:` left to derive");
return Ok(Outcome::Ok);
}
let total: usize = touched.iter().map(|(_, n)| n).sum();
for (path, n) in &touched {
println!(" {:<44} {n:>5} order(s) dropped", path.display());
}
if dry_run {
println!("\nnothing written");
return Ok(Outcome::Ok);
}
println!(
"\nrewrote {} file(s), {total} integer(s) gone\n\nNext:\n coursebank validate\n \
coursebank lecture objectives L1.2 # check the order still reads right",
touched.len()
);
Ok(Outcome::Ok)
}
/// Takes the citations out of the notes and puts them in fields.
fn migrate_references(cli: &Cli, dry_run: bool) -> Result<Outcome> {
let (touched, notes) = migrate::references(&cli.course, !dry_run)?;
for (path, n) in &touched {
println!(" {:<44} {n:>5} note(s) taken apart", path.display());
}
for note in &notes {
println!(" note: {note}");
}
if touched.is_empty() {
println!("nothing to do: no note is carrying a citation");
return Ok(Outcome::Ok);
}
if dry_run {
println!("\nnothing written");
return Ok(Outcome::Ok);
}
println!(
"\nrewrote {} file(s)\n\nNext:\n coursebank validate\n coursebank references list\n\n\
An issue number is never inferred, not even from a DOI that encodes one, so add those \
by hand.",
touched.len()
);
Ok(Outcome::Ok)
}
/// Fills in the stored variant column.
fn migrate_variants(cli: &Cli, dry_run: bool) -> Result<Outcome> {
let touched = migrate::store_variants(&cli.course, !dry_run)?;
if touched.is_empty() {
println!("nothing to fill in: every stored row already names its variant");
return Ok(Outcome::Ok);
}
for (path, n) in &touched {
println!(" {:<44} {n:>5} row(s)", path.display());
}
if dry_run {
println!("\nnothing written");
return Ok(Outcome::Ok);
}
println!("\nrewrote {} data file(s)", touched.len());
Ok(Outcome::Ok)
}
/// Drops the version fields 2.0 ignores.
fn migrate_stems(cli: &Cli, dry_run: bool) -> Result<Outcome> {
let touched = migrate::stems(&cli.course, !dry_run)?;
if touched.is_empty() {
println!("nothing to migrate: no `version:` or `history:` left to drop");
return Ok(Outcome::Ok);
}
for (path, n) in &touched {
println!(" {:<44} {n:>5} line(s) dropped", path.display());
}
if dry_run {
println!("\nnothing written");
return Ok(Outcome::Ok);
}
println!(
"\nrewrote {} file(s)\n\nNext:\n coursebank validate\n\nFrom here, rewording a stem \
is an error rather than a version bump: give the new\n wording a new id and \
`supersedes:` the old one.",
touched.len()
);
Ok(Outcome::Ok)
}
/// Rewrites option letters as names, showing every name before writing.
fn migrate_options(cli: &Cli, dry_run: bool) -> Result<Outcome> {
let (map, problems) = migrate::options_plan(&cli.course)?;
for (item, options) in &map {
println!("{item}");
for (letter, name) in options {
println!(" {letter} -> {name}");
}
}
if !problems.is_empty() {
println!("\n{} item(s) need naming by hand:", problems.len());
for problem in &problems {
println!(" - {problem}");
}
}
if map.is_empty() {
println!("nothing to migrate: every option is already named");
return Ok(Outcome::Ok);
}
let touched = migrate::apply_options(&cli.course, &map, !dry_run)?;
println!();
for (path, n) in &touched {
println!(" {:<44} {n:>5} rename(s)", path.display());
}
if dry_run {
println!("\nnothing written");
return Ok(if problems.is_empty() {
Outcome::Ok
} else {
Outcome::Findings
});
}
println!(
"\nrewrote {} file(s)\n\nSeals keep their letters on purpose; see `coursebank migrate \
--help`.\nNext:\n coursebank validate\n coursebank lint",
touched.len()
);
Ok(if problems.is_empty() {
Outcome::Ok
} else {
Outcome::Findings
})
}
/// Rewrites pre-2.0 bank-qualified item ids everywhere they are stored.
fn migrate_ids(cli: &Cli, dry_run: bool) -> Result<Outcome> {
let files = migrate::qualified_ids(&cli.course)?;
for (path, n, _) in &files {
println!(" {:<44} {n:>5} id(s)", path.display());
}
if !dry_run {
migrate::apply_ids(&cli.course, &files)?;
}
let data = migrate::store_ids(&cli.course, !dry_run)?;
for (path, n) in &data {
println!(" {:<44} {n:>5} row(s)", path.display());
}
if files.is_empty() && data.is_empty() {
println!("nothing to migrate: every item id already names the item course-wide");
return Ok(Outcome::Ok);
}
if dry_run {
println!("\nnothing written");
return Ok(Outcome::Ok);
}
println!(
"\nrewrote {} file(s) and {} data file(s)\n\nNext:\n coursebank validate\n \
coursebank analyze items --all",
files.len(),
data.len()
);
Ok(Outcome::Ok)
}
/// Lists the fragments a course is assembled from, with what each defines.
fn course_files(cli: &Cli) -> Result<Outcome> {
let layout = Layout::new(&cli.course);
let course = CourseFile::load_dir(&cli.course)?;
for (path, role) in fragment::files(&layout)? {
if !path.exists() {
continue;
}
let shown = path.strip_prefix(&cli.course).unwrap_or(&path);
let mut defines: Vec<String> = Vec::new();
for section in Section::ALL {
let n = course
.origins
.iter()
.filter(|((s, _), p)| *s == section && p.as_path() == shown)
.count();
if n == 0 {
continue;
}
defines.push(match section {
// These are declared once for the whole course, so a count
// would always be 1 and would read as though it could be more.
Section::Course | Section::Policy => section.key().to_string(),
_ => format!("{n} {}", section.key()),
});
}
println!(
"{:<40} {:<11} {}",
shown.display(),
role.label(),
defines.join(", ")
);
}
Ok(Outcome::Ok)
}
/// Prints or writes the merged course.
fn course_build(cli: &Cli, out: Option<&Path>) -> Result<Outcome> {
let course = CourseFile::load_dir(&cli.course)?;
match out {
Some(path) => {
course.write_resolved(path)?;
println!(
"wrote {} from {} file(s)",
path.display(),
course.fragment_paths().len()
);
}
None => print!("{}", yaml::to_string(&course)?),
}
Ok(Outcome::Ok)
}
/// Says which file defines an id.
fn course_where(cli: &Cli, id: &str) -> Result<Outcome> {
let course = CourseFile::load_dir(&cli.course)?;
match course.origin(id) {
Some((section, path)) => {
println!(
"{} defines `{id}` under `{}`",
path.display(),
section.key()
);
Ok(Outcome::Ok)
}
None => Err(Error::Unresolved {
kind: "id",
id: id.to_string(),
context: Some(cli.course.display().to_string()),
}),
}
}
/// Splits `course.yaml` into fragments, then checks the result reassembles.
fn course_split(cli: &Cli, dry_run: bool) -> Result<Outcome> {
let plan = migrate::split_plan(&cli.course)?;
println!("{} file(s):", plan.files.len());
for (path, lines) in plan.lines() {
let summary = plan
.files
.iter()
.find(|f| f.path == path)
.map(|f| f.summary.clone())
.unwrap_or_default();
println!(" {:<44} {lines:>5} lines {summary}", path.display());
}
for note in &plan.notes {
println!("\nnote: {note}");
}
if dry_run {
println!("\nnothing written");
return Ok(Outcome::Ok);
}
// Loaded before anything is written, since it is the thing the result is
// checked against.
let before = CourseFile::load(&Layout::new(&cli.course).course_file())?;
let written = migrate::apply(&cli.course, &plan)?;
println!("\nwrote {} file(s)", written.len());
let after = CourseFile::load_dir(&cli.course)?;
let diffs = migrate::differences(&before, &after)?;
if diffs.is_empty() {
println!(
"reassembled and compared against {COURSE_FILE}.bak: identical\n\nNext:\n \
coursebank validate\n coursebank schema\n git add -A && git diff --cached --stat"
);
return Ok(Outcome::Ok);
}
println!(
"\n{} difference(s) between the original and the reassembled course:",
diffs.len()
);
for diff in &diffs {
println!(" - {diff}");
}
println!(
"\nThe original is at {COURSE_FILE}.bak. Restore it with\n mv {COURSE_FILE}.bak \
{COURSE_FILE} && rm -r lectures objectives {}",
fragment::REFERENCES_FILE
);
Ok(Outcome::Findings)
}
/// `references`: list the bibliography, or export it in a citation format.
pub(crate) fn references(cli: &Cli, sub: &ReferencesCommand) -> Result<Outcome> {
let course = CourseFile::load_dir(&cli.course)?;
match sub {
ReferencesCommand::List => {
let mut unreachable = 0;
for (key, reference) in &course.references {
let link = match reference.href(None, None) {
Some(url) => url,
None => {
unreachable += 1;
"(no link)".to_string()
}
};
println!(
"{:<28} {:<10} {:<9} {:<6} {link}",
truncate(key, 27),
format!("{:?}", reference.kind).to_lowercase(),
format!("{:?}", reference.role).to_lowercase(),
reference.label_or("-"),
);
}
if unreachable > 0 && !cli.quiet {
println!(
"\n{unreachable} work(s) with no link. A student report can name one but \
cannot send anyone to it; for a manuscript, add its `doi`."
);
}
Ok(Outcome::Ok)
}
ReferencesCommand::Export { format, out } => {
let format = format.as_format();
let text = references::render(&course, format)?;
match out {
Some(path) => {
yaml::write_text(path, &text)?;
println!(
"wrote {} ({} work(s) as {})",
path.display(),
course.references.len(),
format.label()
);
}
None => print!("{text}"),
}
Ok(Outcome::Ok)
}
}
}
/// What reconciling [`GITIGNORE`] against a file already on disk would do.
struct GitignoreMerge {
/// The file to write. Identical to the input when nothing was missing.
@@ -260,6 +688,13 @@ pub(crate) fn validate(cli: &Cli) -> Result<Outcome> {
}
}
let seals = coursebank::seal::SealFile::load_all(&catalog.layout.seals())?;
all.extend(catalog.validate_seals(&seals));
// The statistics are kept in a different file from the questions they
// describe, so the link between them is worth checking rather than assuming.
all.extend(catalog.calibration.validate(&catalog));
if all.is_empty() {
if !cli.quiet {
println!(
+1
View File
@@ -337,6 +337,7 @@ pub fn ingest(
item_number: number,
item_ref,
item_version: None,
variant: None,
selected,
selected_source: Vec::new(),
eliminated: Vec::new(),
+4 -3
View File
@@ -248,7 +248,8 @@ impl FormDecoder {
for (index, placement) in printed.iter().enumerate() {
let entry = catalog.require(&placement.item)?;
let item = &entry.item;
let order = select::option_order(form, &placement.item, item.options.len());
let shown = item.administered(&placement.key, &placement.distractors);
let order = select::option_order(form, &placement.item, shown.len());
let canonical_key: BTreeSet<String> = if placement.key.is_empty() {
item.key_letters().into_iter().collect()
@@ -260,8 +261,7 @@ impl FormDecoder {
let mut to_printed = BTreeMap::new();
let mut printed_key = Vec::new();
for (position, source_index) in order.iter().enumerate() {
let canonical = item
.options
let canonical = shown
.get(*source_index)
.map(|c| c.id.clone())
.unwrap_or_else(|| printed_letter(*source_index));
@@ -738,6 +738,7 @@ mod tests {
form_position: None,
item_ref: None,
item_version: None,
variant: None,
selected: vec![selected.into()],
eliminated: Vec::new(),
selected_source: Vec::new(),
+1
View File
@@ -645,6 +645,7 @@ pub fn to_responses(questions: &[Question], ctx: &Context) -> Import {
item_number: q.number,
item_ref: None,
item_version: None,
variant: None,
selected,
selected_source: Vec::new(),
eliminated,
+93 -1
View File
@@ -75,6 +75,16 @@ pub struct Response {
/// The item version as administered.
pub item_version: Option<u32>,
/// The variant administered: which option set this row's student saw.
///
/// The grouping key for pooled statistics. Set at ingest from the
/// assessment record, and stored so that anything reading the Parquet
/// without this tool can group the same way — a store that only has
/// `item_ref` cannot tell two option sets of one stem apart, and will
/// average them.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub variant: Option<String>,
/// Option letters the student chose.
pub selected: Vec<String>,
/// The selected options in the bank's own lettering, written at ingest by
@@ -314,6 +324,49 @@ impl ResponseSet {
per_item.values().sum()
}
/// How many rows each option set of each item has.
///
/// What a calibration report needs to be honest about sample size: an item
/// administered three times with three different option sets has three
/// cells, not one, and reporting "n = 72" of it would be wrong three ways.
///
/// # Returns
///
/// Row counts keyed by item id and variant, with an empty variant for rows
/// that carry none.
pub fn variants(&self) -> BTreeMap<(String, String), usize> {
let mut out: BTreeMap<(String, String), usize> = BTreeMap::new();
for row in &self.rows {
let Some(item) = &row.item_ref else { continue };
let key = (item.clone(), row.variant.clone().unwrap_or_default());
*out.entry(key).or_insert(0) += 1;
}
out
}
/// The rows for one option set of one item.
///
/// # Arguments
///
/// * `item_ref` - the item id.
/// * `variant` - the variant digest, or `None` for rows carrying none.
///
/// # Returns
///
/// A set holding only those rows, keeping the warnings of the original.
pub fn for_variant(&self, item_ref: &str, variant: Option<&str>) -> ResponseSet {
ResponseSet {
rows: self
.rows
.iter()
.filter(|r| r.item_ref.as_deref() == Some(item_ref))
.filter(|r| r.variant.as_deref() == variant)
.cloned()
.collect(),
warnings: self.warnings.clone(),
}
}
/// Builds the response matrix for psychometrics.
///
/// # Arguments
@@ -416,6 +469,9 @@ impl ResponseSet {
};
r.item_ref = Some(p.item.clone());
r.item_version = p.version;
// Recorded when the record says so; derived from the option set
// otherwise, which is the case for every administration before 2.0.
r.variant = p.variant.clone();
r.bonus = r.bonus || p.bonus;
r.dropped = r.dropped || p.dropped;
r.dropped_full_credit = r.dropped_full_credit || p.dropped_with_credit();
@@ -434,6 +490,9 @@ impl ResponseSet {
if let Some(cat) = catalog {
if let Some(entry) = cat.get(&p.item) {
if r.variant.is_none() {
r.variant = Some(p.variant_of(&entry.item));
}
r.level = Some(entry.item.level);
r.learning_targets = if p.learning_targets.is_empty() {
entry.item.learning_targets.clone()
@@ -636,6 +695,8 @@ pub struct FlatResponse {
pub item_ref: String,
/// The item version, 0 when unknown.
pub item_version: u32,
/// The administered variant, empty when unknown.
pub variant: String,
/// Comma-joined selected letters.
pub selected: String,
/// Comma-joined eliminated letters.
@@ -710,6 +771,7 @@ impl FlatResponse {
item_number: r.item_number,
item_ref: r.item_ref.clone().unwrap_or_default(),
item_version: r.item_version.unwrap_or(0),
variant: r.variant.clone().unwrap_or_default(),
selected: r.selected.join(","),
selected_source: r.selected_source.join(","),
eliminated: r.eliminated.join(","),
@@ -767,7 +829,11 @@ impl FlatResponse {
email: none_if_empty(&self.email),
section: none_if_empty(&self.section),
item_number: self.item_number,
item_ref: none_if_empty(&self.item_ref),
// Canonicalized on read, so a term ingested before 2.0 pools with
// one ingested after it instead of splitting into two items.
item_ref: none_if_empty(&self.item_ref)
.map(|id| crate::item::canonical_id(&id).to_string()),
variant: none_if_empty(&self.variant),
item_version: if self.item_version == 0 {
None
} else {
@@ -825,6 +891,7 @@ mod tests {
item_number: number,
item_ref: None,
item_version: None,
variant: None,
selected: vec!["A".into()],
selected_source: vec![],
eliminated: vec![],
@@ -843,6 +910,31 @@ mod tests {
}
}
#[test]
fn rows_group_by_the_option_set_they_administered() {
let mut set = ResponseSet::new();
for (student, variant) in [("s1", "v1"), ("s2", "v1"), ("s3", "v2")] {
let mut r = row(student, 1, 1.0);
r.item_ref = Some("q-x".into());
r.variant = Some(variant.into());
set.rows.push(r);
}
// A row from before the column existed.
let mut old = row("s4", 1, 1.0);
old.item_ref = Some("q-x".into());
set.rows.push(old);
let counts = set.variants();
assert_eq!(counts[&("q-x".to_string(), "v1".to_string())], 2);
assert_eq!(counts[&("q-x".to_string(), "v2".to_string())], 1);
// Unknown groups on its own rather than joining either set.
assert_eq!(counts[&("q-x".to_string(), String::new())], 1);
assert_eq!(set.for_variant("q-x", Some("v1")).rows.len(), 2);
assert_eq!(set.for_variant("q-x", None).rows.len(), 1);
assert_eq!(set.for_variant("q-other", Some("v1")).rows.len(), 0);
}
#[test]
fn matrix_is_students_by_items() {
let mut set = ResponseSet::new();
+68 -1
View File
@@ -316,6 +316,70 @@ pub fn read_path(path: &Path) -> Result<ResponseSet> {
}
}
/// Reads a response file without turning its rows into [`Response`]s.
///
/// What a migration wants: the rows exactly as they sit on disk, so rewriting
/// one column cannot disturb another through a round trip.
///
/// # Arguments
///
/// * `path` - the file to read.
///
/// # Returns
///
/// The rows.
///
/// # Errors
///
/// Returns [`Error::Other`] for an unrecognized extension, [`Error::Csv`] or
/// [`Error::Other`] on a parse failure, and [`Error::FeatureDisabled`] for
/// Parquet without the feature.
pub fn read_flat(path: &Path) -> Result<Vec<FlatResponse>> {
match Format::from_path(path) {
Some(Format::Csv) => {
let mut r = csv::Reader::from_path(path).map_err(|e| Error::Csv {
path: path.to_path_buf(),
source: e,
})?;
let mut out = Vec::new();
for rec in r.deserialize::<FlatResponse>() {
out.push(rec.map_err(|e| Error::Csv {
path: path.to_path_buf(),
source: e,
})?);
}
Ok(out)
}
Some(Format::Parquet) => crate::store_parquet::read(path),
None => Err(Error::Other(format!(
"{} is not a response file; expected a .parquet or .csv",
path.display()
))),
}
}
/// Writes flat responses back to the file they came from.
///
/// # Arguments
///
/// * `path` - the destination, whose extension picks the format.
/// * `rows` - the rows.
///
/// # Errors
///
/// Returns [`Error::Other`] for an unrecognized extension and
/// [`Error::FeatureDisabled`] for Parquet without the feature.
pub fn write_flat(path: &Path, rows: &[FlatResponse]) -> Result<()> {
match Format::from_path(path) {
Some(Format::Csv) => write_csv(path, rows),
Some(Format::Parquet) => write_parquet(path, rows),
None => Err(Error::Other(format!(
"{} is not a response file; expected a .parquet or .csv",
path.display()
))),
}
}
/// Writes flat responses as CSV.
///
/// # Arguments
@@ -550,6 +614,7 @@ mod tests {
item_number: number,
item_ref: Some("bank::q-x-001".into()),
item_version: Some(2),
variant: None,
selected: vec!["C".into()],
selected_source: vec![],
eliminated: vec![],
@@ -591,7 +656,9 @@ mod tests {
let back = store.read("BIOSC1540/2026s/exam-4").unwrap();
assert_eq!(back.rows.len(), 2);
assert_eq!(back.rows[0].item_ref.as_deref(), Some("bank::q-x-001"));
// Canonicalized on the way in: the row was written with a pre-2.0
// `bank::item` key, and reading it yields the item it names.
assert_eq!(back.rows[0].item_ref.as_deref(), Some("q-x-001"));
assert_eq!(back.rows[0].selected, vec!["C".to_string()]);
assert_eq!(back.rows[0].learning_targets, vec!["lo-a".to_string()]);
+7
View File
@@ -49,6 +49,7 @@ pub fn schema() -> Schema {
Field::new("item_number", DataType::UInt32, false),
Field::new("item_ref", DataType::Utf8, false),
Field::new("item_version", DataType::UInt32, false),
Field::new("variant", DataType::Utf8, false),
Field::new("selected", DataType::Utf8, false),
Field::new("eliminated", DataType::Utf8, false),
Field::new("correct", DataType::Utf8, false),
@@ -114,6 +115,7 @@ fn to_batch(rows: &[FlatResponse]) -> Result<RecordBatch> {
u32c(|r| r.item_number),
s(|r| &r.item_ref),
u32c(|r| r.item_version),
s(|r| &r.variant),
s(|r| &r.selected),
s(|r| &r.eliminated),
s(|r| &r.correct),
@@ -279,6 +281,9 @@ fn from_batch(batch: &RecordBatch, path: &Path) -> Result<Vec<FlatResponse>> {
let item_number = uints("item_number")?;
let item_ref = strings("item_ref")?;
let item_version = uints("item_version")?;
// Added after the first stores were written, so absent rather than fatal in
// a file from before 2.0; `coursebank migrate variants` fills it in.
let variant = optional_strings("variant");
let selected = strings("selected")?;
let eliminated = strings("eliminated")?;
let correct = strings("correct")?;
@@ -314,6 +319,7 @@ fn from_batch(batch: &RecordBatch, path: &Path) -> Result<Vec<FlatResponse>> {
item_number: item_number.value(i),
item_ref: item_ref.value(i).to_string(),
item_version: item_version.value(i),
variant: variant.map(|c| c.value(i).to_string()).unwrap_or_default(),
selected: selected.value(i).to_string(),
selected_source: selected_source
.map(|a| a.value(i).to_string())
@@ -366,6 +372,7 @@ mod tests {
item_number: number,
item_ref: "bank::q-a-001".into(),
item_version: 3,
variant: "4c81fa".into(),
selected: "C".into(),
eliminated: String::new(),
correct: "1".into(),
+2
View File
@@ -12,6 +12,7 @@
//! | [`site`] | a Quarto partial and an encrypted bundle | a course page with password-gated solutions |
//! | [`report`] | Markdown and HTML | students, and yourself |
//! | [`lecture`] | Markdown | the reading list on the course website |
//! | [`references`] | Hayagriva, CSL-JSON, BibTeX | Typst, Zotero, LaTeX |
//!
//! [`qti`] and [`typst`] share one rule that is easy to get wrong: a form's answer
//! key must be generated from the same permutation that produced its question
@@ -26,6 +27,7 @@
pub mod lecture;
pub mod practice;
pub mod qti;
pub mod references;
pub mod report;
pub mod site;
pub mod typst;
+2 -1
View File
@@ -374,7 +374,7 @@ fn entry(
///
/// A linked citation when the location has a URL, and a plain one when it does not.
fn heading(reading: &Reading, key: &str, reference: &Reference, style: Style) -> String {
let label = reference.label.as_deref().unwrap_or(key);
let label = reference.label_or(key);
let locator = reading.locator.as_deref().unwrap_or("");
let linked = match reading.resolve_url(reference) {
Some(url) if !locator.is_empty() => format!("[{locator}]({url})"),
@@ -428,6 +428,7 @@ mod tests {
date: None,
unit: None,
slides_url: None,
teaches: Vec::new(),
readings: vec![
Reading {
reference: Some("kuriyan2013molecules".into()),
+15 -23
View File
@@ -30,7 +30,7 @@
use crate::assessment::{AssessmentFile, Form, Placement};
use crate::catalog::Catalog;
use crate::course::{CourseFile, Reference};
use crate::course::CourseFile;
use crate::error::Result;
use crate::item::{Choice, Citation, Item};
use crate::markup;
@@ -289,7 +289,7 @@ fn worksheet_question(
out.push_str("\n\n");
if item.has_options() {
let ordered = ordered_options(item, form, &placement.item);
let ordered = ordered_options(item, placement, form);
for (position, source) in ordered.iter().enumerate() {
out.push_str(&format!(
"{}. {}\n",
@@ -320,7 +320,7 @@ fn solution_question(
out.push_str("\n\n");
if item.has_options() {
let ordered = ordered_options(item, form, &placement.item);
let ordered = ordered_options(item, placement, form);
for (position, source) in ordered.iter().enumerate() {
let mark = if source.correct { " ✓" } else { "" };
let note = source
@@ -427,9 +427,9 @@ fn cite(course: &CourseFile, citation: &Citation) -> String {
let Some(reference) = course.references.get(key) else {
return citation.display();
};
let label = reference.label.as_deref().unwrap_or(key);
let label = reference.label_or(key);
let locator = citation.locator.as_deref().unwrap_or("");
match resolve_url(citation, reference) {
match citation.href(reference) {
Some(url) if !locator.is_empty() => format!("`{label}` [{locator}]({url})"),
Some(url) => format!("`{label}` [{}]({url})", reference.title),
None if !locator.is_empty() => format!("`{label}` {locator}"),
@@ -437,21 +437,6 @@ fn cite(course: &CourseFile, citation: &Citation) -> String {
}
}
/// The URL for a citation: its own `url`, else the reference `base_url` joined with
/// the citation `path`.
fn resolve_url(citation: &Citation, reference: &Reference) -> Option<String> {
if let Some(url) = &citation.url {
return Some(url.clone());
}
let path = citation.path.as_deref()?;
let base = reference.base_url.as_deref()?;
Some(match (base.ends_with('/'), path.starts_with('/')) {
(true, true) => format!("{base}{}", &path[1..]),
(false, false) => format!("{base}/{path}"),
_ => format!("{base}{path}"),
})
}
/// The `## Question N` heading, marking a bonus item.
fn heading(number: usize, placement: &Placement) -> String {
let bonus = if placement.bonus { " (bonus)" } else { "" };
@@ -473,10 +458,11 @@ fn meta_line(placement: &Placement, item: &Item) -> String {
/// Salted with the item's global id, the same value the Typst and QTI exports use,
/// so a worksheet built for form B lists options in the order that form's paper and
/// its Canvas quiz do.
fn ordered_options<'a>(item: &'a Item, form: &Form, uid: &str) -> Vec<&'a Choice> {
select::option_order(form, uid, item.options.len())
fn ordered_options<'a>(item: &'a Item, placement: &Placement, form: &Form) -> Vec<&'a Choice> {
let shown = item.administered(&placement.key, &placement.distractors);
select::option_order(form, &placement.item, shown.len())
.into_iter()
.map(|i| &item.options[i])
.map(|i| shown[i])
.collect()
}
@@ -593,6 +579,9 @@ items:
number: 1,
item: "l11::q-enthalpy-001".into(),
version: None,
stem_digest: None,
distractors: Vec::new(),
variant: None,
fingerprint: None,
points: Some(1.0),
bonus: false,
@@ -608,6 +597,9 @@ items:
number: 2,
item: "l11::q-enthalpy-op-001".into(),
version: None,
stem_digest: None,
distractors: Vec::new(),
variant: None,
fingerprint: None,
points: Some(2.0),
bonus: false,
+19 -3
View File
@@ -476,10 +476,12 @@ pub fn build(catalog: &Catalog, record: &AssessmentFile, opts: &QtiOptions) -> R
if !placement.bonus {
total_points += points;
}
let shown = item.administered(&placement.key, &placement.distractors);
items.push(build_item(
&record.assessment.id,
&placement.item,
item,
&shown,
points,
opts,
));
@@ -564,14 +566,21 @@ pub fn build(catalog: &Catalog, record: &AssessmentFile, opts: &QtiOptions) -> R
/// # Returns
///
/// The element.
fn build_item(assessment_id: &str, uid: &str, item: &Item, points: f64, opts: &QtiOptions) -> Node {
fn build_item(
assessment_id: &str,
uid: &str,
item: &Item,
shown: &[&crate::item::Choice],
points: f64,
opts: &QtiOptions,
) -> Node {
// An open-response item is an essay in Canvas: no choices, graded by hand.
if !item.format.has_options() {
return build_essay_item(assessment_id, uid, item, points, opts);
}
let order = select::option_order(&opts.form, uid, item.options.len());
let ordered: Vec<&crate::item::Choice> = order.iter().map(|i| &item.options[*i]).collect();
let order = select::option_order(&opts.form, uid, shown.len());
let ordered: Vec<&crate::item::Choice> = order.iter().map(|i| shown[*i]).collect();
// Option identifiers are numeric, mirroring Canvas's own exports, and are
// derived from the item id so they survive regeneration.
@@ -1259,6 +1268,7 @@ mod tests {
defense: None,
feedback_student: None,
selection_rate_expected: None,
retired: None,
}
}
@@ -1389,6 +1399,9 @@ items:
number: 1,
item: "b::q-mcq".into(),
version: None,
stem_digest: None,
distractors: Vec::new(),
variant: None,
fingerprint: None,
points: Some(1.0),
bonus: false,
@@ -1404,6 +1417,9 @@ items:
number: 2,
item: "b::q-open".into(),
version: None,
stem_digest: None,
distractors: Vec::new(),
variant: None,
fingerprint: None,
points: Some(2.0),
bonus: false,
+498
View File
@@ -0,0 +1,498 @@
// SPDX-License-Identifier: Prosperity-3.0.0
// Copyright Scientific Computing Studio
// Source: https://git.scient.ing/education/coursebank
//! The bibliography, in the formats other tools read.
//!
//! `references.yaml` is the authoritative copy, and it is shaped for a reading
//! list: it carries a `label` that reports print, and a `role` saying whether
//! the course requires the work or offers it as background. No general citation
//! format has either field, which is why the course keeps its own.
//!
//! What the other formats are for is everything downstream of the reading list:
//!
//! | Format | Read by | Gets you |
//! |:--|:--|:--|
//! | [`Format::Hayagriva`] | Typst | real citations in a printed exam or report |
//! | [`Format::CslJson`] | Zotero, Pandoc, CSL processors | a bibliography in any style |
//! | [`Format::Bibtex`] | LaTeX, most reference managers | the lowest common denominator |
//!
//! All three are generated. The argument is the same one the lecture reading
//! list makes: two copies of a citation drift within a term, and one copy plus a
//! build step does not.
//!
//! # What does not survive the trip
//!
//! `label`, `role`, `base_url`, and `note` have nowhere to go in any of the
//! three, so they stay behind. That is the reason this is an export rather than
//! a migration: the course file is not recoverable from its own bibliography
//! export, and nothing reads these files back in.
use std::collections::BTreeMap;
use serde::Serialize;
use serde_json::{Value, json};
use crate::course::{CourseFile, Reference, ReferenceKind};
use crate::error::Result;
use crate::yaml;
/// Which citation format to write.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Format {
/// Hayagriva YAML, which Typst's `#bibliography` reads natively.
Hayagriva,
/// CSL-JSON, which Zotero, Pandoc, and every CSL processor read.
CslJson,
/// BibTeX, for LaTeX and for reference managers that read nothing else.
Bibtex,
}
impl Format {
/// The conventional file extension.
pub fn extension(self) -> &'static str {
match self {
Format::Hayagriva => "yml",
Format::CslJson => "json",
Format::Bibtex => "bib",
}
}
/// The name used on the command line.
pub fn label(self) -> &'static str {
match self {
Format::Hayagriva => "hayagriva",
Format::CslJson => "csl-json",
Format::Bibtex => "bibtex",
}
}
}
/// Renders a course's bibliography.
///
/// # Arguments
///
/// * `course` - the loaded course.
/// * `format` - which format to write.
///
/// # Returns
///
/// The document, keyed or ordered by citation key so the output is stable
/// between runs.
///
/// # Errors
///
/// Returns [`crate::error::Error::Other`] if the intermediate structure cannot
/// be serialized.
pub fn render(course: &CourseFile, format: Format) -> Result<String> {
match format {
Format::Hayagriva => hayagriva(course),
Format::CslJson => csl_json(course),
Format::Bibtex => Ok(bibtex(course)),
}
}
// --- Hayagriva ---
/// One Hayagriva entry.
#[derive(Debug, Serialize)]
struct Entry {
#[serde(rename = "type")]
kind: &'static str,
title: String,
#[serde(skip_serializing_if = "Vec::is_empty")]
author: Vec<String>,
#[serde(skip_serializing_if = "Option::is_none")]
date: Option<u32>,
#[serde(skip_serializing_if = "Option::is_none")]
edition: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
publisher: Option<String>,
#[serde(rename = "page-range", skip_serializing_if = "Option::is_none")]
page_range: Option<String>,
#[serde(rename = "serial-number", skip_serializing_if = "BTreeMap::is_empty")]
serial_number: BTreeMap<&'static str, String>,
#[serde(skip_serializing_if = "Option::is_none")]
url: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
parent: Option<Parent>,
}
/// The container a Hayagriva entry sits inside.
#[derive(Debug, Serialize)]
struct Parent {
#[serde(rename = "type")]
kind: &'static str,
title: String,
#[serde(skip_serializing_if = "Option::is_none")]
volume: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
issue: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
publisher: Option<String>,
}
/// Renders the bibliography as Hayagriva YAML.
fn hayagriva(course: &CourseFile) -> Result<String> {
let entries: BTreeMap<&String, Entry> = course
.references
.iter()
.map(|(key, reference)| (key, entry(reference)))
.collect();
yaml::to_string(&entries)
}
/// Maps one reference onto a Hayagriva entry.
fn entry(reference: &Reference) -> Entry {
let mut serial: BTreeMap<&'static str, String> = BTreeMap::new();
for (field, value) in [
("doi", &reference.doi),
("isbn", &reference.isbn),
("arxiv", &reference.arxiv),
("pmid", &reference.pmid),
("pmcid", &reference.pmcid),
] {
if let Some(value) = value {
serial.insert(field, value.clone());
}
}
// A chapter sits in a book and an article sits in a periodical, and
// Hayagriva wants that said with a parent rather than a flat field.
let parent = reference
.container
.as_ref()
.map(|title| match reference.kind {
ReferenceKind::Chapter => Parent {
kind: "book",
title: title.clone(),
volume: reference.volume.clone(),
issue: None,
publisher: reference.publisher.clone(),
},
_ => Parent {
kind: "periodical",
title: title.clone(),
volume: reference.volume.clone(),
issue: reference.issue.clone(),
publisher: None,
},
});
Entry {
kind: hayagriva_kind(reference.kind),
title: reference.title.clone(),
author: reference.authors.clone(),
date: reference.year,
edition: reference.edition.clone(),
// The publisher belongs to the container when there is one.
publisher: if parent.is_some() {
None
} else {
reference.publisher.clone()
},
page_range: reference.pages.clone(),
serial_number: serial,
url: reference.url.clone(),
parent,
}
}
/// The Hayagriva entry type for a course reference kind.
fn hayagriva_kind(kind: ReferenceKind) -> &'static str {
match kind {
ReferenceKind::Book => "book",
ReferenceKind::Chapter => "chapter",
// Hayagriva has no preprint type. An article with no periodical parent
// is what a preprint is anyway.
ReferenceKind::Article | ReferenceKind::Preprint => "article",
ReferenceKind::Thesis => "thesis",
ReferenceKind::Website => "web",
ReferenceKind::Software | ReferenceKind::Dataset => "repository",
ReferenceKind::Video => "video",
ReferenceKind::Other => "misc",
}
}
// --- CSL-JSON ---
/// Renders the bibliography as CSL-JSON.
fn csl_json(course: &CourseFile) -> Result<String> {
let items: Vec<Value> = course
.references
.iter()
.map(|(key, reference)| csl_item(key, reference))
.collect();
serde_json::to_string_pretty(&items)
.map(|text| format!("{text}\n"))
.map_err(crate::error::Error::other)
}
/// One CSL-JSON item.
fn csl_item(key: &str, reference: &Reference) -> Value {
let mut item = json!({
"id": key,
"type": csl_kind(reference.kind),
"title": reference.title,
});
let map = item.as_object_mut().expect("built from a JSON object");
if !reference.authors.is_empty() {
map.insert(
"author".to_string(),
Value::Array(reference.authors.iter().map(|a| csl_name(a)).collect()),
);
}
if let Some(year) = reference.year {
map.insert("issued".to_string(), json!({ "date-parts": [[year]] }));
}
for (field, value) in [
("container-title", &reference.container),
("publisher", &reference.publisher),
("volume", &reference.volume),
("issue", &reference.issue),
("page", &reference.pages),
("edition", &reference.edition),
("DOI", &reference.doi),
("ISBN", &reference.isbn),
("PMID", &reference.pmid),
("PMCID", &reference.pmcid),
("URL", &reference.url),
] {
if let Some(value) = value {
map.insert(field.to_string(), Value::String(value.clone()));
}
}
item
}
/// Splits `Family, Given` into a CSL name, or keeps it whole.
///
/// A name with no comma is not a name this code can take apart — an
/// organization, or a single mononym — so it goes in `literal`, which is what
/// CSL has the field for.
fn csl_name(author: &str) -> Value {
match author.split_once(',') {
Some((family, given)) => json!({
"family": family.trim(),
"given": given.trim(),
}),
None => json!({ "literal": author.trim() }),
}
}
/// The CSL type for a course reference kind.
fn csl_kind(kind: ReferenceKind) -> &'static str {
match kind {
ReferenceKind::Book => "book",
ReferenceKind::Chapter => "chapter",
ReferenceKind::Article => "article-journal",
ReferenceKind::Preprint => "article",
ReferenceKind::Thesis => "thesis",
ReferenceKind::Website => "webpage",
ReferenceKind::Software => "software",
ReferenceKind::Dataset => "dataset",
ReferenceKind::Video => "motion_picture",
ReferenceKind::Other => "document",
}
}
// --- BibTeX ---
/// Renders the bibliography as BibTeX.
fn bibtex(course: &CourseFile) -> String {
let mut out = String::new();
for (key, reference) in &course.references {
out.push_str(&bibtex_entry(key, reference));
out.push('\n');
}
out
}
/// One BibTeX entry.
fn bibtex_entry(key: &str, reference: &Reference) -> String {
let mut fields: Vec<(&str, String)> = vec![("title", reference.title.clone())];
if !reference.authors.is_empty() {
fields.push(("author", reference.authors.join(" and ")));
}
if let Some(year) = reference.year {
fields.push(("year", year.to_string()));
}
if let Some(container) = &reference.container {
let field = match reference.kind {
ReferenceKind::Chapter => "booktitle",
_ => "journal",
};
fields.push((field, container.clone()));
}
for (field, value) in [
("volume", &reference.volume),
("number", &reference.issue),
("edition", &reference.edition),
("publisher", &reference.publisher),
("doi", &reference.doi),
("isbn", &reference.isbn),
("url", &reference.url),
("note", &reference.note),
] {
if let Some(value) = value {
fields.push((field, value.clone()));
}
}
if let Some(pages) = &reference.pages {
fields.push(("pages", en_dash(pages)));
}
let body: String = fields
.iter()
.map(|(field, value)| format!(" {field} = {{{}}},\n", escape_tex(value)))
.collect();
format!("@{}{{{key},\n{body}}}\n", bibtex_kind(reference.kind))
}
/// The BibTeX entry type for a course reference kind.
fn bibtex_kind(kind: ReferenceKind) -> &'static str {
match kind {
ReferenceKind::Book => "book",
ReferenceKind::Chapter => "incollection",
ReferenceKind::Article => "article",
ReferenceKind::Thesis => "phdthesis",
ReferenceKind::Website => "online",
// BibTeX proper has nothing for these. `misc` with a `note` is what
// every style guide says to do, and biblatex users can convert.
ReferenceKind::Preprint
| ReferenceKind::Software
| ReferenceKind::Dataset
| ReferenceKind::Video
| ReferenceKind::Other => "misc",
}
}
/// Escapes the characters BibTeX treats as syntax.
///
/// Deliberately short: a publisher called `John Wiley & Sons` is the case that
/// actually occurs, and escaping more than this risks mangling the `$...$` in a
/// title that carries real mathematics.
fn escape_tex(value: &str) -> String {
value.replace('&', "\\&").replace('%', "\\%")
}
/// Turns a hyphenated page range into the en dash BibTeX expects.
fn en_dash(pages: &str) -> String {
let parts: Vec<&str> = pages.split('-').collect();
if parts.len() == 2 && parts.iter().all(|p| !p.is_empty()) {
return format!("{}--{}", parts[0].trim(), parts[1].trim());
}
pages.to_string()
}
#[cfg(test)]
mod tests {
use super::*;
use crate::course::ReferenceRole;
fn course() -> CourseFile {
let mut c = CourseFile::skeleton("BIOSC 1540", "Computational Biology", "2026f");
c.references.insert(
"ismail2023bioinformatics".into(),
Reference {
label: Some("IBB".into()),
kind: ReferenceKind::Book,
role: ReferenceRole::Supplemental,
title: "Bioinformatics: A practical guide".into(),
authors: vec!["Ismail, H. D.".into()],
year: Some(2023),
publisher: Some("CRC Press".into()),
base_url: Some("https://library.example.org/ismail2023/".into()),
isbn: Some("9781032366423".into()),
..Reference::default()
},
);
c.references.insert(
"altschul1990basic".into(),
Reference {
label: Some("BLAST".into()),
kind: ReferenceKind::Article,
role: ReferenceRole::Required,
title: "Basic local alignment search tool".into(),
authors: vec![
"Altschul, S. F.".into(),
"Gish, W.".into(),
"Wiley & Sons".into(),
],
year: Some(1990),
container: Some("Journal of Molecular Biology".into()),
volume: Some("215".into()),
issue: Some("3".into()),
pages: Some("403-410".into()),
doi: Some("10.1016/S0022-2836(05)80360-2".into()),
pmid: Some("2231712".into()),
..Reference::default()
},
);
c
}
#[test]
fn hayagriva_nests_an_article_under_its_periodical() {
let out = render(&course(), Format::Hayagriva).unwrap();
assert!(out.contains("altschul1990basic:"), "{out}");
assert!(out.contains("type: article"), "{out}");
assert!(out.contains("type: periodical"), "{out}");
assert!(out.contains("Journal of Molecular Biology"), "{out}");
assert!(out.contains("page-range: 403-410"), "{out}");
assert!(out.contains("doi: 10.1016/S0022-2836(05)80360-2"), "{out}");
// Parses back as YAML, which is what Typst will do to it.
let back: serde_yaml_ng::Value = serde_yaml_ng::from_str(&out).unwrap();
assert!(back.get("ismail2023bioinformatics").is_some());
}
#[test]
fn hayagriva_keeps_a_books_publisher_on_the_book() {
let out = render(&course(), Format::Hayagriva).unwrap();
let parsed: serde_yaml_ng::Value = serde_yaml_ng::from_str(&out).unwrap();
let book = parsed.get("ismail2023bioinformatics").unwrap();
assert_eq!(book.get("publisher").unwrap().as_str(), Some("CRC Press"));
assert!(book.get("parent").is_none());
// `label`, `role`, and `base_url` have nowhere to go and stay behind.
assert!(book.get("label").is_none());
assert!(book.get("base_url").is_none());
}
#[test]
fn csl_json_splits_names_and_keeps_organizations_whole() {
let out = render(&course(), Format::CslJson).unwrap();
let items: Vec<Value> = serde_json::from_str(&out).unwrap();
let article = items
.iter()
.find(|i| i["id"] == "altschul1990basic")
.unwrap();
assert_eq!(article["type"], "article-journal");
assert_eq!(article["author"][0]["family"], "Altschul");
assert_eq!(article["author"][0]["given"], "S. F.");
assert_eq!(article["author"][2]["literal"], "Wiley & Sons");
assert_eq!(article["issued"]["date-parts"][0][0], 1990);
assert_eq!(article["DOI"], "10.1016/S0022-2836(05)80360-2");
}
#[test]
fn bibtex_escapes_ampersands_and_dashes_a_page_range() {
let out = render(&course(), Format::Bibtex).unwrap();
assert!(out.contains("@article{altschul1990basic,"), "{out}");
assert!(
out.contains("journal = {Journal of Molecular Biology},"),
"{out}"
);
assert!(out.contains("pages = {403--410},"), "{out}");
assert!(out.contains("Wiley \\& Sons"), "{out}");
assert!(out.contains("@book{ismail2023bioinformatics,"), "{out}");
}
#[test]
fn a_course_with_no_bibliography_renders_empty_rather_than_failing() {
let mut c = course();
c.references.clear();
assert_eq!(render(&c, Format::Bibtex).unwrap(), "");
assert_eq!(render(&c, Format::CslJson).unwrap().trim(), "[]");
}
}
+15 -22
View File
@@ -34,7 +34,7 @@
use crate::assessment::{AssessmentFile, Form, Placement};
use crate::catalog::Catalog;
use crate::course::{CourseFile, Reference};
use crate::course::CourseFile;
use crate::error::{Error, Result};
use crate::item::{Choice, Citation, Item, Solution};
use crate::markup;
@@ -226,12 +226,13 @@ fn question_block(
if item.has_options() {
b.push_str(":::: {.q-choices}\n");
let order = select::option_order(form, &placement.item, item.options.len());
let shown = item.administered(&placement.key, &placement.distractors);
let order = select::option_order(form, &placement.item, shown.len());
for (position, &source) in order.iter().enumerate() {
b.push_str(&format!(
"{}. {}\n",
position + 1,
markup::to_markdown(&item.options[source].text)
markup::to_markdown(&shown[source].text)
));
}
b.push_str("::::\n\n");
@@ -295,11 +296,12 @@ fn fragment(
/// A single-best-answer or multiple-response fragment: the key, the model answer,
/// the explanation, then per-distractor feedback.
fn choice_fragment(course: &CourseFile, placement: &Placement, item: &Item, form: &Form) -> String {
let order = select::option_order(form, &placement.item, item.options.len());
let shown = item.administered(&placement.key, &placement.distractors);
let order = select::option_order(form, &placement.item, shown.len());
let printed: Vec<(usize, &Choice)> = order
.iter()
.enumerate()
.map(|(position, &source)| (position, &item.options[source]))
.map(|(position, &source)| (position, shown[source]))
.collect();
let mut out = String::new();
@@ -472,7 +474,7 @@ fn cite_html(course: &CourseFile, citation: &Citation) -> String {
let Some(reference) = course.references.get(key) else {
return markup::escape_html(&citation.display());
};
let label = reference.label.as_deref().unwrap_or(key);
let label = reference.label_or(key);
let locator = citation.locator.as_deref().unwrap_or("");
let body = if locator.is_empty() {
markup::escape_html(label)
@@ -483,27 +485,12 @@ fn cite_html(course: &CourseFile, citation: &Citation) -> String {
markup::escape_html(locator)
)
};
match resolve_url(citation, reference) {
match citation.href(reference) {
Some(url) => format!("<a href=\"{}\">{body}</a>", markup::escape_html(&url)),
None => body,
}
}
/// Resolves a citation's link, from an explicit URL or a path joined to the
/// reference's base URL.
fn resolve_url(citation: &Citation, reference: &Reference) -> Option<String> {
if let Some(url) = &citation.url {
return Some(url.clone());
}
let path = citation.path.as_deref()?;
let base = reference.base_url.as_deref()?;
Some(match (base.ends_with('/'), path.starts_with('/')) {
(true, true) => format!("{base}{}", &path[1..]),
(false, false) => format!("{base}/{path}"),
_ => format!("{base}{path}"),
})
}
// --- math-aware markup ---
/// One run of source text, split on math delimiters.
@@ -858,6 +845,9 @@ items:
number: 1,
item: "b::q-mcq".into(),
version: None,
stem_digest: None,
distractors: Vec::new(),
variant: None,
fingerprint: None,
points: Some(1.0),
bonus: false,
@@ -873,6 +863,9 @@ items:
number: 2,
item: "b::q-open".into(),
version: None,
stem_digest: None,
distractors: Vec::new(),
variant: None,
fingerprint: None,
points: Some(2.0),
bonus: false,
+7 -1
View File
@@ -431,6 +431,9 @@ mod tests {
number: 1,
item: "b::q-1".into(),
version: None,
stem_digest: None,
distractors: Vec::new(),
variant: None,
fingerprint: None,
points: None,
bonus: false,
@@ -446,6 +449,9 @@ mod tests {
number: 2,
item: "b::q-2".into(),
version: None,
stem_digest: None,
distractors: Vec::new(),
variant: None,
fingerprint: None,
points: None,
bonus: false,
@@ -464,6 +470,6 @@ mod tests {
.filter(|p| p.was_printed())
.map(|p| p.number)
.collect();
assert_eq!(printable, vec![2]);
assert_eq!(printable, vec![1, 2]);
}
}
+1
View File
@@ -1281,6 +1281,7 @@ mod tests {
blueprint: Vec::new(),
patterns: Vec::new(),
warnings: Vec::new(),
dropped_detail: Vec::new(),
}
}
+3 -2
View File
@@ -411,11 +411,12 @@ pub fn build(
_ => (None, None),
};
let order = select::option_order(form, &placement.item, item.options.len());
let shown = item.administered(&placement.key, &placement.distractors);
let order = select::option_order(form, &placement.item, shown.len());
let options: Vec<Opt> = order
.iter()
.enumerate()
.map(|(position, source_index)| option(&item.options[*source_index], position, config))
.map(|(position, source_index)| option(shown[*source_index], position, config))
.collect();
let key = if config.reveal.shows_key() {
+17 -5
View File
@@ -18,12 +18,19 @@
//!
//! | File | Holds | Written by |
//! |:--|:--|:--|
//! | `course.yaml` | identity, policy, objectives, lectures | you |
//! | `course.yaml` | identity, policy, units | you |
//! | `references.yaml` | the works the course cites | you |
//! | `lectures/*.yaml` | one lecture: readings, and what it teaches | you |
//! | `objectives/*.yaml` | one objective and its learning targets | you |
//! | `banks/*.yaml` | items, with design intent and pooled statistics | you, then `calibrate` |
//! | `assessments/*.yaml` | what was given, to whom, when | `assemble`, then you |
//! | `data/*.parquet` | one row per student per item | `ingest` |
//!
//! Three of the four are hand-editable YAML meant to be reviewed in a pull request.
//! The first four are one course file split by subject; `coursebank course build`
//! prints the merged result, and a course that keeps everything in `course.yaml`
//! still loads unchanged. See [`course::fragment`].
//!
//! Most of these are hand-editable YAML meant to be reviewed in a pull request.
//! Only the response data is machine-only, and it is stored in an open columnar
//! format so pandas, polars, R, and DuckDB can all read it without this tool.
//!
@@ -95,12 +102,17 @@ pub mod data;
pub mod error;
pub mod export;
pub mod guide;
pub mod migrate;
pub mod model;
pub mod util;
pub use util::{date, hash, markup, rng, yaml, zipfile};
pub use util::{citation, date, hash, markup, rng, yaml, zipfile};
pub use model::{assessment, bank, catalog, course, history, item, layout, seal, taxonomy};
pub use model::{
assessment, bank, calibration, catalog, course, history, item, layout, seal, taxonomy,
};
pub use course::fragment;
pub use authoring::{jsonschema, lint, select};
@@ -110,7 +122,7 @@ pub use data::{canvas, decode, gradescope, intake, responses, store};
pub use analysis::{calibrate, classical, diagnostic, irt, students};
pub use export::site;
pub use export::{lecture, practice, qti, report, typst};
pub use export::{lecture, practice, qti, references, report, typst};
pub use catalog::Catalog;
pub use course::{CourseFile, SCHEMA_VERSION};
+2590
View File
File diff suppressed because it is too large Load Diff
+4 -1
View File
@@ -12,7 +12,9 @@
//! ```text
//! taxonomy levels, cognitive processes, error types, status, flags
//! │
//! course course.yaml: identity, policy, objectives, lectures, stimuli
//! course identity, policy, objectives, lectures, stimuli
//! │ └─ course::fragment merges course.yaml, references.yaml,
//! │ lectures/*.yaml, objectives/*.yaml
//! │
//! item one question: stem, options, design intent, calibration
//! │
@@ -26,6 +28,7 @@
pub mod assessment;
pub mod bank;
pub mod calibration;
pub mod catalog;
pub mod course;
pub mod history;
+60 -2
View File
@@ -248,11 +248,28 @@ pub struct Placement {
/// Printed question number. This is the join key to grading exports, which
/// is the entire reason this record exists.
pub number: u32,
/// The item's global id, `bank::item`.
/// The item's id, which names it course-wide. A pre-2.0 `bank::item`
/// value still resolves; `coursebank migrate ids` rewrites it.
pub item: String,
/// The item version used.
#[serde(default, skip_serializing_if = "Option::is_none")]
/// Retained only so a pre-2.0 record still loads. Ignored.
///
/// What it was for — knowing whether the item has changed since this
/// administration — is [`Placement::stem_digest`] and
/// [`Placement::fingerprint`], which say *what* changed rather than that
/// something did.
#[serde(default, skip_serializing)]
pub version: Option<u32>,
/// The stem's digest as administered. See [`crate::item::Item::stem_digest`].
///
/// Distinct from `fingerprint`, which covers the options too. A changed
/// fingerprint means the pooled statistics describe an older wording; a
/// changed stem digest means this is no longer the same question, which
/// [`crate::catalog::Catalog::validate_record`] treats as an error rather
/// than a note.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub stem_digest: Option<String>,
/// The content fingerprint as used, so later edits are detectable.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub fingerprint: Option<String>,
@@ -265,6 +282,26 @@ pub struct Placement {
/// The keyed letters as administered.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub key: Vec<String>,
/// The option ids offered alongside the key.
///
/// Resolved when the assessment is assembled and written out explicitly,
/// never sampled at export time. A blueprint may ask for a draw; the record
/// holds what was drawn. Otherwise a bank edit between assembling and
/// printing silently changes the paper, and the key printed on Tuesday
/// disagrees with the one printed on Wednesday.
///
/// Empty means the whole pool, which is what every pre-2.0 record meant.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub distractors: Vec<String>,
/// The digest of the item as this administration showed it.
///
/// See [`crate::item::Item::variant_digest`]. The key statistics pool on:
/// two administrations of one stem with different distractors are two
/// items, and averaging them is averaging different questions.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub variant: Option<String>,
/// The level as administered, denormalized so a record reads standalone.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub level: Option<Level>,
@@ -340,6 +377,27 @@ pub enum DropStyle {
}
impl Placement {
/// The variant this placement administered.
///
/// Recorded when the assessment was assembled; derived from the option set
/// otherwise, which is what makes every record written before 2.0 poolable
/// without being rewritten. A pre-2.0 placement names its key and no
/// distractors, which means the whole pool — a well-defined option set, and
/// so a well-defined variant.
///
/// # Arguments
///
/// * `item` - the item this placement names.
///
/// # Returns
///
/// The digest.
pub fn variant_of(&self, item: &crate::item::Item) -> String {
self.variant
.clone()
.unwrap_or_else(|| item.variant_digest(&self.key, &self.distractors))
}
/// Whether this placement was dropped by crediting every option.
///
/// # Returns
+80 -92
View File
@@ -352,8 +352,10 @@ fn validate_item(
if it.stem.trim().is_empty() {
issues.push("empty stem".into());
}
if it.version == 0 {
issues.push("version must be at least 1".into());
if let Some(replaced) = &it.supersedes {
if *replaced == it.id {
issues.push("supersedes names this item".into());
}
}
// --- options ----
@@ -379,21 +381,37 @@ fn validate_item(
if o.text.trim().is_empty() {
issues.push(format!("option {pos}: empty text"));
}
let letter_ok = o.id.len() == 1
&& o.id
.chars()
.next()
.map(|c| c.is_ascii_uppercase() && c <= 'H')
.unwrap_or(false);
if !letter_ok {
// Two forms are accepted: the 2.0 name, and the letter that preceded
// it. A bank migrates when `coursebank migrate options` is run on it,
// not when the tool is upgraded, so a course mid-migration still loads.
let slug_ok = o.id.strip_prefix("o-").is_some_and(|rest| {
!rest.is_empty()
&& !rest.starts_with('-')
&& !rest.ends_with('-')
&& !rest.contains("--")
&& rest
.chars()
.all(|c| c.is_ascii_lowercase() || c.is_ascii_digit() || c == '-')
});
if !slug_ok && !Item::is_legacy_option_id(&o.id) {
issues.push(format!(
"option {pos}: id `{}` must be a single letter A through H",
"option {pos}: id `{}` is neither a name such as `o-fourth-line` nor a pre-2.0 \
letter A through H",
o.id
));
}
if seen.contains(&o.id.as_str()) {
issues.push(format!("option {pos}: duplicate option id `{}`", o.id));
}
if let Some(retirement) = &o.retired {
if retirement.reason.trim().is_empty() {
issues.push(format!(
"option {pos}: retired without a reason. The reason is the finding — what \
the option did or failed to do — and it is the only part of a retirement \
that is worth anything later."
));
}
}
seen.push(&o.id);
let credit = o.credit();
@@ -433,14 +451,27 @@ fn validate_item(
}
// --- key ---
// Counted over the pool that can still be drawn: a retired option is a
// record, not an offer.
let (live_keys, live_distractors) = it.pool();
let keys = it.key_indices();
match it.format {
Format::SingleBestAnswer => {
if keys.len() != 1 {
issues.push(format!(
"single_best_answer needs exactly one keyed option, has {}",
keys.len()
));
// Several defensible keys is a pool, not a bug — it is what lets you
// test whether "fourth" or "last of the four" is doing the work.
// Exactly one of them reaches a student, and that is the
// placement's business: see
// [`crate::catalog::Catalog::validate_record`].
if live_keys.is_empty() {
issues.push("single_best_answer needs at least one keyed option".into());
}
if live_distractors.is_empty() {
let retired = it.options.iter().any(|o| o.retired.is_some());
issues.push(if retired {
"every distractor is retired, so nothing can be drawn against the key".into()
} else {
"has no option that is not keyed correct, so it asks nothing".to_string()
});
}
}
Format::MultipleResponse => {
@@ -499,57 +530,12 @@ fn validate_item(
}
// --- calibration plausibility ------
if let Some(c) = &it.calibration {
if let Some(p) = c.p_value {
if !(0.0..=1.0).contains(&p) {
issues.push(format!(
"calibration.p_value must be between 0 and 1, got {p}"
));
}
}
if let Some(r) = c.point_biserial {
if !(-1.0..=1.0).contains(&r) {
issues.push(format!(
"calibration.point_biserial must be between -1 and 1, got {r}"
));
}
}
for letter in c.option_stats.keys() {
if it.option(letter).is_none() {
issues.push(format!(
"calibration.option_stats has `{letter}`, which is not an option of this item"
));
}
}
if let Some(irt) = &c.irt {
if irt.a <= 0.0 {
issues.push(format!("calibration.irt.a must be positive, got {}", irt.a));
}
if let Some(cp) = irt.c {
if !(0.0..1.0).contains(&cp) {
issues.push(format!("calibration.irt.c must be in [0, 1), got {cp}"));
}
}
}
}
// --- history must be coherent ------
let mut last_version = 0u32;
for (i, h) in it.history.iter().enumerate() {
if h.version <= last_version {
issues.push(format!(
"history entry {} has version {} which does not increase",
i + 1,
h.version
));
}
last_version = h.version;
}
if !it.history.is_empty() && last_version > it.version {
issues.push(format!(
"history records version {last_version} but the item says version {}",
it.version
));
if it.calibration.is_some() {
issues.push(
"has a `calibration:` block, but statistics live in analysis/calibration.yaml since \
2.0. A bank's diff should be a change of intent, not the output of a grading run."
.to_string(),
);
}
// --- retirement -----
@@ -790,20 +776,43 @@ mod tests {
options:
- { id: A, text: a, correct: true }
- { id: B, text: b, correct: true }
- id: q-a-003
status: draft
level: 1
format: single_best_answer
stem: s
options:
- { id: o-key-one, text: a, correct: true }
- { id: o-key-two, text: b, correct: true }
- { id: o-wrong-one, text: c }
- { id: o-wrong-two, text: d }
"#,
);
let issues = b.validate(None);
// No key at all is still a bank problem: nothing can be drawn from it.
assert!(
issues
.iter()
.any(|i| i.contains("exactly one keyed option"))
.any(|i| i.starts_with("q-a-001") && i.contains("at least one keyed option")),
"{issues:?}"
);
assert_eq!(
// Keying every option is still a bank problem, for the older reason: a
// question with nothing to choose against asks nothing.
assert!(
issues
.iter()
.filter(|i| i.contains("exactly one keyed option"))
.count(),
2
.any(|i| i.starts_with("q-a-002") && i.contains("asks nothing")),
"{issues:?}"
);
// Two defensible keys alongside real distractors is not a problem. It
// is the pool doing its job — it is what lets you test whether the
// wording of the key is what students are answering. Exactly one of
// them reaches a student, and that is checked against the placement
// that administers it, in `Catalog::validate_placement`.
assert!(
!issues.iter().any(|i| i.starts_with("q-a-003")
&& (i.contains("keyed option") || i.contains("asks nothing"))),
"{issues:?}"
);
}
@@ -1027,27 +1036,6 @@ learning_targets:
);
}
#[test]
fn history_versions_must_increase() {
let b = bank(
r#"
- id: q-a-001
version: 2
status: draft
level: 1
stem: s
options:
- { id: A, text: a, correct: true }
- { id: B, text: b }
history:
- { version: 2, date: 2026-01-01, change: second }
- { version: 1, date: 2026-01-02, change: first }
"#,
);
let issues = b.validate(None);
assert!(issues.iter().any(|i| i.contains("does not increase")));
}
#[test]
fn level_counts_exclude_drafts_and_bonuses() {
let b = bank(
+838
View File
@@ -0,0 +1,838 @@
// SPDX-License-Identifier: Prosperity-3.0.0
// Copyright Scientific Computing Studio
// Source: https://git.scient.ing/education/coursebank
//! Where the statistics live, which is not in the bank.
//!
//! A bank file is a reviewed artifact: someone wrote the question, someone
//! argued about the distractors, and the diff on it should be a change of
//! intent. Statistics are neither reviewed nor intended — they are what
//! happened — and writing them back into the bank means every grading run
//! produces a diff on a file whose history is supposed to be about wording.
//!
//! So the evidence lives in `analysis/`, in two kinds of file:
//!
//! | File | Format | Rewritten? | Holds |
//! |:--|:--|:--|:--|
//! | `analysis/administrations/<id>-items.csv` | CSV | never | one row per question |
//! | `analysis/administrations/<id>-options.csv` | CSV | never | one row per question and option |
//! | `analysis/calibration.yaml` | YAML | by `calibrate` | the pooled per-item view |
//!
//! The format follows the shape. An administration record is a table — fixed
//! columns, one row per question, machine-written, never hand-edited — so it is
//! CSV: one line per item rather than fifteen, which diffs better, and it loads
//! straight into pandas or DuckDB, which is much of the point of committing it.
//! The pooled view is not a table. It is three levels deep, item to variant to
//! option history, with fitted parameters and variable-length lists, and as CSV
//! that would be three files joined by keys — a relational schema for the one
//! file a person actually reads in a pull request. That stays YAML.
//!
//! Neither is Parquet, and the reason is the review workflow: a binary file
//! shows nothing in a diff and cannot be merged. Parquet is right for `data/`
//! precisely because that is bulk, ignored, and never reviewed.
//!
//! An administration file is written once and not touched again, for the same
//! reason a seal is not: it is a record of a thing that happened on a day. The
//! calibration file is the accepted rollup — what `lint` compares your
//! predictions against, and what a report reads — and `calibrate` proposes
//! changes to it as a diff you review before committing.
//!
//! Both are meant to be committed. Neither can carry student data, and that is
//! a property of the types rather than a promise: there is no field for a
//! student key, an identifier, a section, or an ability estimate, and the
//! structures reject unknown keys, so a file carrying one fails to load rather
//! than being quietly accepted. Everything per-person stays in `data/`, which
//! is what your `.gitignore` is for.
//!
//! # Linking back to the bank
//!
//! By item id, which since 2.0 names the item course-wide and has no file name
//! in it, and by variant digest, which says which option set the numbers
//! describe. A record also carries the stem digest it was measured against, so
//! [`CalibrationFile::validate`] can say that an item has been reworded since —
//! the statistics then describe a question that no longer exists under that id.
use std::collections::BTreeMap;
use std::path::{Path, PathBuf};
use serde::{Deserialize, Serialize};
use crate::course::SCHEMA_VERSION;
use crate::date::Date;
use crate::error::{Error, Result};
use crate::item::{Calibration, IrtModel, IrtParams, OptionStat};
use crate::taxonomy::Flag;
use crate::yaml;
/// The file name of the pooled calibration store, under `analysis/`.
pub const CALIBRATION_FILE: &str = "calibration.yaml";
/// The pooled per-item statistics: `analysis/calibration.yaml`.
///
/// Keyed by item id. This is the file `calibrate` rewrites and the one
/// everything else reads; [`crate::catalog::Catalog::load`] fills each item's
/// in-memory calibration from it, so nothing downstream has to know where the
/// numbers came from.
#[derive(Debug, Clone, Default, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct CalibrationFile {
/// Schema version.
#[serde(
default = "default_version",
deserialize_with = "yaml::flexible_string"
)]
pub schema_version: String,
/// One entry per calibrated item, by item id.
#[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
pub items: BTreeMap<String, Calibration>,
}
impl CalibrationFile {
/// Loads the store, or an empty one when the file does not exist yet.
///
/// Absence is not an error: a course that has not graded anything has no
/// statistics, and every command that reads them has to work anyway.
///
/// # Arguments
///
/// * `path` - the file, usually `analysis/calibration.yaml`.
///
/// # Returns
///
/// The store.
///
/// # Errors
///
/// Returns [`Error::Yaml`] when the file exists and does not parse.
pub fn load(path: &Path) -> Result<CalibrationFile> {
if !path.is_file() {
return Ok(CalibrationFile::default());
}
yaml::read(path)
}
/// Writes the store.
///
/// # Arguments
///
/// * `path` - the destination.
///
/// # Errors
///
/// Returns [`Error::Io`] on a write failure.
pub fn save(&self, path: &Path) -> Result<()> {
yaml::write(path, self)
}
/// The calibration recorded for one item.
///
/// # Arguments
///
/// * `item` - the item id.
///
/// # Returns
///
/// The record, or `None` when the item has never been calibrated.
pub fn get(&self, item: &str) -> Option<&Calibration> {
self.items.get(item)
}
/// Checks the store against the bank it describes.
///
/// The checks that matter for a file kept apart from what it refers to: a
/// record for an item that no longer exists, an option id the item does not
/// have, and — the one worth having — statistics measured against a stem
/// that has since been reworded, which since 2.0 means they describe a
/// different question wearing the same id.
///
/// # Arguments
///
/// * `catalog` - the loaded course.
///
/// # Returns
///
/// One message per problem, empty when the store agrees with the bank.
pub fn validate(&self, catalog: &crate::catalog::Catalog) -> Vec<String> {
let mut issues = Vec::new();
for (id, calibration) in &self.items {
let Some(entry) = catalog.get(id) else {
issues.push(format!(
"calibration for `{id}`: no such item. Statistics outlive an item only if \
it is retired, not deleted — a retired item keeps its id so its numbers \
still mean something."
));
continue;
};
let item = &entry.item;
for variant in &calibration.variants {
for option in variant.option_stats.keys() {
if item.option(option).is_none() {
issues.push(format!(
"calibration for `{id}`: variant `{}` has statistics for `{option}`, \
which is not an option of this item",
short(&variant.variant)
));
}
}
if !variant.key.is_empty()
&& variant.variant != item.variant_digest(&variant.key, &variant.distractors)
{
issues.push(format!(
"calibration for `{id}`: variant `{}` was measured against an option set \
that has since been reworded, so its numbers describe wording no \
student now sees",
short(&variant.variant)
));
}
}
for option in calibration.options.keys() {
if item.option(option).is_none() {
issues.push(format!(
"calibration for `{id}`: an option history names `{option}`, which is \
not an option of this item"
));
}
}
}
issues
}
}
/// What one administration measured: `analysis/administrations/<id>.yaml`.
///
/// Written once, when the exam is analyzed, and never rewritten. It is the
/// audit trail under [`CalibrationFile`]: the pooled numbers say an item sits
/// at 0.63, and these say which exams that came from and what each one saw.
/// Keeping them also means the history survives losing `data/`, which is
/// ignored by git and rotates.
#[derive(Debug, Clone, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct MeasurementFile {
/// Schema version.
#[serde(
default = "default_version",
deserialize_with = "yaml::flexible_string"
)]
pub schema_version: String,
/// What was administered, and how it was analyzed.
pub administration: MeasurementMeta,
/// One entry per question, in the order it was printed.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub items: Vec<Measurement>,
}
/// What an administration was, for a reader two years later.
#[derive(Debug, Clone, Default, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct MeasurementMeta {
/// The administration id these numbers came from.
pub id: String,
/// The assessment that was administered.
pub assessment: String,
/// The term.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub term: Option<String>,
/// The date it was given.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub date: Option<Date>,
/// The forms in play.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub forms: Vec<String>,
/// How many examinees the numbers pool over.
///
/// The one number to read before any of the others. A point-biserial on
/// twenty-seven students is a different kind of claim than one on three
/// hundred, and nothing below records how thin it is.
pub n_examinees: usize,
/// The item response model fitted, when one was.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub model: Option<IrtModel>,
/// When the analysis was run.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub generated: Option<Date>,
/// The version of the tool that ran it, since the numbers depend on it.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub coursebank: Option<String>,
}
/// One question's statistics from one administration.
///
/// Cohort aggregates only. There is deliberately no per-section or per-form
/// breakdown: those get small, and a small cell crossed with anything else is
/// how an aggregate stops being one.
#[derive(Debug, Clone, Default, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct Measurement {
/// The item id.
pub item: String,
/// The question number it was printed as.
pub number: u32,
/// The option set administered. See [`crate::item::Item::variant_digest`].
#[serde(default, skip_serializing_if = "Option::is_none")]
pub variant: Option<String>,
/// The stem as administered. See [`crate::item::Item::stem_digest`].
#[serde(default, skip_serializing_if = "Option::is_none")]
pub stem_digest: Option<String>,
/// Examinees who saw it.
pub n: usize,
/// Proportion correct.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub p_value: Option<f64>,
/// Corrected item-total point-biserial correlation.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub point_biserial: Option<f64>,
/// Upper-minus-lower-group discrimination index.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub discrimination_index: Option<f64>,
/// The option ids keyed correct, so a row in the options file says whether
/// it describes the answer or a distractor without a join.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub key: Vec<String>,
/// Per-option behaviour, by option id.
#[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
pub option_stats: BTreeMap<String, OptionStat>,
/// Fitted parameters, when the sample supported a fit.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub irt: Option<IrtParams>,
/// Machine-detected problems with this question on this administration.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub flags: Vec<Flag>,
}
/// The columns of an items file, and the only ones accepted on read.
///
/// An allowlist rather than a type-level guarantee. In YAML the structures
/// reject unknown keys, so a file carrying a student column could not be
/// loaded; CSV readers are tolerant of extra columns, so the same assurance has
/// to be an explicit check. This is it, and [`MeasurementFile::read_csv`]
/// refuses any header not named here.
pub const ITEM_COLUMNS: [&str; 16] = [
"administration_id",
"assessment",
"term",
"date",
"forms",
"n_examinees",
"coursebank",
"generated",
"item",
"number",
"variant",
"stem_digest",
"n",
"p_value",
"point_biserial",
"discrimination_index",
];
/// The columns of an options file, and the only ones accepted on read.
pub const OPTION_COLUMNS: [&str; 9] = [
"administration_id",
"item",
"number",
"option",
"keyed",
"selection_rate",
"point_biserial",
"upper_group_rate",
"lower_group_rate",
];
/// One row of an items file.
#[derive(Debug, Clone, Serialize, Deserialize)]
struct ItemRow {
administration_id: String,
assessment: String,
term: String,
date: String,
forms: String,
n_examinees: usize,
coursebank: String,
generated: String,
item: String,
number: u32,
variant: String,
stem_digest: String,
n: usize,
p_value: String,
point_biserial: String,
discrimination_index: String,
}
/// One row of an options file.
#[derive(Debug, Clone, Serialize, Deserialize)]
struct OptionRow {
administration_id: String,
item: String,
number: u32,
option: String,
keyed: bool,
selection_rate: String,
point_biserial: String,
upper_group_rate: String,
lower_group_rate: String,
}
impl MeasurementFile {
/// The two file names this administration writes, items first.
///
/// # Arguments
///
/// * `dir` - usually `analysis/administrations`.
///
/// # Returns
///
/// The items path and the options path.
pub fn paths(&self, dir: &Path) -> (PathBuf, PathBuf) {
let stem = crate::course::slugify(&self.administration.id);
(
dir.join(format!("{stem}-items.csv")),
dir.join(format!("{stem}-options.csv")),
)
}
/// Writes the two files, refusing to overwrite either.
///
/// An administration is a thing that happened once, so replacing its record
/// is a deliberate act: delete the files first if you mean to re-analyze.
///
/// Two files rather than one because the data is two shapes — one row per
/// question, one row per question and option — and a single sparse table
/// serves neither. The administration's metadata repeats on every row,
/// which is what makes each file independently loadable and is the same
/// convention the response store already uses.
///
/// # Arguments
///
/// * `dir` - the directory to write into.
///
/// # Returns
///
/// The paths written.
///
/// # Errors
///
/// Returns [`Error::Usage`] when either file exists, and [`Error::Io`] or
/// [`Error::Csv`] on a write failure.
pub fn write_csv(&self, dir: &Path) -> Result<Vec<PathBuf>> {
let (items_path, options_path) = self.paths(dir);
for path in [&items_path, &options_path] {
if path.exists() {
return Err(Error::usage(format!(
"{} already records this administration. It happened once, so replacing it \
is a deliberate act: delete it first if you mean to re-analyze.",
path.display()
)));
}
}
std::fs::create_dir_all(dir).map_err(|e| Error::io(dir, e))?;
let meta = &self.administration;
let mut items = csv::Writer::from_path(&items_path).map_err(|e| Error::Csv {
path: items_path.clone(),
source: e,
})?;
let mut options = csv::Writer::from_path(&options_path).map_err(|e| Error::Csv {
path: options_path.clone(),
source: e,
})?;
for measurement in &self.items {
items
.serialize(ItemRow {
administration_id: meta.id.clone(),
assessment: meta.assessment.clone(),
term: meta.term.clone().unwrap_or_default(),
date: meta.date.map(|d| d.to_string()).unwrap_or_default(),
forms: meta.forms.join(";"),
n_examinees: meta.n_examinees,
coursebank: meta.coursebank.clone().unwrap_or_default(),
generated: meta.generated.map(|d| d.to_string()).unwrap_or_default(),
item: measurement.item.clone(),
number: measurement.number,
variant: measurement.variant.clone().unwrap_or_default(),
stem_digest: measurement.stem_digest.clone().unwrap_or_default(),
n: measurement.n,
p_value: number(measurement.p_value),
point_biserial: number(measurement.point_biserial),
discrimination_index: number(measurement.discrimination_index),
})
.map_err(|e| Error::Csv {
path: items_path.clone(),
source: e,
})?;
for (option, stat) in &measurement.option_stats {
options
.serialize(OptionRow {
administration_id: meta.id.clone(),
item: measurement.item.clone(),
number: measurement.number,
option: option.clone(),
keyed: measurement.key.iter().any(|k| k == option),
selection_rate: number(stat.selection_rate),
point_biserial: number(stat.point_biserial),
upper_group_rate: number(stat.upper_group_rate),
lower_group_rate: number(stat.lower_group_rate),
})
.map_err(|e| Error::Csv {
path: options_path.clone(),
source: e,
})?;
}
}
items.flush().map_err(|e| Error::io(&items_path, e))?;
options.flush().map_err(|e| Error::io(&options_path, e))?;
Ok(vec![items_path, options_path])
}
/// Reads one administration back from its two files.
///
/// # Arguments
///
/// * `items_path` - the items file. The options file is found beside it.
///
/// # Returns
///
/// The administration, with its per-option statistics reattached.
///
/// # Errors
///
/// Returns [`Error::Csv`] on a parse failure and [`Error::Invalid`] when a
/// file carries a column that is not in [`ITEM_COLUMNS`] or
/// [`OPTION_COLUMNS`] — which is how a student column is caught.
pub fn read_csv(items_path: &Path) -> Result<MeasurementFile> {
let options_path = PathBuf::from(
items_path
.to_string_lossy()
.replace("-items.csv", "-options.csv"),
);
let mut reader = open_csv(items_path, &ITEM_COLUMNS)?;
let mut meta = MeasurementMeta::default();
let mut items: Vec<Measurement> = Vec::new();
for row in reader.deserialize::<ItemRow>() {
let row = row.map_err(|e| Error::Csv {
path: items_path.to_path_buf(),
source: e,
})?;
meta = MeasurementMeta {
id: row.administration_id.clone(),
assessment: row.assessment.clone(),
term: some(&row.term),
date: parse_date(&row.date),
forms: row
.forms
.split(';')
.filter(|f| !f.is_empty())
.map(str::to_string)
.collect(),
n_examinees: row.n_examinees,
model: meta.model,
generated: parse_date(&row.generated),
coursebank: some(&row.coursebank),
};
items.push(Measurement {
item: row.item,
number: row.number,
variant: some(&row.variant),
stem_digest: some(&row.stem_digest),
n: row.n,
p_value: parse(&row.p_value),
point_biserial: parse(&row.point_biserial),
discrimination_index: parse(&row.discrimination_index),
..Measurement::default()
});
}
if options_path.is_file() {
let mut reader = open_csv(&options_path, &OPTION_COLUMNS)?;
for row in reader.deserialize::<OptionRow>() {
let row = row.map_err(|e| Error::Csv {
path: options_path.clone(),
source: e,
})?;
let Some(target) = items.iter_mut().find(|i| i.number == row.number) else {
continue;
};
if row.keyed && !target.key.contains(&row.option) {
target.key.push(row.option.clone());
}
target.option_stats.insert(
row.option,
OptionStat {
selection_rate: parse(&row.selection_rate),
point_biserial: parse(&row.point_biserial),
upper_group_rate: parse(&row.upper_group_rate),
lower_group_rate: parse(&row.lower_group_rate),
},
);
}
}
Ok(MeasurementFile {
schema_version: default_version(),
administration: meta,
items,
})
}
/// Reads every administration under a directory, oldest first.
///
/// # Arguments
///
/// * `dir` - usually `analysis/administrations`.
///
/// # Returns
///
/// The administrations, empty when the directory does not exist.
///
/// # Errors
///
/// Propagates read failures.
pub fn load_all(dir: &Path) -> Result<Vec<MeasurementFile>> {
if !dir.is_dir() {
return Ok(Vec::new());
}
let mut paths: Vec<PathBuf> = std::fs::read_dir(dir)
.map_err(|e| Error::io(dir, e))?
.filter_map(|e| e.ok().map(|e| e.path()))
.filter(|p| p.to_string_lossy().ends_with("-items.csv"))
.collect();
paths.sort();
let mut out = Vec::new();
for path in &paths {
out.push(MeasurementFile::read_csv(path)?);
}
out.sort_by(|a, b| {
a.administration
.date
.cmp(&b.administration.date)
.then(a.administration.id.cmp(&b.administration.id))
});
Ok(out)
}
}
/// Opens a CSV and refuses any column that is not on the allowlist.
fn open_csv(path: &Path, allowed: &[&str]) -> Result<csv::Reader<std::fs::File>> {
let mut reader = csv::Reader::from_path(path).map_err(|e| Error::Csv {
path: path.to_path_buf(),
source: e,
})?;
let headers = reader
.headers()
.map_err(|e| Error::Csv {
path: path.to_path_buf(),
source: e,
})?
.clone();
let unexpected: Vec<&str> = headers.iter().filter(|h| !allowed.contains(h)).collect();
if !unexpected.is_empty() {
return Err(Error::Invalid(vec![format!(
"{}: unexpected column(s) {}. These files are committed, so they hold cohort \
aggregates and nothing else — anything per-student belongs in data/, which is \
ignored.",
path.display(),
unexpected.join(", ")
)]));
}
Ok(reader)
}
/// A float as a CSV cell, empty when absent.
fn number(value: Option<f64>) -> String {
value.map(|v| format!("{v}")).unwrap_or_default()
}
/// A CSV cell as a float, absent when empty or unparseable.
fn parse(cell: &str) -> Option<f64> {
cell.trim().parse().ok()
}
/// A CSV cell as a string, absent when empty.
fn some(cell: &str) -> Option<String> {
(!cell.trim().is_empty()).then(|| cell.trim().to_string())
}
/// A CSV cell as a date, absent when empty or unparseable.
fn parse_date(cell: &str) -> Option<Date> {
some(cell).and_then(|c| c.parse().ok())
}
/// The schema version new files are written with.
fn default_version() -> String {
SCHEMA_VERSION.to_string()
}
/// A digest shortened for a message.
fn short(digest: &str) -> String {
digest.chars().take(8).collect()
}
#[cfg(test)]
mod tests {
use super::*;
use crate::item::VariantCalibration;
fn tmp(tag: &str) -> std::path::PathBuf {
let p = std::env::temp_dir().join(format!("coursebank-cal-{tag}-{}", std::process::id()));
let _ = std::fs::remove_dir_all(&p);
std::fs::create_dir_all(&p).unwrap();
p
}
#[test]
fn an_absent_store_is_empty_rather_than_an_error() {
let dir = tmp("absent");
let store = CalibrationFile::load(&dir.join(CALIBRATION_FILE)).unwrap();
assert!(store.items.is_empty());
assert!(store.get("q-x").is_none());
}
#[test]
fn the_store_round_trips() {
let dir = tmp("round");
let path = dir.join(CALIBRATION_FILE);
let mut store = CalibrationFile::default();
store.items.insert(
"q-x".to_string(),
Calibration {
n_examinees: Some(27),
p_value: Some(0.63),
variants: vec![VariantCalibration {
variant: "4c81fa".into(),
n_examinees: Some(27),
..VariantCalibration::default()
}],
..Calibration::default()
},
);
store.save(&path).unwrap();
let back = CalibrationFile::load(&path).unwrap();
assert_eq!(back.get("q-x").unwrap().p_value, Some(0.63));
assert_eq!(back.get("q-x").unwrap().variants.len(), 1);
let _ = std::fs::remove_dir_all(&dir);
}
#[test]
fn an_administration_round_trips_through_two_csvs() {
let dir = tmp("csv");
let file = MeasurementFile {
schema_version: default_version(),
administration: MeasurementMeta {
id: "e1-2026f".into(),
assessment: "e1".into(),
term: Some("2026f".into()),
date: Some(Date::new(2026, 9, 15).unwrap()),
forms: vec!["A".into(), "B".into()],
n_examinees: 27,
model: None,
generated: Some(Date::new(2026, 9, 27).unwrap()),
coursebank: Some("0.0.0".into()),
},
items: vec![Measurement {
item: "q-fastq-quality-length-match".into(),
number: 1,
variant: Some("237d62f9af222f78".into()),
stem_digest: Some("8b22e0".into()),
n: 27,
p_value: Some(0.5926),
point_biserial: Some(0.31),
discrimination_index: None,
key: vec!["o-fourth-line".into()],
option_stats: [
(
"o-fourth-line".to_string(),
OptionStat {
selection_rate: Some(0.5926),
point_biserial: Some(0.31),
upper_group_rate: None,
lower_group_rate: None,
},
),
(
"o-third-line".to_string(),
OptionStat {
selection_rate: Some(0.1852),
point_biserial: Some(-0.18),
upper_group_rate: None,
lower_group_rate: None,
},
),
]
.into_iter()
.collect(),
irt: None,
flags: Vec::new(),
}],
};
let written = file.write_csv(&dir).unwrap();
assert_eq!(written.len(), 2, "one table per shape");
assert!(written[0].ends_with("e1-2026f-items.csv"));
assert!(written[1].ends_with("e1-2026f-options.csv"));
// One line per question, loadable on its own: the administration's
// metadata repeats on the row, as the response store already does.
let items = std::fs::read_to_string(&written[0]).unwrap();
assert!(
items.starts_with("administration_id,assessment,term,date,forms"),
"{items}"
);
assert!(
items.contains("e1-2026f,e1,2026f,2026-09-15,A;B,27"),
"{items}"
);
let options = std::fs::read_to_string(&written[1]).unwrap();
assert!(options.contains("o-fourth-line,true"), "{options}");
assert!(options.contains("o-third-line,false"), "{options}");
let back = MeasurementFile::read_csv(&written[0]).unwrap();
assert_eq!(back.administration.id, "e1-2026f");
assert_eq!(back.administration.n_examinees, 27);
assert_eq!(back.administration.forms, vec!["A", "B"]);
assert_eq!(back.items.len(), 1);
assert_eq!(back.items[0].p_value, Some(0.5926));
assert_eq!(back.items[0].key, vec!["o-fourth-line"]);
assert_eq!(back.items[0].option_stats.len(), 2);
assert_eq!(back.items[0].discrimination_index, None);
assert_eq!(MeasurementFile::load_all(&dir).unwrap().len(), 1);
// It happened once, so the record is not replaced by accident.
let err = file.write_csv(&dir).unwrap_err().to_string();
assert!(err.contains("deliberate"), "{err}");
let _ = std::fs::remove_dir_all(&dir);
}
#[test]
fn a_student_column_is_refused_on_read() {
let dir = tmp("identifiers");
let path = dir.join("e1-2026f-items.csv");
std::fs::write(
&path,
"administration_id,assessment,item,number,n,student_key\n e1-2026f,e1,q-x,1,27,abc123\n",
)
.unwrap();
// In YAML this fell out of the type, which rejects unknown keys. A CSV
// reader tolerates extra columns, so the same assurance has to be an
// explicit allowlist — and this is the test that it is one.
let err = MeasurementFile::read_csv(&path).unwrap_err().to_string();
assert!(err.contains("student_key"), "{err}");
assert!(err.contains("cohort aggregates"), "{err}");
let _ = std::fs::remove_dir_all(&dir);
}
}
+250 -62
View File
@@ -5,7 +5,7 @@
//! Loading a whole course at once, and reporting on what it contains.
//!
//! A [`Catalog`] is every bank in a course, indexed so that an item can be found
//! by its global id (`bank::item`), and so that questions like "how many Apply
//! by its id, and so that questions like "how many Apply
//! level items do I have on lecture 12" have a cheap answer.
//!
//! The global id is the join key for everything downstream: assessment records
@@ -20,19 +20,24 @@
use std::collections::{BTreeMap, BTreeSet};
use std::path::{Path, PathBuf};
use crate::assessment::AssessmentFile;
use crate::assessment::{AssessmentFile, Placement};
use crate::bank::BankFile;
use crate::course::CourseFile;
use crate::error::{Error, Result};
use crate::item::Item;
use crate::layout::Layout;
use crate::taxonomy::{Level, Status, Tier};
use crate::taxonomy::{Format, Level, Status, Tier};
use crate::yaml;
/// One item plus everything needed to locate it again.
#[derive(Debug, Clone)]
pub struct Entry {
/// The globally unique id, `bank::item`.
/// The item's id, which names it course-wide.
///
/// The bank is in [`Entry::bank`] and the file in [`Entry::path`], neither
/// of which is part of the identity: this is the join key that response
/// data carries, and a join key with a file name in it renames itself every
/// time the files are reorganized.
pub uid: String,
/// The bank id.
pub bank: String,
@@ -70,6 +75,12 @@ pub struct Catalog {
pub entries: Vec<Entry>,
/// Bank metadata by bank id.
pub banks: BTreeMap<String, crate::bank::BankMeta>,
/// The statistics, loaded from `analysis/` rather than from the banks.
///
/// Each entry's [`crate::item::Item::calibration`] is filled from this at
/// load, so everything downstream reads one item and does not have to know
/// that the numbers and the wording come from different files.
pub calibration: crate::calibration::CalibrationFile,
/// Map from global id to index into `entries`.
index: BTreeMap<String, usize>,
}
@@ -92,14 +103,17 @@ impl Catalog {
/// id, since either makes the join key ambiguous.
pub fn load(root: &Path) -> Result<Catalog> {
let layout = Layout::new(root);
let course = CourseFile::load(&layout.course_file())?;
let course = CourseFile::load_dir(root)?;
let mut catalog = Catalog {
course,
layout,
entries: Vec::new(),
banks: BTreeMap::new(),
index: BTreeMap::new(),
calibration: crate::calibration::CalibrationFile::default(),
};
catalog.calibration =
crate::calibration::CalibrationFile::load(&catalog.layout.calibration_file())?;
let mut problems = Vec::new();
let mut files = yaml::list_yaml(&catalog.layout.banks())?;
@@ -117,9 +131,13 @@ impl Catalog {
}
catalog.banks.insert(bank_id.clone(), bank.bank.clone());
for (i, item) in bank.items.into_iter().enumerate() {
let uid = format!("{bank_id}::{}", item.id);
if catalog.index.contains_key(&uid) {
problems.push(format!("duplicate global item id `{uid}`"));
let uid = item.id.clone();
if let Some(first) = catalog.index.get(&uid) {
problems.push(format!(
"item id `{uid}` is used twice: in bank `{}` and in bank `{bank_id}`. An \
id names one question course-wide, because response data joins on it.",
catalog.entries[*first].bank
));
continue;
}
catalog.index.insert(uid.clone(), catalog.entries.len());
@@ -136,20 +154,36 @@ impl Catalog {
if !problems.is_empty() {
return Err(Error::Invalid(problems));
}
// Statistics are attached here rather than parsed from the bank, which
// is why an item can be read as one thing while its wording and its
// evidence live in files with different review cycles.
for entry in &mut catalog.entries {
if let Some(calibration) = catalog.calibration.items.get(&entry.uid) {
entry.item.calibration = Some(calibration.clone());
}
}
Ok(catalog)
}
/// Looks up an item by global id.
///
/// A pre-2.0 `bank::item` id resolves to the item it used to name, so an
/// assessment record or a parquet file written before the change still
/// joins. See [`crate::item::canonical_id`].
///
/// # Arguments
///
/// * `uid` - the global id, `bank::item`.
/// * `uid` - the item id, in either form.
///
/// # Returns
///
/// The entry, or `None`.
pub fn get(&self, uid: &str) -> Option<&Entry> {
self.index.get(uid).map(|i| &self.entries[*i])
self.index
.get(uid)
.or_else(|| self.index.get(crate::item::canonical_id(uid)))
.map(|i| &self.entries[*i])
}
/// Looks up an item by global id, erroring when absent.
@@ -173,45 +207,37 @@ impl Catalog {
})
}
/// Resolves a possibly-unqualified id to a global id.
/// Resolves an id written in either form to the canonical one.
///
/// Typing `q-mm-kinetics-001` on the command line should work when that id is
/// unambiguous across the course, because remembering which bank a question
/// lives in is exactly the sort of bookkeeping this tool exists to remove.
/// Since 2.0 an item id is already course-wide, so this is the identity for
/// anything current. What it is still for is the old `bank::item` form,
/// which appears in assessment records, seals, and response files written
/// before the change, and which a person may well still type.
///
/// # Arguments
///
/// * `id` - a global id, or a bare item id.
/// * `id` - an item id, in either form.
///
/// # Returns
///
/// The global id.
/// The canonical id.
///
/// # Errors
///
/// Returns [`Error::Unresolved`] when nothing matches, or [`Error::Usage`]
/// when a bare id matches items in more than one bank.
/// Returns [`Error::Unresolved`] when nothing matches.
pub fn resolve(&self, id: &str) -> Result<String> {
if self.index.contains_key(id) {
return Ok(id.to_string());
}
let matches: Vec<&Entry> = self.entries.iter().filter(|e| e.item.id == id).collect();
match matches.len() {
0 => Err(Error::Unresolved {
kind: "item",
id: id.to_string(),
context: None,
}),
1 => Ok(matches[0].uid.clone()),
_ => Err(Error::usage(format!(
"`{id}` is ambiguous; it exists in {}. Use the full `bank::item` form.",
matches
.iter()
.map(|e| e.bank.as_str())
.collect::<Vec<_>>()
.join(", ")
))),
let canonical = crate::item::canonical_id(id);
if self.index.contains_key(canonical) {
return Ok(canonical.to_string());
}
Err(Error::Unresolved {
kind: "item",
id: id.to_string(),
context: None,
})
}
/// Every item that may be placed on a graded assessment.
@@ -232,12 +258,10 @@ impl Catalog {
///
/// Problems, prefixed with the file they came from.
pub fn validate(&self) -> Result<Vec<String>> {
let mut issues: Vec<String> = self
.course
.validate()
.into_iter()
.map(|m| format!("course.yaml: {m}"))
.collect();
// Not prefixed here: `CourseFile::validate` attributes each message to
// the fragment that defined the id, which for an unsplit course is
// `course.yaml` and for a split one is the file worth opening.
let mut issues: Vec<String> = self.course.validate();
for path in yaml::list_yaml(&self.layout.banks())? {
let bank = BankFile::load_resolved(&path)?;
@@ -255,6 +279,154 @@ impl Catalog {
Ok(issues)
}
/// Checks one placement's option set against the item's pool.
///
/// The checks that moved here from the bank when options became a pool. A
/// bank holding two defensible keys and six distractors is sound; what has
/// to hold for a *form* is that exactly one key reached the student, that
/// none of the distractors was true, and that the count matches policy.
/// None of that can be decided by looking at the item alone.
///
/// # Arguments
///
/// * `p` - the placement.
/// * `item` - the item it names.
///
/// # Returns
///
/// One message per problem.
fn validate_placement(&self, p: &Placement, item: &Item) -> Vec<String> {
let mut issues = Vec::new();
if !item.format.has_options() {
return issues;
}
let at = |number: u32| format!("question {number} ({})", p.item);
for id in p.key.iter().chain(p.distractors.iter()) {
if item.option(id).is_none() {
issues.push(format!(
"{}: `{id}` is not an option of this item",
at(p.number)
));
}
}
for id in &p.distractors {
if p.key.iter().any(|k| k == id) {
issues.push(format!(
"{}: `{id}` is listed as both the key and a distractor",
at(p.number)
));
}
if item.option(id).is_some_and(|o| o.correct) {
issues.push(format!(
"{}: `{id}` is offered as a distractor but the bank keys it correct",
at(p.number)
));
}
}
for id in &p.key {
if item.option(id).is_some_and(|o| !o.correct) {
issues.push(format!(
"{}: `{id}` is keyed correct here but the bank does not key it. An option is \
true or it is not; which true option a form uses is this record's choice, \
but not whether it is true.",
at(p.number)
));
}
}
let single = item.format == Format::SingleBestAnswer;
if single && p.key.len() > 1 {
issues.push(format!(
"{}: single_best_answer administers exactly one key, this names {}",
at(p.number),
p.key.len()
));
}
// No distractor list means the whole pool, which is what a pre-2.0
// record means and what an item whose pool is its form still means.
// The record still has to say which key, when the pool offers a choice.
if p.distractors.is_empty() {
let (keys, _) = item.pool();
if p.key.is_empty() && keys.len() > 1 && single {
issues.push(format!(
"{}: the item offers {} defensible keys, so the record has to say which one \
this assessment used",
at(p.number),
keys.len()
));
}
} else {
let shown = item.administered(&p.key, &p.distractors);
let expected = self.course.policy.options_per_item;
if shown.len() != expected {
issues.push(format!(
"{}: administers {} option(s), but course policy is {expected} per item",
at(p.number),
shown.len()
));
}
if single && p.key.is_empty() {
issues.push(format!(
"{}: names its distractors but not its key, so what was marked correct is \
left to whatever the bank says today",
at(p.number)
));
}
}
if let Some(recorded) = &p.variant {
if *recorded != item.variant_digest(&p.key, &p.distractors) {
issues.push(format!(
"{}: an administered option has been reworded since this assessment. \
Statistics pooled under this variant describe the older wording.",
at(p.number)
));
}
}
issues
}
/// Checks every sealed administration's stems against the bank.
///
/// The seal is the authority on what was administered, so this is the
/// comparison that matters: a record can be edited, but a seal is written
/// before the exam is printed and digested against tampering. A stem that
/// no longer matches the one a cohort answered means the id now names a
/// different question, and every statistic pooled under it is describing
/// two things at once.
///
/// # Arguments
///
/// * `seals` - the sealed administrations to check.
///
/// # Returns
///
/// One message per stem that has moved out from under its seal.
pub fn validate_seals(&self, seals: &[crate::seal::SealFile]) -> Vec<String> {
let mut issues = Vec::new();
for seal in seals {
for item in &seal.items {
let Some(digest) = &item.stem_digest else {
continue;
};
let Some(entry) = self.get(&item.item) else {
continue;
};
if *digest != entry.item.stem_digest() {
issues.push(format!(
"{}: question {} (`{}`) was administered with a different stem than the \
bank now holds. Statistics from that administration describe the older \
wording; give the new wording its own id.",
seal.seal.assessment, item.number, item.item
));
}
}
}
issues
}
/// Validates a record's internal invariants, then its references against this
/// catalog: unknown items, keys that drifted, and fingerprints showing the
/// item was reworded since it was administered.
@@ -264,6 +436,22 @@ impl Catalog {
match self.get(&p.item) {
None => issues.push(format!("question {}: unknown item `{}`", p.number, p.item)),
Some(entry) => {
// The rule the stem digest exists to enforce. A changed
// fingerprint is a note: the statistics describe an older
// wording. A changed stem is an error: whatever was
// administered is not the question the bank now holds, so
// the id is being reused for two different questions.
if let Some(digest) = &p.stem_digest {
if *digest != entry.item.stem_digest() {
issues.push(format!(
"question {} ({}): the stem has been reworded since this \
assessment. A reworded stem is a new question: give the new \
wording a new id with `supersedes: {}`, and leave this one as \
it was administered.",
p.number, p.item, p.item
));
}
}
if let Some(fp) = &p.fingerprint {
if *fp != entry.item.fingerprint() {
issues.push(format!(
@@ -273,11 +461,7 @@ impl Catalog {
));
}
}
if !p.key.is_empty() && p.key != entry.item.key_letters() {
issues.push(format!(
"question {} ({}): the recorded key {:?} differs from the item's current key {:?}",
p.number, p.item, p.key, entry.item.key_letters()));
}
issues.extend(self.validate_placement(p, &entry.item));
}
}
}
@@ -705,13 +889,32 @@ items:
let cat = Catalog::load(&dir).expect("catalog loads");
assert_eq!(cat.entries.len(), 1);
assert_eq!(cat.entries[0].uid, "b1::q-x-001");
// The id names the item course-wide; the bank is where it is kept.
assert_eq!(cat.entries[0].uid, "q-x-001");
assert_eq!(cat.entries[0].bank, "b1");
assert!(cat.get("q-x-001").is_some());
assert_eq!(cat.resolve("q-x-001").unwrap(), "q-x-001");
// A record or a parquet file written before 2.0 still joins.
assert!(cat.get("b1::q-x-001").is_some());
assert_eq!(cat.resolve("q-x-001").unwrap(), "b1::q-x-001");
assert_eq!(cat.resolve("b1::q-x-001").unwrap(), "q-x-001");
assert!(cat.get("b1::q-nonexistent").is_none());
assert!(cat.validate().unwrap().is_empty());
let _ = std::fs::remove_dir_all(&dir);
}
#[test]
fn one_item_id_in_two_banks_is_fatal() {
let dir = tmp("dupitem");
write_course(&dir, "");
write_bank(&dir, "b1.yaml", APPROVED);
write_bank(&dir, "b2.yaml", &APPROVED.replace("id: b1", "id: b2"));
let message = Catalog::load(&dir).unwrap_err().to_string();
assert!(message.contains("`q-x-001` is used twice"), "{message}");
assert!(message.contains("course-wide"), "{message}");
let _ = std::fs::remove_dir_all(&dir);
}
#[test]
fn duplicate_bank_ids_are_fatal() {
let dir = tmp("dupbank");
@@ -723,21 +926,6 @@ items:
let _ = std::fs::remove_dir_all(&dir);
}
#[test]
fn ambiguous_bare_ids_are_rejected() {
let dir = tmp("ambig");
write_course(&dir, "");
write_bank(&dir, "a.yaml", APPROVED);
write_bank(&dir, "b.yaml", &APPROVED.replace("id: b1", "id: b2"));
let cat = Catalog::load(&dir).expect("distinct banks load");
assert_eq!(cat.entries.len(), 2);
let err = cat.resolve("q-x-001").expect_err("bare id is ambiguous");
assert!(format!("{err}").contains("ambiguous"));
// The fully qualified form still works.
assert_eq!(cat.resolve("b2::q-x-001").unwrap(), "b2::q-x-001");
let _ = std::fs::remove_dir_all(&dir);
}
#[test]
fn coverage_finds_real_gaps() {
let dir = tmp("coverage");
+524 -32
View File
@@ -4,20 +4,29 @@
//! The course file: identity plus the registries every bank references.
//!
//! Learning objectives, their learning targets, and lectures are declared once,
//! in `course.yaml`, and referenced by id from items. That is the single most load-bearing decision in
//! Learning objectives, their learning targets, and lectures are declared once
//! and referenced by id from items. That is the single most load-bearing decision in
//! the schema. It means an objective's wording lives in exactly one place, so
//! rewording it updates every report; it means a report can name what a student
//! missed by objective rather than by question number; and it means a dangling
//! reference is a hard error instead of a silently misspelled string that splits
//! your coverage table into two near-identical rows.
//!
//! "Once" is a claim about ids, not about files. [`CourseFile`] is the resolved
//! model, and it may be assembled from a directory of fragments — one file per
//! lecture, one per objective — as well as from a single `course.yaml`. Either
//! way an id has exactly one definition site, and [`CourseFile::origins`]
//! records which file that was. See [`fragment`] for the merge and the rules
//! that keep it honest.
//!
//! The course file also declares the term. Items live across terms, so the term
//! belongs to the course and the administration, never to the item.
pub mod fragment;
use std::collections::BTreeMap;
use std::fmt;
use std::path::Path;
use std::path::{Path, PathBuf};
use serde::de::{self, MapAccess, Visitor};
use serde::ser::SerializeMap;
@@ -28,8 +37,19 @@ use crate::error::{Error, Result};
use crate::taxonomy::Level;
use crate::yaml;
use fragment::Section;
/// The schema version this build of the tool writes.
pub const SCHEMA_VERSION: &str = "1.0";
pub const SCHEMA_VERSION: &str = "2.0";
/// The schema major versions this build can read.
///
/// A 1.0 repository loads unchanged. What 2.0 changes is the shape of two
/// things, and both are tolerated on the way in: a course file may be split
/// into fragments, and an item is named course-wide rather than as
/// `bank::item`. `coursebank migrate` rewrites files into the 2.0 form when you
/// are ready; nothing forces it.
pub const SUPPORTED_MAJORS: [&str; 2] = ["1", "2"];
/// The canonical file name inside a course directory.
pub const COURSE_FILE: &str = "course.yaml";
@@ -91,6 +111,16 @@ pub struct CourseFile {
/// Shared stimuli for case-based testlets, keyed by id.
#[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
pub stimuli: BTreeMap<String, Stimulus>,
/// Which file defined each id, relative to the course root.
///
/// Populated by [`fragment::assemble`] and empty for a course parsed
/// straight out of one file by [`CourseFile::load`]. It is what makes a
/// validation message able to name the file to open, which matters rather a
/// lot once one course is forty files. Not serialized: it describes where
/// the model came from, not what it says.
#[serde(skip)]
pub origins: BTreeMap<(Section, String), PathBuf>,
}
/// Course identity.
@@ -285,6 +315,15 @@ pub struct Lecture {
/// Where the slides live, for study guidance in student reports.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub slides_url: Option<String>,
/// The objectives this session develops.
///
/// The registration direction: you write what a lecture covers while
/// planning the lecture, and each named objective gains this lecture in its
/// [`Objective::lectures`] list during [`fragment::assemble`]. Declaring the
/// pair from the objective's side instead is equivalent, and declaring it
/// from both is redundant rather than contradictory — the two are unioned.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub teaches: Vec<String>,
/// Assigned readings for the session, in the order you assign them.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub readings: Vec<Reading>,
@@ -335,8 +374,21 @@ pub struct Reference {
#[serde(default, skip_serializing_if = "Option::is_none")]
pub pages: Option<String>,
/// DOI, bare: `10.1038/nature12373`.
///
/// For a manuscript this is usually the only link worth storing: it is the
/// identifier of the work rather than of one copy of it, and [`Reference::href`]
/// turns it into a URL.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub doi: Option<String>,
/// arXiv id, bare: `2301.00001` or `q-bio/0501001`.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub arxiv: Option<String>,
/// PubMed Central id, which hosts the full text: `PMC3084216`.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub pmcid: Option<String>,
/// PubMed id, which hosts a record about the work: `21471563`.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub pmid: Option<String>,
/// ISBN, for a book.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub isbn: Option<String>,
@@ -353,6 +405,123 @@ pub struct Reference {
pub note: Option<String>,
}
impl Reference {
/// Where to send a reader, most specific first.
///
/// The one link resolution in the crate. Every exporter used to carry its
/// own copy of the `base_url` join, which meant a reading list, a printed
/// key, a practice sheet, and a student report could disagree about where a
/// citation points — and that none of them linked a journal article, since
/// an article has no `base_url` to join a path to.
///
/// The order is from the exact location outward: a link to §1.4 beats a link
/// to the work, and a link to the work beats nothing.
///
/// # Arguments
///
/// * `url` - a full URL for the exact location, from a reading or citation.
/// * `path` - a location under this work's `base_url`.
///
/// # Returns
///
/// The most specific link available, or `None` for a work with no online
/// location at all.
pub fn href(&self, url: Option<&str>, path: Option<&str>) -> Option<String> {
if let Some(url) = url {
return Some(url.to_string());
}
if let (Some(base), Some(path)) = (self.base_url.as_deref(), path) {
return Some(join_url(base, path));
}
if let Some(url) = &self.url {
return Some(url.clone());
}
self.identifier_url()
}
/// The link this work's identifiers resolve to, ignoring any location inside
/// it.
///
/// DOI first, because it names the work rather than one copy of it. Then
/// arXiv and PubMed Central, which host the article itself, before PubMed,
/// which hosts a record about it.
///
/// # Returns
///
/// A URL, or `None` when the work carries no identifier.
pub fn identifier_url(&self) -> Option<String> {
if let Some(doi) = self.doi.as_deref().map(bare_doi) {
return Some(format!("https://doi.org/{doi}"));
}
if let Some(id) = self.arxiv.as_deref().map(bare_arxiv) {
return Some(format!("https://arxiv.org/abs/{id}"));
}
if let Some(id) = self.pmcid.as_deref().map(str::trim) {
let id = if id.starts_with("PMC") {
id.to_string()
} else {
format!("PMC{id}")
};
return Some(format!("https://www.ncbi.nlm.nih.gov/pmc/articles/{id}/"));
}
if let Some(id) = self.pmid.as_deref().map(str::trim) {
return Some(format!("https://pubmed.ncbi.nlm.nih.gov/{id}/"));
}
None
}
/// The short form a reading list shows: the label, or the citation key.
///
/// # Arguments
///
/// * `key` - the citation key, used when the work declares no label.
///
/// # Returns
///
/// The label to print.
pub fn label_or<'a>(&'a self, key: &'a str) -> &'a str {
self.label.as_deref().unwrap_or(key)
}
}
/// A DOI with any resolver prefix stripped, so `href` cannot produce
/// `https://doi.org/https://doi.org/10...`.
fn bare_doi(doi: &str) -> &str {
let doi = doi.trim();
for prefix in [
"https://doi.org/",
"http://doi.org/",
"https://dx.doi.org/",
"http://dx.doi.org/",
"doi:",
] {
if let Some(rest) = doi.strip_prefix(prefix) {
return rest;
}
}
doi
}
/// An arXiv id with the `arXiv:` prefix stripped.
fn bare_arxiv(id: &str) -> &str {
let id = id.trim();
for prefix in ["arXiv:", "arxiv:", "https://arxiv.org/abs/"] {
if let Some(rest) = id.strip_prefix(prefix) {
return rest;
}
}
id
}
/// Joins a base URL and a path without doubling or dropping the separator.
fn join_url(base: &str, path: &str) -> String {
match (base.ends_with('/'), path.starts_with('/')) {
(true, true) => format!("{base}{}", &path[1..]),
(false, false) => format!("{base}/{path}"),
_ => format!("{base}{path}"),
}
}
/// The kind of work, chosen to map onto BibTeX entry types.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
#[serde(rename_all = "kebab-case")]
@@ -447,19 +616,10 @@ impl Reading {
///
/// # Returns
///
/// `url` when given, otherwise the reference's `base_url` joined with `path`,
/// otherwise `None`.
/// The most specific link available, which for a manuscript with a DOI and
/// no `path` is the DOI. See [`Reference::href`].
pub fn resolve_url(&self, reference: &Reference) -> Option<String> {
if let Some(url) = &self.url {
return Some(url.clone());
}
let path = self.path.as_deref()?;
let base = reference.base_url.as_deref()?;
Some(match (base.ends_with('/'), path.starts_with('/')) {
(true, true) => format!("{base}{}", &path[1..]),
(false, false) => format!("{base}/{path}"),
_ => format!("{base}{path}"),
})
reference.href(self.url.as_deref(), self.path.as_deref())
}
/// A short citation for a report: `KKW §6.1`.
@@ -476,7 +636,7 @@ impl Reading {
if let Some(text) = &self.text {
return text.clone();
}
let label = reference.label.as_deref().unwrap_or(key);
let label = reference.label_or(key);
match &self.locator {
Some(locator) => format!("{label} {locator}"),
None => label.to_string(),
@@ -640,15 +800,25 @@ pub struct Objective {
/// The lectures that develop it.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub lectures: Vec<String>,
/// Position in teaching order, low first.
/// Position in teaching order, low first. Derived; authoring it is
/// deprecated.
///
/// The registry is a map, so declaration order is lost on load, and sorting
/// The registry is a map, so declaration order is lost on load and sorting
/// by id would put `lo-enthalpy` before `lo-first-law` when the second is a
/// prerequisite of the first. Anything that prints objectives in the order
/// you teach them, a lecture page above all, needs this. Objectives without
/// it sort last, by id.
/// prerequisite of the first. Something has to supply the order.
///
/// Since 2.0 that something is [`Lecture::teaches`], which is a sequence:
/// the position of an objective in the list of what a lecture covers, and
/// the position of the lecture in the course, together say when it is
/// taught. [`fragment::assemble`] fills this in from those two, so a course
/// that used to carry thirty-nine hand-kept integers now carries none, and
/// inserting an objective is a one-line edit rather than a renumber.
///
/// An authored value still wins, so a 1.0 course loads unchanged.
/// `coursebank migrate order` removes them.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub order: Option<u32>,
/// The highest level you intend to assess this objective at. Assembling an
/// item above the ceiling is a warning: either the item overreaches or the
/// ceiling needs raising.
@@ -705,8 +875,20 @@ pub struct Target {
///
/// Ordered within its objective rather than across the course, so two
/// targets under different objectives never compete for a position and
/// inserting one renumbers nothing outside its own group.
#[serde(default, skip_serializing_if = "Option::is_none")]
/// Retained only so a pre-2.0 course still loads. Ignored.
///
/// Targets do not have an order. An objective's targets are a set of
/// question templates, not steps in a sequence: they are not taught in
/// order, an exam samples from them rather than working through them, and
/// the study workflow reads the list as a checklist and counts what it can
/// do cold. A position would assert a sequence that does not exist.
///
/// Where a list has to be printed, [`CourseFile::targets`] orders it by
/// ceiling and then by id. Where one target genuinely depends on another,
/// that is [`Target::prerequisites`], which says so directly.
///
/// `coursebank migrate order` removes it.
#[serde(default, skip_serializing)]
pub order: Option<u32>,
/// The highest level you intend to assess this target at. Omit it to inherit
/// the objective's.
@@ -761,7 +943,12 @@ impl CourseFile {
yaml::read(path)
}
/// Finds and loads the course file for a course directory.
/// Loads the course for a course directory, merging every fragment it holds.
///
/// This is the entry point every command uses. A directory holding only
/// `course.yaml` gives the same result it always did; one that also holds
/// `lectures/`, `objectives/`, or `references.yaml` gets them merged in. See
/// [`fragment::assemble`].
///
/// # Arguments
///
@@ -769,34 +956,167 @@ impl CourseFile {
///
/// # Returns
///
/// The parsed course file.
/// The merged course file.
///
/// # Errors
///
/// Propagates load errors, including absence of `course.yaml`.
/// Propagates load errors, including absence of `course.yaml`, and returns
/// [`Error::Invalid`] when two files define the same id.
pub fn load_dir(dir: &Path) -> Result<CourseFile> {
CourseFile::load(&dir.join(COURSE_FILE))
fragment::assemble(dir)
}
/// Which file defined an id, and which registry it was in.
///
/// # Arguments
///
/// * `id` - a unit, lecture, objective, target, reference, or stimulus id.
///
/// # Returns
///
/// The section and the path relative to the course root, or `None` for an
/// unknown id or a course that was not assembled from fragments.
pub fn origin(&self, id: &str) -> Option<(Section, &Path)> {
Section::ALL.iter().find_map(|section| {
self.origins
.get(&(*section, id.to_string()))
.map(|path| (*section, path.as_path()))
})
}
/// Every file this course was assembled from, in sorted order.
///
/// # Returns
///
/// The paths relative to the course root, empty for a course parsed from a
/// single file by [`CourseFile::load`].
pub fn fragment_paths(&self) -> Vec<&Path> {
let mut paths: Vec<&Path> = self.origins.values().map(PathBuf::as_path).collect();
paths.sort_unstable();
paths.dedup();
paths
}
/// Writes the course file back out as YAML.
///
/// Refuses to write a course that was assembled from more than one file,
/// because the merged model has no home on disk: writing it to
/// `course.yaml` would leave every fragment defining ids the root file also
/// defines, which is the one thing [`fragment::assemble`] treats as an
/// error. Use [`CourseFile::write_resolved`] for an inspection copy.
///
/// # Arguments
///
/// * `path` - destination path.
///
/// # Errors
///
/// Returns [`Error::Io`] on a write failure.
/// Returns [`Error::Usage`] for a fragmented course and [`Error::Io`] on a
/// write failure.
pub fn save(&self, path: &Path) -> Result<()> {
let sources = self.fragment_paths();
if sources.len() > 1 {
return Err(Error::usage(format!(
"this course is assembled from {} files, so it cannot be written back to one. \
Edit the fragment that owns what you are changing, or use `coursebank course \
build` for a merged copy.",
sources.len()
)));
}
yaml::write(path, self)
}
/// Writes the merged course as YAML, for reading rather than for loading.
///
/// The output carries a banner saying so. It is what `coursebank course
/// build` writes, and nothing in the tool reads it back: a generated file
/// that commands depend on is a file that goes stale.
///
/// # Arguments
///
/// * `path` - destination path.
///
/// # Errors
///
/// Returns [`Error::Io`] on a write failure, or [`Error::Other`] if the
/// model cannot be represented as YAML.
pub fn write_resolved(&self, path: &Path) -> Result<()> {
let body = yaml::to_string(self)?;
let banner = format!(
"# Generated by `coursebank course build` from {} file(s). Do not edit: nothing\n\
# reads this, and the next build overwrites it. Edit the fragments instead.\n",
self.fragment_paths().len().max(1)
);
yaml::write_text(path, &format!("{banner}{body}"))
}
/// Checks internal consistency of the registries.
///
/// Each message is prefixed with the file that defined the id it is about,
/// when that is known. For an unsplit course that is always `course.yaml`,
/// which is what the messages used to say.
///
/// # Returns
///
/// Every problem found, empty when the file is sound.
pub fn validate(&self) -> Vec<String> {
self.problems()
.into_iter()
.map(|issue| self.attribute(issue))
.collect()
}
/// Prefixes one validation message with the fragment it concerns.
///
/// The id is taken from the first backticked token in the message, since
/// every message that is about a registry entry names it first. Messages
/// about `course`, `policy`, or `units` are attributed by section instead,
/// because the first thing they quote is a field or a grade letter.
///
/// # Arguments
///
/// * `issue` - the message.
///
/// # Returns
///
/// The message, prefixed with a path when one is known.
fn attribute(&self, issue: String) -> String {
let section = if issue.starts_with("course.") {
Some(Section::Course)
} else if issue.starts_with("policy.") {
Some(Section::Policy)
} else if issue.starts_with("units") {
Some(Section::Units)
} else {
None
};
let path = match section {
Some(section) => self.section_origin(section),
None => issue
.split('`')
.nth(1)
.and_then(|id| self.origin(id))
.map(|(_, path)| path),
};
match path {
Some(path) => format!("{}: {issue}", path.display()),
None => issue,
}
}
/// The file that declared a whole section, for the sections that are not
/// keyed by id.
fn section_origin(&self, section: Section) -> Option<&Path> {
self.origins
.iter()
.find(|((s, _), _)| *s == section)
.map(|(_, path)| path.as_path())
}
/// The validation messages, before they are attributed to files.
fn problems(&self) -> Vec<String> {
let mut issues = Vec::new();
if self.course.code.trim().is_empty() {
@@ -874,6 +1194,52 @@ impl CourseFile {
if reference.title.trim().is_empty() {
issues.push(format!("reference `{key}`: empty title"));
}
// Checked rather than silently coerced: `href` strips a resolver
// prefix, but something that is not a DOI at all would become a
// link that 404s on a student's reading list.
if let Some(doi) = &reference.doi {
if !bare_doi(doi).starts_with("10.") {
issues.push(format!(
"reference `{key}`: `{doi}` is not a DOI. Write it bare, as \
10.1038/nature12373."
));
}
}
// A manuscript with no journal is a citation nobody can print. The
// fields exist; a note that carries them instead is data the reading
// list cannot link and the bibliography exporters cannot use.
if matches!(
reference.kind,
ReferenceKind::Article | ReferenceKind::Preprint
) && reference.container.is_none()
{
issues.push(format!(
"reference `{key}`: an {} needs a `container` — the journal, preprint \
server, or proceedings it appeared in. Run `coursebank migrate references` \
if it is sitting in the `note`.",
match reference.kind {
ReferenceKind::Preprint => "preprint",
_ => "article",
}
));
}
if let Some(note) = &reference.note {
if crate::citation::looks_like_a_citation(note) {
issues.push(format!(
"reference `{key}`: the note still carries a citation. Volume, pages, \
and DOI have their own fields, and a DOI in a note is a link nobody \
can follow. `coursebank migrate references` takes it apart."
));
}
}
if let Some(pmid) = &reference.pmid {
if !pmid.trim().chars().all(|c| c.is_ascii_digit()) {
issues.push(format!(
"reference `{key}`: pmid `{pmid}` is not a number. A `PMC...` id goes in \
`pmcid`."
));
}
}
if let Some(label) = &reference.label {
labels.entry(label.as_str()).or_default().push(key);
}
@@ -897,6 +1263,20 @@ impl CourseFile {
issues.push(format!("lecture `{id}`: unknown unit `{u}`"));
}
}
for objective in &lec.teaches {
if !self.learning_objectives.contains_key(objective) {
if self.learning_targets.contains_key(objective) {
issues.push(format!(
"lecture `{id}`: `teaches` names the target `{objective}`, but it \
registers objectives. A target is reached through its objective."
));
} else {
issues.push(format!(
"lecture `{id}`: `teaches` names an unknown objective `{objective}`"
));
}
}
}
let mut seen: Vec<(&str, &str)> = Vec::new();
for (index, reading) in lec.readings.iter().enumerate() {
issues.extend(self.reading_issues(id, index, reading, &mut seen));
@@ -1315,9 +1695,20 @@ impl CourseFile {
.filter(|(_, target)| target.objective == objective)
.map(|(id, _)| id)
.collect();
// By ceiling, then by id. A target is a question template rather than a
// step in a sequence — an objective's targets are not taught in an
// order, and an exam samples from them — so there is no teaching order
// to print. What there is is depth, and grouping by it puts the
// checklist in the order the study methods apply: recall for a Level 1
// target, explanation for Level 2, variations and written solutions
// above that. Ties break by id so the list is stable.
ids.sort_by_key(|id| {
let target = &self.learning_targets[*id];
(target.order.unwrap_or(u32::MAX), (*id).clone())
(
self.effective_level_ceiling(id)
.map(|l| l.code())
.unwrap_or(0),
(*id).clone(),
)
});
ids.into_iter().map(String::as_str).collect()
}
@@ -1770,6 +2161,7 @@ impl CourseFile {
date: None,
unit: Some("u-intro".to_string()),
slides_url: None,
teaches: Vec::new(),
readings: Vec::new(),
},
);
@@ -1825,6 +2217,7 @@ impl CourseFile {
learning_targets: targets,
references: BTreeMap::new(),
stimuli: BTreeMap::new(),
origins: BTreeMap::new(),
}
}
}
@@ -1893,7 +2286,7 @@ course:
term: Spring 2026
"#,
);
assert_eq!(c.schema_version, "1.0");
assert_eq!(c.schema_version, SCHEMA_VERSION);
assert_eq!(c.policy.options_per_item, 4);
assert_eq!(c.course.slug(), "biosc-1540");
assert!(c.validate().is_empty());
@@ -2040,6 +2433,105 @@ learning_objectives:
assert_eq!(reading.cite("kuriyan2013molecules", reference), "KKW §6.1");
}
#[test]
fn a_link_resolves_from_the_exact_location_outward() {
let mut reference = Reference {
title: "Basic local alignment search tool".into(),
kind: ReferenceKind::Article,
..Reference::default()
};
// Nothing at all to link to.
assert_eq!(reference.href(None, None), None);
// A DOI is a link to the work, which beats nothing.
reference.doi = Some("10.1016/S0022-2836(05)80360-2".into());
assert_eq!(
reference.href(None, None).as_deref(),
Some("https://doi.org/10.1016/S0022-2836(05)80360-2")
);
// The work's own URL is more use than its identifier.
reference.url = Some("https://example.org/blast".into());
assert_eq!(
reference.href(None, None).as_deref(),
Some("https://example.org/blast")
);
// A location inside the work beats the work.
reference.base_url = Some("https://example.org/blast/".into());
assert_eq!(
reference.href(None, Some("/§2")).as_deref(),
Some("https://example.org/blast/§2")
);
assert_eq!(
reference
.href(Some("https://example.org/exact"), Some("§2"))
.as_deref(),
Some("https://example.org/exact")
);
}
#[test]
fn identifiers_are_normalized_before_they_become_links() {
let doi_as_url = Reference {
title: "T".into(),
doi: Some("https://doi.org/10.1/x".into()),
..Reference::default()
};
assert_eq!(
doi_as_url.href(None, None).as_deref(),
Some("https://doi.org/10.1/x")
);
let preprint = Reference {
title: "T".into(),
kind: ReferenceKind::Preprint,
arxiv: Some("arXiv:2301.00001".into()),
..Reference::default()
};
assert_eq!(
preprint.href(None, None).as_deref(),
Some("https://arxiv.org/abs/2301.00001")
);
// A bare number is still a PMC id.
let open_access = Reference {
title: "T".into(),
pmcid: Some("3084216".into()),
..Reference::default()
};
assert_eq!(
open_access.href(None, None).as_deref(),
Some("https://www.ncbi.nlm.nih.gov/pmc/articles/PMC3084216/")
);
}
#[test]
fn something_that_is_not_a_doi_is_reported() {
let c = parse(
r#"
course: { code: X, title: Y, term: Z }
references:
bad:
title: A work
doi: nature12373
worse:
title: Another work
pmid: PMC3084216
"#,
);
let issues = c.validate();
assert!(
issues.iter().any(|i| i.contains("is not a DOI")),
"{issues:?}"
);
assert!(
issues.iter().any(|i| i.contains("is not a number")),
"{issues:?}"
);
}
#[test]
fn a_bare_string_reading_still_parses_and_round_trips() {
let c = parse(
+974
View File
@@ -0,0 +1,974 @@
// SPDX-License-Identifier: Prosperity-3.0.0
// Copyright Scientific Computing Studio
// Source: https://git.scient.ing/education/coursebank
//! One course, several files.
//!
//! A course of forty lectures does not fit in a file anyone wants to scroll. So
//! the registries [`CourseFile`] holds may be spread across a directory and
//! merged on load:
//!
//! ```text
//! course.yaml course, policy, units
//! references.yaml references
//! lectures/l-1-2.yaml one lecture, its readings, and what it teaches
//! objectives/lo-x.yaml one objective and its targets
//! ```
//!
//! The merge happens in memory on every command. Nothing is generated on disk
//! and no command depends on a build step, because a generated file that other
//! commands read is a file that goes stale. `coursebank course build` exists to
//! show you the merged result, and nothing reads what it writes.
//!
//! # What this does not relax
//!
//! Splitting a file is only worth doing if it cannot introduce a second
//! definition of the same thing. Two rules keep that true, and both are enforced
//! here rather than left to convention:
//!
//! * **One definition site per id.** Two files defining `lo-read-file-formats`
//! is an error naming both paths. The winner is not the last file loaded,
//! because there is no winner.
//! * **A section belongs to a kind of file.** A file under `lectures/` may not
//! define `learning_objectives`. Otherwise the layout decays into forty files
//! that each might hold anything, which is the same navigation problem in a
//! worse shape.
//!
//! `course.yaml` is exempt from the second rule: a course that has not been
//! split is a single fragment that happens to define everything, and it keeps
//! loading unchanged.
//!
//! # Two derivations
//!
//! Splitting by lecture makes two fields tedious to maintain by hand, so they
//! are derived instead:
//!
//! * A lecture's `teaches:` list adds that lecture to each named objective's
//! `lectures`. You write what a lecture covers while planning the lecture,
//! which is when you know.
//! * A target with no `lectures` of its own inherits its objective's, the same
//! way it already inherits `level_ceiling`.
//!
//! Both are unions and both are idempotent, so declaring a pair on both sides
//! is redundant rather than contradictory.
use std::collections::BTreeMap;
use std::fmt;
use std::path::{Path, PathBuf};
use serde::{Deserialize, Serialize};
use super::{
COURSE_FILE, Course, CourseFile, Lecture, Objective, Policy, Reference, SCHEMA_VERSION,
SUPPORTED_MAJORS, Stimulus, Target, Unit,
};
use crate::error::{Error, Result};
use crate::layout::Layout;
use crate::yaml;
/// The file holding the bibliography when it is kept out of `course.yaml`.
pub const REFERENCES_FILE: &str = "references.yaml";
/// One registry section of a course.
///
/// Used to say which file a fact came from, and to keep a fragment from
/// defining something that belongs somewhere else.
#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)]
pub enum Section {
/// Course identity.
Course,
/// Course-wide policy.
Policy,
/// Units.
Units,
/// Lectures.
Lectures,
/// Learning objectives.
Objectives,
/// Learning targets.
Targets,
/// Works the course cites.
References,
/// Shared stimuli.
Stimuli,
}
impl Section {
/// Every section, in the order a merged course lists them.
pub const ALL: [Section; 8] = [
Section::Course,
Section::Policy,
Section::Units,
Section::Lectures,
Section::Objectives,
Section::Targets,
Section::References,
Section::Stimuli,
];
/// The YAML key this section is written under.
pub fn key(self) -> &'static str {
match self {
Section::Course => "course",
Section::Policy => "policy",
Section::Units => "units",
Section::Lectures => "lectures",
Section::Objectives => "learning_objectives",
Section::Targets => "learning_targets",
Section::References => "references",
Section::Stimuli => "stimuli",
}
}
/// What one of its entries is called in a message.
pub fn noun(self) -> &'static str {
match self {
Section::Course => "course identity",
Section::Policy => "policy",
Section::Units => "unit",
Section::Lectures => "lecture",
Section::Objectives => "objective",
Section::Targets => "target",
Section::References => "reference",
Section::Stimuli => "stimulus",
}
}
/// Where a file defining this section is expected to live.
pub fn home(self) -> &'static str {
match self {
Section::Course | Section::Policy | Section::Units => COURSE_FILE,
Section::Lectures => "lectures/*.yaml",
Section::Objectives | Section::Targets | Section::Stimuli => "objectives/*.yaml",
Section::References => REFERENCES_FILE,
}
}
}
impl fmt::Display for Section {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
f.write_str(self.key())
}
}
/// What kind of file a fragment is, which fixes what it may define.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Role {
/// `course.yaml`. May define anything, so an unsplit course still loads.
Root,
/// `references.yaml`.
References,
/// A file under `lectures/`.
Lecture,
/// A file under `objectives/`.
Objective,
}
impl Role {
/// Whether a file in this role may define a section.
pub fn allows(self, section: Section) -> bool {
match self {
Role::Root => true,
Role::References => section == Section::References,
Role::Lecture => section == Section::Lectures,
Role::Objective => matches!(
section,
Section::Objectives | Section::Targets | Section::Stimuli
),
}
}
/// A short name for a message.
pub fn label(self) -> &'static str {
match self {
Role::Root => "course",
Role::References => "references",
Role::Lecture => "lecture",
Role::Objective => "objective",
}
}
}
/// One file's worth of course registries.
///
/// Every section is optional, which is what makes this both the fragment schema
/// and — with every section filled in — the schema of an unsplit `course.yaml`.
/// [`CourseFile`] is the resolved model the rest of the crate reads; this is
/// only what one file on disk is allowed to say.
#[derive(Debug, Clone, Default, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct Fragment {
/// Schema version this file targets.
#[serde(
default,
deserialize_with = "yaml::flexible_string_opt",
skip_serializing_if = "Option::is_none"
)]
pub schema_version: Option<String>,
/// Course identity. Exactly one fragment must carry it.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub course: Option<Course>,
/// Course-wide policy.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub policy: Option<Policy>,
/// Units, in teaching order.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub units: Vec<Unit>,
/// Lectures by id.
#[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
pub lectures: BTreeMap<String, Lecture>,
/// Learning objectives by id.
#[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
pub learning_objectives: BTreeMap<String, Objective>,
/// Learning targets by id.
#[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
pub learning_targets: BTreeMap<String, Target>,
/// Works the course cites, by citation key.
#[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
pub references: BTreeMap<String, Reference>,
/// Shared stimuli by id.
#[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
pub stimuli: BTreeMap<String, Stimulus>,
}
impl Fragment {
/// Loads one fragment from disk.
///
/// # Arguments
///
/// * `path` - the file to read.
///
/// # Returns
///
/// The parsed fragment.
///
/// # Errors
///
/// Returns [`Error::Io`] if unreadable and [`Error::Yaml`] if it does not
/// match the schema. Unknown keys are errors, so a misspelled section name
/// is caught here rather than silently contributing nothing.
pub fn load(path: &Path) -> Result<Fragment> {
yaml::read(path)
}
/// Which sections this fragment actually defines.
pub fn sections(&self) -> Vec<Section> {
let mut out = Vec::new();
if self.course.is_some() {
out.push(Section::Course);
}
if self.policy.is_some() {
out.push(Section::Policy);
}
if !self.units.is_empty() {
out.push(Section::Units);
}
if !self.lectures.is_empty() {
out.push(Section::Lectures);
}
if !self.learning_objectives.is_empty() {
out.push(Section::Objectives);
}
if !self.learning_targets.is_empty() {
out.push(Section::Targets);
}
if !self.references.is_empty() {
out.push(Section::References);
}
if !self.stimuli.is_empty() {
out.push(Section::Stimuli);
}
out
}
}
/// The fragment files of a course directory, in load order, with their roles.
///
/// `course.yaml` is listed whether or not it exists, so a directory that is not
/// a course fails with a message naming the file it wanted rather than an empty
/// merge. Directory contents are sorted, which is what makes the merged course
/// independent of filesystem order.
///
/// # Arguments
///
/// * `layout` - the resolved course layout.
///
/// # Returns
///
/// Paths paired with what each file is allowed to define.
///
/// # Errors
///
/// Returns [`Error::Io`] when a fragment directory exists but cannot be read.
pub fn files(layout: &Layout) -> Result<Vec<(PathBuf, Role)>> {
let mut out = vec![(layout.course_file(), Role::Root)];
let references = layout.references_file();
if references.is_file() {
out.push((references, Role::References));
}
for path in yaml::list_yaml(&layout.lectures())? {
out.push((path, Role::Lecture));
}
for path in yaml::list_yaml(&layout.objectives())? {
out.push((path, Role::Objective));
}
Ok(out)
}
/// Loads every fragment in a course directory and merges them into one course.
///
/// # Arguments
///
/// * `root` - the course directory.
///
/// # Returns
///
/// The merged course, with [`CourseFile::origins`] recording which file defined
/// each id.
///
/// # Errors
///
/// Propagates load errors, and returns [`Error::Invalid`] with every merge
/// problem at once: an id defined twice, a section in the wrong kind of file, a
/// fragment written against another major schema version, or no `course:`
/// section anywhere.
///
/// Cross-references are *not* checked here. A dangling objective id is a
/// content problem, and content problems are [`CourseFile::validate`]'s, so that
/// they are reported the same way whether or not the course is split.
pub fn assemble(root: &Path) -> Result<CourseFile> {
let layout = Layout::new(root);
let mut merge = Merge::default();
for (path, role) in files(&layout)? {
let fragment = Fragment::load(&path)?;
let shown = path.strip_prefix(root).unwrap_or(&path).to_path_buf();
merge.take(&shown, role, fragment);
}
merge.resolve();
merge.finish(root)
}
/// Accumulates fragments, remembering where each id came from.
#[derive(Debug, Default)]
struct Merge {
schema_version: Option<String>,
course: Option<Course>,
policy: Option<Policy>,
units: Vec<Unit>,
lectures: BTreeMap<String, Lecture>,
objectives: BTreeMap<String, Objective>,
targets: BTreeMap<String, Target>,
references: BTreeMap<String, Reference>,
stimuli: BTreeMap<String, Stimulus>,
origins: BTreeMap<(Section, String), PathBuf>,
issues: Vec<String>,
}
impl Merge {
/// Folds one fragment in.
///
/// # Arguments
///
/// * `path` - the fragment's path relative to the course root, for messages.
/// * `role` - what this file is allowed to define.
/// * `fragment` - the parsed fragment.
fn take(&mut self, path: &Path, role: Role, fragment: Fragment) {
for section in fragment.sections() {
if !role.allows(section) {
self.issues.push(format!(
"{}: a {} file may not define `{}`; that section belongs in {}",
path.display(),
role.label(),
section.key(),
section.home()
));
}
}
if let Some(declared) = &fragment.schema_version {
if !SUPPORTED_MAJORS.contains(&major(declared)) {
self.issues.push(format!(
"{}: declares schema_version {declared}, which this build cannot read. It \
writes {SCHEMA_VERSION} and reads {}.",
path.display(),
SUPPORTED_MAJORS
.iter()
.map(|m| format!("{m}.x"))
.collect::<Vec<_>>()
.join(" and ")
));
}
if self.schema_version.is_none() {
self.schema_version = Some(declared.clone());
}
}
if role.allows(Section::Course) {
if let Some(course) = fragment.course {
if self.claim(Section::Course, path) {
self.course = Some(course);
}
}
}
if role.allows(Section::Policy) {
if let Some(policy) = fragment.policy {
if self.claim(Section::Policy, path) {
self.policy = Some(policy);
}
}
}
if role.allows(Section::Units) {
for unit in fragment.units {
let key = (Section::Units, unit.id.clone());
if let Some(first) = self.origins.get(&key) {
let message = duplicate(Section::Units, &unit.id, first.as_path(), path);
self.issues.push(message);
continue;
}
self.origins.insert(key, path.to_path_buf());
self.units.push(unit);
}
}
if role.allows(Section::Lectures) {
absorb(
&mut self.lectures,
fragment.lectures,
Section::Lectures,
path,
&mut self.origins,
&mut self.issues,
);
}
if role.allows(Section::Objectives) {
absorb(
&mut self.objectives,
fragment.learning_objectives,
Section::Objectives,
path,
&mut self.origins,
&mut self.issues,
);
}
if role.allows(Section::Targets) {
absorb(
&mut self.targets,
fragment.learning_targets,
Section::Targets,
path,
&mut self.origins,
&mut self.issues,
);
}
if role.allows(Section::References) {
absorb(
&mut self.references,
fragment.references,
Section::References,
path,
&mut self.origins,
&mut self.issues,
);
}
if role.allows(Section::Stimuli) {
absorb(
&mut self.stimuli,
fragment.stimuli,
Section::Stimuli,
path,
&mut self.origins,
&mut self.issues,
);
}
}
/// Records a section that may only be declared once.
///
/// # Arguments
///
/// * `section` - the section being claimed.
/// * `path` - the file claiming it.
///
/// # Returns
///
/// Whether the claim was the first, and so whether the caller should store
/// what it parsed.
fn claim(&mut self, section: Section, path: &Path) -> bool {
let key = (section, String::new());
if let Some(first) = self.origins.get(&key) {
let message = format!(
"`{}` is declared twice: {} and {}. It applies to the whole course, so it has \
one definition site.",
section.key(),
first.display(),
path.display()
);
self.issues.push(message);
return false;
}
self.origins.insert(key, path.to_path_buf());
true
}
/// Fills in the two fields a split layout would otherwise duplicate.
fn resolve(&mut self) {
// A lecture says what it teaches; the objective's lecture list follows.
for (lecture_id, lecture) in &self.lectures {
for objective_id in &lecture.teaches {
if let Some(objective) = self.objectives.get_mut(objective_id) {
if !objective.lectures.iter().any(|l| l == lecture_id) {
objective.lectures.push(lecture_id.clone());
}
}
}
}
// Teaching order, from the two sequences that already declare it: the
// lectures in course order, and each lecture's `teaches` list. This is
// what lets an objective stop carrying a hand-kept integer.
let mut position = 0u32;
let ordered: Vec<String> = self.lectures.keys().cloned().collect();
for lecture_id in ordered {
let teaches = self
.lectures
.get(&lecture_id)
.map(|l| l.teaches.clone())
.unwrap_or_default();
for objective_id in teaches {
position += 1;
if let Some(objective) = self.objectives.get_mut(&objective_id) {
if objective.order.is_none() {
objective.order = Some(position);
}
}
}
}
// A target with no lecture of its own is taught wherever its objective
// is. Collected first: the read of `objectives` and the write to
// `targets` cannot overlap in one pass.
let inherited: Vec<(String, Vec<String>)> = self
.targets
.iter()
.filter(|(_, target)| target.lectures.is_empty())
.filter_map(|(id, target)| {
self.objectives
.get(&target.objective)
.map(|objective| (id.clone(), objective.lectures.clone()))
})
.collect();
for (id, lectures) in inherited {
if let Some(target) = self.targets.get_mut(&id) {
target.lectures = lectures;
}
}
}
/// Builds the course, or reports every merge problem at once.
fn finish(self, root: &Path) -> Result<CourseFile> {
let mut issues = self.issues;
let course = match self.course {
Some(course) => course,
None => {
issues.push(format!(
"no file in {} declares a `course:` section, so the course has no code, \
title, or term",
root.display()
));
return Err(Error::Invalid(issues));
}
};
if !issues.is_empty() {
return Err(Error::Invalid(issues));
}
Ok(CourseFile {
schema_version: self
.schema_version
.unwrap_or_else(|| SCHEMA_VERSION.to_string()),
course,
policy: self.policy.unwrap_or_default(),
units: self.units,
lectures: self.lectures,
learning_objectives: self.objectives,
learning_targets: self.targets,
references: self.references,
stimuli: self.stimuli,
origins: self.origins,
})
}
}
/// Moves one section's entries across, refusing a second definition.
fn absorb<T>(
into: &mut BTreeMap<String, T>,
from: BTreeMap<String, T>,
section: Section,
path: &Path,
origins: &mut BTreeMap<(Section, String), PathBuf>,
issues: &mut Vec<String>,
) {
for (id, value) in from {
let key = (section, id.clone());
if let Some(first) = origins.get(&key) {
issues.push(duplicate(section, &id, first.as_path(), path));
continue;
}
origins.insert(key, path.to_path_buf());
into.insert(id, value);
}
}
/// The message for an id defined in two files.
fn duplicate(section: Section, id: &str, first: &Path, second: &Path) -> String {
format!(
"{} `{id}` is defined in two places: {} and {}. An id has one definition site; delete \
one or rename it.",
section.noun(),
first.display(),
second.display()
)
}
/// The part of a schema version before the first dot.
fn major(version: &str) -> &str {
version.split('.').next().unwrap_or(version)
}
#[cfg(test)]
mod tests {
use super::*;
fn tmp(tag: &str) -> PathBuf {
let p = std::env::temp_dir().join(format!("coursebank-frag-{tag}-{}", std::process::id()));
let _ = std::fs::remove_dir_all(&p);
std::fs::create_dir_all(&p).unwrap();
p
}
fn write(root: &Path, relative: &str, body: &str) {
let path = root.join(relative);
std::fs::create_dir_all(path.parent().unwrap()).unwrap();
std::fs::write(path, body).unwrap();
}
const ROOT: &str = r#"
course:
code: BIOSC 1540
title: Computational Biology
term: 2026f
policy:
points_per_item: 1.0
units:
- id: u1
title: Search and Similarity
"#;
#[test]
fn a_split_course_merges_into_one_model() {
let root = tmp("merge");
write(&root, "course.yaml", ROOT);
write(
&root,
"references.yaml",
"references:\n ismail2023:\n title: Bioinformatics\n",
);
write(
&root,
"lectures/l-1-2.yaml",
"lectures:\n L1.2:\n title: The Digital Genome\n unit: u1\n \
teaches: [lo-read-file-formats]\n",
);
write(
&root,
"objectives/lo-read-file-formats.yaml",
"learning_objectives:\n lo-read-file-formats:\n text: Read the text formats.\n \
unit: u1\nlearning_targets:\n t-fastq-structure:\n text: Identify the four \
lines.\n objective: lo-read-file-formats\n",
);
let course = assemble(&root).unwrap();
assert_eq!(course.course.code, "BIOSC 1540");
assert_eq!(course.units.len(), 1);
assert_eq!(course.references.len(), 1);
assert!(course.validate().is_empty(), "{:?}", course.validate());
// Derived: the lecture registered the objective, and the target
// inherited the objective's lecture.
assert_eq!(
course.learning_objectives["lo-read-file-formats"].lectures,
vec!["L1.2".to_string()]
);
assert_eq!(
course.learning_targets["t-fastq-structure"].lectures,
vec!["L1.2".to_string()]
);
assert_eq!(course.lecture_targets("L1.2"), vec!["t-fastq-structure"]);
}
#[test]
fn an_unsplit_course_file_still_loads() {
let root = tmp("monolith");
write(
&root,
"course.yaml",
&format!(
"{ROOT}lectures:\n L1.2:\n title: The Digital Genome\nlearning_objectives:\n \
lo-x:\n text: Do the thing.\n lectures: [L1.2]\nlearning_targets:\n \
t-x:\n text: Do the smaller thing.\n objective: lo-x\nreferences:\n \
ismail2023:\n title: Bioinformatics\n"
),
);
let course = assemble(&root).unwrap();
assert!(course.validate().is_empty(), "{:?}", course.validate());
assert_eq!(course.lecture_objectives("L1.2"), vec!["lo-x"]);
assert_eq!(
course.origin("lo-x").map(|(_, p)| p.to_path_buf()),
Some(PathBuf::from(COURSE_FILE))
);
}
#[test]
fn an_id_defined_twice_names_both_files() {
let root = tmp("dup");
write(&root, "course.yaml", ROOT);
let body = "learning_objectives:\n lo-x:\n text: Do the thing.\n";
write(&root, "objectives/lo-x.yaml", body);
write(&root, "objectives/lo-x-old.yaml", body);
let err = assemble(&root).unwrap_err();
let message = err.to_string();
assert!(message.contains("objectives/lo-x.yaml"), "{message}");
assert!(message.contains("objectives/lo-x-old.yaml"), "{message}");
assert!(message.contains("one definition site"), "{message}");
}
#[test]
fn a_section_in_the_wrong_kind_of_file_is_rejected() {
let root = tmp("misplaced");
write(&root, "course.yaml", ROOT);
write(
&root,
"lectures/l-1-2.yaml",
"lectures:\n L1.2:\n title: The Digital Genome\nlearning_objectives:\n lo-x:\n \
text: Do the thing.\n",
);
let message = assemble(&root).unwrap_err().to_string();
assert!(
message.contains("may not define `learning_objectives`"),
"{message}"
);
assert!(message.contains("objectives/*.yaml"), "{message}");
}
#[test]
fn a_course_with_no_identity_says_so() {
let root = tmp("no-course");
write(&root, "course.yaml", "units:\n - id: u1\n title: One\n");
let message = assemble(&root).unwrap_err().to_string();
assert!(message.contains("`course:` section"), "{message}");
}
#[test]
fn a_fragment_from_another_major_version_is_refused() {
let root = tmp("version");
write(&root, "course.yaml", ROOT);
write(
&root,
"objectives/lo-x.yaml",
"schema_version: '9.0'\nlearning_objectives:\n lo-x:\n text: Do the thing.\n",
);
let message = assemble(&root).unwrap_err().to_string();
assert!(message.contains("schema_version 9.0"), "{message}");
}
#[test]
fn origins_point_at_the_fragment_that_defined_each_id() {
let root = tmp("origins");
write(&root, "course.yaml", ROOT);
write(
&root,
"lectures/l-1-2.yaml",
"lectures:\n L1.2:\n title: The Digital Genome\n",
);
write(
&root,
"objectives/lo-x.yaml",
"learning_objectives:\n lo-x:\n text: Do the thing.\n",
);
let course = assemble(&root).unwrap();
let (section, path) = course.origin("lo-x").unwrap();
assert_eq!(section, Section::Objectives);
assert_eq!(path, Path::new("objectives/lo-x.yaml"));
let (section, path) = course.origin("L1.2").unwrap();
assert_eq!(section, Section::Lectures);
assert_eq!(path, Path::new("lectures/l-1-2.yaml"));
assert!(course.origin("nothing-like-this").is_none());
}
#[test]
fn teaching_order_is_derived_from_the_two_sequences_that_declare_it() {
let root = tmp("order");
write(&root, "course.yaml", ROOT);
write(
&root,
"lectures/l-1-2.yaml",
"lectures:\n L1.2:\n title: One\n teaches: [lo-second, lo-first]\n",
);
write(
&root,
"lectures/l-1-3.yaml",
"lectures:\n L1.3:\n title: Two\n teaches: [lo-third]\n",
);
write(
&root,
"objectives/lo-first.yaml",
r#"learning_objectives:
lo-first:
text: A.
lo-second:
text: B.
lo-third:
text: C.
"#,
);
let course = assemble(&root).unwrap();
// Position in `teaches`, not id order: the lecture lists `lo-second`
// first and that is what teaching it first means.
assert_eq!(course.learning_objectives["lo-second"].order, Some(1));
assert_eq!(course.learning_objectives["lo-first"].order, Some(2));
assert_eq!(course.learning_objectives["lo-third"].order, Some(3));
assert_eq!(
course.lecture_objectives("L1.2"),
vec!["lo-second", "lo-first"]
);
}
#[test]
fn targets_print_by_ceiling_rather_than_in_a_sequence() {
let root = tmp("target-order");
write(&root, "course.yaml", ROOT);
write(
&root,
"lectures/l-1-2.yaml",
"lectures:\n L1.2:\n title: One\n teaches: [lo-x]\n",
);
write(
&root,
"objectives/lo-x.yaml",
r#"learning_objectives:
lo-x:
text: A.
level_ceiling: 3
learning_targets:
t-predict:
text: Predict the effect.
objective: lo-x
level_ceiling: 3
t-define:
text: Define the term.
objective: lo-x
level_ceiling: 1
t-explain:
text: Explain the mechanism.
objective: lo-x
level_ceiling: 2
"#,
);
let course = assemble(&root).unwrap();
assert!(course.validate().is_empty(), "{:?}", course.validate());
// Shallowest first, which is the order the study methods apply in:
// recall, then explanation, then variations. Not id order, which would
// put `t-define` after `t-predict` for no reason at all.
assert_eq!(
course.targets("lo-x"),
vec!["t-define", "t-explain", "t-predict"]
);
// A target with no ceiling of its own inherits the objective's, so it
// sorts where that puts it.
assert_eq!(course.learning_targets["t-define"].order, None);
}
#[test]
fn teaching_the_same_objective_from_two_lectures_unions() {
let root = tmp("union");
write(&root, "course.yaml", ROOT);
write(
&root,
"lectures/l-1-2.yaml",
"lectures:\n L1.2:\n title: One\n teaches: [lo-x]\n",
);
write(
&root,
"lectures/l-1-3.yaml",
"lectures:\n L1.3:\n title: Two\n teaches: [lo-x]\n",
);
write(
&root,
"objectives/lo-x.yaml",
"learning_objectives:\n lo-x:\n text: Do the thing.\n",
);
let course = assemble(&root).unwrap();
assert_eq!(
course.learning_objectives["lo-x"].lectures,
vec!["L1.2".to_string(), "L1.3".to_string()]
);
}
#[test]
fn a_declaration_on_both_sides_is_not_duplicated() {
let root = tmp("both-sides");
write(&root, "course.yaml", ROOT);
write(
&root,
"lectures/l-1-2.yaml",
"lectures:\n L1.2:\n title: One\n teaches: [lo-x]\n",
);
write(
&root,
"objectives/lo-x.yaml",
"learning_objectives:\n lo-x:\n text: Do the thing.\n lectures: [L1.2]\n",
);
let course = assemble(&root).unwrap();
assert_eq!(
course.learning_objectives["lo-x"].lectures,
vec!["L1.2".to_string()]
);
}
#[test]
fn teaching_an_unknown_objective_is_a_validation_problem_not_a_merge_one() {
let root = tmp("unknown-teaches");
write(&root, "course.yaml", ROOT);
write(
&root,
"lectures/l-1-2.yaml",
"lectures:\n L1.2:\n title: One\n teaches: [lo-nope]\n",
);
let course = assemble(&root).unwrap();
let issues = course.validate();
assert!(issues.iter().any(|i| i.contains("lo-nope")), "{issues:?}");
}
}
+478 -19
View File
@@ -25,6 +25,7 @@ use serde::de::{self, MapAccess, Visitor};
use serde::ser::SerializeMap;
use serde::{Deserialize, Deserializer, Serialize, Serializer};
use crate::course::Reference;
use crate::date::Date;
use crate::hash::fingerprint;
use crate::taxonomy::{
@@ -44,9 +45,28 @@ pub struct Item {
/// Revision counter, bumped whenever the content changes in a way that
/// invalidates pooled statistics.
#[serde(default = "one_u32")]
/// Retained only so a pre-2.0 bank still loads. Ignored.
///
/// A version number on a question answered the wrong question. It recorded
/// that *something* changed without constraining what, which meant an item
/// at version 3 might have a reworded distractor — fair, the statistics
/// still describe the same question — or a reworded stem, which makes it a
/// different question wearing the same id. Since 2.0 the stem *is* the
/// identity: reword it and you have a new item, with a new id and
/// [`Item::supersedes`] pointing back. [`Item::stem_digest`] is what
/// enforces that, against the seals of every administration.
///
/// `coursebank migrate stems` removes it.
#[serde(default, skip_serializing)]
pub version: u32,
/// The item this one replaces, when it is a rewording of an earlier stem.
///
/// Lineage rather than versioning: both items stay in the bank, each with
/// its own statistics, and a report can say which one a cohort answered.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub supersedes: Option<String>,
/// Workflow state; only [`Status::Approved`] items may be assembled.
pub status: Status,
@@ -137,7 +157,14 @@ pub struct Item {
pub design: Option<Design>,
/// What the evidence says, accumulated across administrations.
#[serde(default, skip_serializing_if = "Option::is_none")]
/// What the statistics say, filled in from `analysis/` at load.
///
/// Read from the store and never written back: `skip_serializing` means a
/// bank file cannot acquire a `calibration:` block by being round-tripped
/// through this type. A bank is a reviewed artifact whose diff should be a
/// change of intent, and every grading run would otherwise produce a diff
/// on it. See [`crate::calibration`].
#[serde(default, skip_serializing)]
pub calibration: Option<Calibration>,
/// The last review decision recorded for this item.
@@ -145,7 +172,16 @@ pub struct Item {
pub review: Option<Review>,
/// Append-only change log.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
/// Retained only so a pre-2.0 bank still loads. Ignored.
///
/// A hand-maintained change log inside a version-controlled file, every
/// entry of which duplicated what `git log -p` already knew, with no
/// guarantee of agreeing with it. What git cannot express is a claim about
/// the item rather than a record of an edit, and that has its own fields:
/// [`Item::retired`] and [`Item::supersedes`].
///
/// `coursebank migrate stems` removes it.
#[serde(default, skip_serializing)]
pub history: Vec<HistoryEntry>,
/// The author of record.
@@ -170,7 +206,17 @@ pub struct Item {
#[derive(Debug, Clone, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct Choice {
/// Option letter, `A` through `H`.
/// The option's id, unique within its item: `o-fourth-line`.
///
/// A name rather than a position. Until 2.0 this was a letter, which put a
/// position in a field that pooled statistics, student feedback, and
/// `credit_overrides` all join on — so reordering a YAML block silently
/// moved the misconception recorded against one option onto another. The
/// letter a student sees is derived per form from the form's seed and lives
/// in the seal; see [`crate::seal::printed_letter`].
///
/// A single letter `A` through `H` still loads, so a bank migrates when you
/// run `coursebank migrate options` rather than when you upgrade.
pub id: String,
/// The option text.
@@ -218,6 +264,17 @@ pub struct Choice {
/// Your a priori guess at how often this option is chosen.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub selection_rate_expected: Option<f64>,
/// Why this option is no longer drawn, when it is not.
///
/// A retired option stays in the file forever. It has to: a seal and four
/// terms of response rows refer to it by id, and deleting it would turn
/// every one of those references into a dangling one. What retirement does
/// is take it out of the pool an assessment draws from, with the reason
/// attached — "selected by 1 of 96 across two administrations" is a finding
/// about the option, and the place for it is next to the option.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub retired: Option<Retirement>,
}
impl Choice {
@@ -319,6 +376,19 @@ impl Citation {
(None, None) => String::new(),
}
}
/// The link for this location, resolved against the work it points into.
///
/// # Arguments
///
/// * `reference` - the work, looked up from the citation key.
///
/// # Returns
///
/// The most specific link available. See [`Reference::href`].
pub fn href(&self, reference: &Reference) -> Option<String> {
reference.href(self.url.as_deref(), self.path.as_deref())
}
}
/// Writes a citation as a mapping, or as a bare string when that is all it holds.
@@ -553,6 +623,89 @@ pub struct Calibration {
/// Machine-detected problems.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub flags: Vec<Flag>,
/// One record per option set ever administered.
///
/// What the flat fields above cannot express once options are a pool. A
/// stem shown with distractors `{third, second, first}` is a measurably
/// easier item than the same stem with `{third, plus-line, line-two}`, so a
/// p-value pooled across both is the average of two different questions.
/// Statistics are computed and compared per variant; the flat fields remain
/// as the pre-2.0 summary, and for an item whose pool is its form the two
/// agree.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub variants: Vec<VariantCalibration>,
/// One record per option, pooled across every set it appeared in.
///
/// The capability the pool is worth the trouble for. Selection rates are
/// shares of a fixed set, so they are only comparable *within* a variant —
/// which means this view supports exactly one kind of claim, and it is the
/// useful one: this option draws nobody, anywhere. That is the evidence
/// that retires a distractor, and one administration cannot supply it.
#[serde(default, skip_serializing_if = "std::collections::BTreeMap::is_empty")]
pub options: std::collections::BTreeMap<String, OptionHistory>,
}
/// Statistics for one option set, as administered.
#[derive(Debug, Clone, Default, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct VariantCalibration {
/// The digest this record describes. See [`Item::variant_digest`].
pub variant: String,
/// The option ids keyed correct.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub key: Vec<String>,
/// The option ids offered alongside them.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub distractors: Vec<String>,
/// The administrations pooled into these numbers.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub administrations: Vec<String>,
/// Examinees pooled.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub n_examinees: Option<usize>,
/// Proportion correct.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub p_value: Option<f64>,
/// Corrected item-total point-biserial correlation.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub point_biserial: Option<f64>,
/// Upper-minus-lower-group discrimination index.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub discrimination_index: Option<f64>,
/// Per-option behaviour within this set, keyed by option id.
#[serde(default, skip_serializing_if = "std::collections::BTreeMap::is_empty")]
pub option_stats: std::collections::BTreeMap<String, OptionStat>,
/// Fitted item response theory parameters for this set.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub irt: Option<IrtParams>,
/// Machine-detected problems with this set.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub flags: Vec<Flag>,
}
/// What one option has done across every set it has appeared in.
///
/// Deliberately coarse. Averaging selection rates across variants is not
/// meaningful — each is a share of a different set — so `mean_selection_rate`
/// is a summary for reading, not a statistic to act on. `never_chosen` is the
/// one field that carries weight, and it needs several administrations to earn.
#[derive(Debug, Clone, Default, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct OptionHistory {
/// How many distinct variants this option has appeared in.
#[serde(default)]
pub appearances: usize,
/// Examinees who saw it, summed across those variants.
#[serde(default)]
pub n_examinees: usize,
/// Mean of its within-variant selection rates. For reading only.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub mean_selection_rate: Option<f64>,
/// Whether it has never been chosen, anywhere.
#[serde(default, skip_serializing_if = "is_false")]
pub never_chosen: bool,
}
/// How one option behaved.
@@ -661,7 +814,7 @@ pub struct Retirement {
#[derive(Debug, Clone, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct HistoryEntry {
/// The version this change produced.
/// The version this change produced. Ignored since 2.0.
pub version: u32,
/// When it was made.
pub date: Date,
@@ -672,6 +825,32 @@ pub struct HistoryEntry {
pub change: String,
}
/// The separator a pre-2.0 bank-qualified item id used: `b-1-2::q-fastq-line`.
pub const LEGACY_QUALIFIER: &str = "::";
/// An item id with any pre-2.0 bank qualifier removed.
///
/// Until 2.0 an item was named `bank::item`, which made the file it happened to
/// live in part of its identity — and therefore part of the join key on every
/// row of response data ever collected. Moving a question between banks renamed
/// it. Since 2.0 the id names the item course-wide and the bank is only where it
/// is kept, so anything reading an old id strips the qualifier rather than
/// failing to match.
///
/// # Arguments
///
/// * `id` - an item id in either form.
///
/// # Returns
///
/// The part after the qualifier, or the whole id when there is none.
pub fn canonical_id(id: &str) -> &str {
match id.split_once(LEGACY_QUALIFIER) {
Some((_, rest)) => rest,
None => id,
}
}
impl Item {
/// Builds a draft item with everything optional left empty.
///
@@ -725,6 +904,7 @@ impl Item {
author: None,
notes_private: None,
retired: None,
supersedes: None,
}
}
@@ -758,19 +938,38 @@ impl Item {
out
}
/// Looks up an option by letter.
/// Looks up an option by id.
///
/// # Arguments
///
/// * `letter` - the option id, case insensitive.
/// * `id` - the option id. A pre-2.0 letter matches case-insensitively,
/// which a slug never needs but a hand-typed `d` does.
///
/// # Returns
///
/// The option, or `None`.
pub fn option(&self, letter: &str) -> Option<&Choice> {
pub fn option(&self, id: &str) -> Option<&Choice> {
self.options
.iter()
.find(|o| o.id.eq_ignore_ascii_case(letter))
.find(|o| o.id == id)
.or_else(|| self.options.iter().find(|o| o.id.eq_ignore_ascii_case(id)))
}
/// Whether an option id is a pre-2.0 letter rather than a name.
///
/// # Arguments
///
/// * `id` - the option id.
///
/// # Returns
///
/// `true` for `A` through `H`.
pub fn is_legacy_option_id(id: &str) -> bool {
id.len() == 1
&& id
.chars()
.next()
.is_some_and(|c| c.is_ascii_uppercase() && c <= 'H')
}
/// Whether the item keys more than one option.
@@ -834,6 +1033,159 @@ impl Item {
fingerprint(parts.iter().map(|s| s.as_str()))
}
/// The options an assessment administers, in the order the bank declares
/// them.
///
/// Since 2.0 `options` is a *pool*: it may hold several defensible keys and
/// more distractors than any one form shows, and which of them a student
/// saw is a property of the placement rather than of the item. Everything
/// that renders, seals, decodes, or scores an administration has to work
/// from this rather than from `options`, or the paper and the key disagree.
///
/// Bank order, not administered order: the per-form permutation is
/// [`crate::select::option_order`]'s business, and keeping the two separate
/// is what lets one item appear on three forms with one set of statistics.
///
/// # Arguments
///
/// * `key` - the option ids keyed correct for this administration.
/// * `distractors` - the option ids offered alongside them.
///
/// # Returns
///
/// The named options, or the whole live pool when `distractors` is empty.
///
/// `distractors` is what says the set was chosen, not `key`. A pre-2.0
/// record names its key and nothing else — `key: [D]` with no distractor
/// list — and it means "all of them, and D is the right one". Reading that
/// as "administer D alone" would print a one-option paper for every
/// assessment ever recorded.
pub fn administered(&self, key: &[String], distractors: &[String]) -> Vec<&Choice> {
if distractors.is_empty() {
return self
.options
.iter()
.filter(|o| o.retired.is_none())
.collect();
}
self.options
.iter()
.filter(|o| key.contains(&o.id) || distractors.contains(&o.id))
.collect()
}
/// The options that may still be drawn.
///
/// # Returns
///
/// Every option not retired, split into candidate keys and distractors.
pub fn pool(&self) -> (Vec<&Choice>, Vec<&Choice>) {
let live = || self.options.iter().filter(|o| o.retired.is_none());
(
live().filter(|o| o.correct).collect(),
live().filter(|o| !o.correct).collect(),
)
}
/// A digest of the item as one administration showed it.
///
/// The pooling key for statistics, and the reason
/// [`Item::fingerprint`] cannot be. A stem with distractors
/// `{third, second, first}` is a measurably easier item than the same stem
/// with `{third, plus-line, line-two}`, so pooling a p-value across both is
/// averaging two different questions. Covers the stem, the administered
/// options, and which of them was keyed — the last because the same option
/// set with a different key is again a different item.
///
/// # Arguments
///
/// * `key` - the option ids keyed correct for this administration.
/// * `distractors` - the option ids offered alongside them.
///
/// # Returns
///
/// The digest as hex.
pub fn variant_digest(&self, key: &[String], distractors: &[String]) -> String {
let mut parts = vec![self.stem_digest()];
let mut shown: Vec<&Choice> = self.administered(key, distractors);
shown.sort_by(|a, b| a.id.cmp(&b.id));
for option in shown {
let keyed = if key.is_empty() {
option.correct
} else {
key.contains(&option.id)
};
parts.push(format!(
"{}|{}|{}",
option.id,
if keyed { "1" } else { "0" },
option.text.trim()
));
}
fingerprint(parts.iter().map(|s| s.as_str()))
}
/// A digest of what the item asks, without its options.
///
/// The identity check. [`Item::fingerprint`] covers the options too, which
/// is right for calibration — reword a distractor and the pooled selection
/// rates no longer describe what students saw — but wrong for identity,
/// because a question whose distractors changed is still the same question.
/// This covers the stem and the stimulus, and nothing else.
///
/// Compared against the digest each seal recorded, which is what makes
/// "a reworded stem is a new stem" a rule the tool enforces rather than a
/// convention that decays.
///
/// # Returns
///
/// The digest as hex.
pub fn stem_digest(&self) -> String {
let mut parts: Vec<String> = vec![self.stem.trim().to_string()];
if let Some(s) = &self.stimulus {
parts.push(format!("stimulus:{s}"));
}
fingerprint(parts.iter().map(|s| s.as_str()))
}
/// The calibration recorded for one option set.
///
/// # Arguments
///
/// * `variant` - the digest from [`Item::variant_digest`].
///
/// # Returns
///
/// The record, or `None` when this set has not been calibrated.
pub fn calibration_for(&self, variant: &str) -> Option<&VariantCalibration> {
self.calibration
.as_ref()?
.variants
.iter()
.find(|v| v.variant == variant)
}
/// Whether a variant's recorded statistics still describe it.
///
/// Staleness gets *narrower* with a pool rather than wider: rewording one
/// distractor used to invalidate the item's whole calibration, and now it
/// invalidates only the sets that distractor appeared in.
///
/// # Arguments
///
/// * `variant` - the digest to check.
///
/// # Returns
///
/// `false` only when a record exists for that digest and the digest no
/// longer matches what the option ids now say.
pub fn variant_is_current(&self, variant: &str) -> bool {
match self.calibration_for(variant) {
Some(record) => variant == self.variant_digest(&record.key, &record.distractors),
None => true,
}
}
/// Whether the recorded calibration matches the current content.
///
/// # Returns
@@ -889,16 +1241,20 @@ impl Item {
}
}
/// Appends a change-log entry and bumps the version.
/// Appends a change-log entry.
///
/// Kept for the pre-2.0 banks that still carry a `history:` block, so
/// reading one and writing it back does not silently drop entries. New
/// entries belong in a commit message.
///
/// # Arguments
///
/// * `change` - a description of what changed.
/// * `author` - who made the change.
pub fn record_change(&mut self, change: &str, author: Option<&str>) {
self.version += 1;
let version = self.history.iter().map(|h| h.version).max().unwrap_or(0) + 1;
self.history.push(HistoryEntry {
version: self.version,
version,
date: Date::today(),
author: author.map(|a| a.to_string()),
change: change.to_string(),
@@ -906,9 +1262,6 @@ impl Item {
}
}
fn one_u32() -> u32 {
1
}
fn default_format() -> Format {
Format::SingleBestAnswer
}
@@ -938,7 +1291,6 @@ options:
#[test]
fn minimal_item_parses_with_defaults() {
let it = item(MINIMAL);
assert_eq!(it.version, 1);
assert_eq!(it.format, Format::SingleBestAnswer);
assert!(!it.bonus);
assert_eq!(it.key_letters(), vec!["A"]);
@@ -1010,6 +1362,101 @@ options:
assert_eq!(a.fingerprint(), b.fingerprint());
}
#[test]
fn a_pool_administers_a_subset_and_defaults_to_everything() {
let mut it = item(MINIMAL);
let all: Vec<String> = it.options.iter().map(|o| o.id.clone()).collect();
// Unstated means the whole pool, which is what a pre-2.0 record meant.
assert_eq!(it.administered(&[], &[]).len(), all.len());
// A retired option leaves the pool but not the file.
it.options[1].retired = Some(Retirement {
on: Date::new(2026, 9, 20).unwrap(),
reason: "chosen by 1 of 96 across two administrations".into(),
replaced_by: None,
});
let shown = it.administered(&[], &[]);
assert_eq!(shown.len(), all.len() - 1);
assert!(!shown.iter().any(|o| o.id == all[1]));
// Still resolvable: a seal and four terms of rows refer to it.
assert!(it.option(&all[1]).is_some());
// A key with no distractor list is a pre-2.0 record, and it means all
// of them. Reading it as "administer the key alone" would print a
// one-option paper for every assessment already recorded.
assert_eq!(it.administered(&[all[2].clone()], &[]).len(), all.len() - 1);
// Named explicitly, bank order is kept whatever order the lists are in.
let shown = it.administered(&[all[2].clone()], &[all[0].clone()]);
assert_eq!(
shown.iter().map(|o| o.id.clone()).collect::<Vec<_>>(),
vec![all[0].clone(), all[2].clone()]
);
}
#[test]
fn the_variant_digest_tracks_the_option_set_and_the_stem_does_not() {
let it = item(MINIMAL);
let ids: Vec<String> = it.options.iter().map(|o| o.id.clone()).collect();
let one = it.variant_digest(&[ids[0].clone()], &[ids[1].clone()]);
let two = it.variant_digest(&[ids[0].clone()], &[ids[2].clone()]);
// A different distractor is a different item: same stem, different
// difficulty, so pooling a p-value across both would average two
// questions.
assert_ne!(one, two, "a swapped distractor is a new variant");
// The stem is unmoved by any of it.
assert_eq!(it.stem_digest(), item(MINIMAL).stem_digest());
// Order of the lists is not part of the identity.
assert_eq!(
it.variant_digest(&[ids[0].clone()], &[ids[2].clone(), ids[1].clone()]),
it.variant_digest(&[ids[0].clone()], &[ids[1].clone(), ids[2].clone()])
);
}
#[test]
fn a_variant_goes_stale_alone_rather_than_taking_the_item_with_it() {
let mut it = item(MINIMAL);
let ids: Vec<String> = it.options.iter().map(|o| o.id.clone()).collect();
let one = it.variant_digest(&[ids[0].clone()], &[ids[1].clone()]);
let two = it.variant_digest(&[ids[0].clone()], &[ids[2].clone()]);
it.calibration = Some(Calibration {
variants: vec![
VariantCalibration {
variant: one.clone(),
key: vec![ids[0].clone()],
distractors: vec![ids[1].clone()],
n_examinees: Some(96),
p_value: Some(0.84),
..VariantCalibration::default()
},
VariantCalibration {
variant: two.clone(),
key: vec![ids[0].clone()],
distractors: vec![ids[2].clone()],
n_examinees: Some(32),
..VariantCalibration::default()
},
],
..Calibration::default()
});
assert_eq!(it.calibration_for(&one).unwrap().n_examinees, Some(96));
assert!(it.calibration_for("nothing-like-this").is_none());
assert!(it.variant_is_current(&one));
assert!(it.variant_is_current(&two));
// Rewording the option that only the second set used leaves the first
// set's numbers standing. Before the pool, one distractor edit
// invalidated every statistic the item had.
it.options[2].text = "a different distractor".into();
assert!(it.variant_is_current(&one), "the first set never showed it");
assert!(!it.variant_is_current(&two));
}
#[test]
fn stale_calibration_is_detectable() {
let mut it = item(MINIMAL);
@@ -1053,12 +1500,24 @@ options:
}
#[test]
fn record_change_bumps_version_and_logs() {
fn the_stem_is_the_identity_rather_than_a_version_number() {
let mut it = item(MINIMAL);
let before = it.stem_digest();
// A change log entry numbers itself and leaves the item alone: since
// 2.0 nothing reads `version`, and rewording a stem is not a version
// bump but a new item.
it.record_change("clarified the stem", Some("Alex"));
assert_eq!(it.version, 2);
assert_eq!(it.version, 0);
assert_eq!(it.history.len(), 1);
assert_eq!(it.history[0].version, 2);
assert_eq!(it.history[0].version, 1);
assert_eq!(it.stem_digest(), before);
// The options are the fingerprint's business, not the stem's.
it.options[1].text = "a different distractor".into();
assert_eq!(it.stem_digest(), before);
it.stem = "What is y?".into();
assert_ne!(it.stem_digest(), before);
}
#[test]
+46
View File
@@ -3,6 +3,11 @@
// Source: https://git.scient.ing/education/coursebank
//! The on-disk layout of a course directory.
//!
//! Two of these directories hold fragments of the course file rather than files
//! of their own kind: `lectures/` and `objectives/` are merged into one
//! [`crate::course::CourseFile`] on load, along with `references.yaml`. See
//! [`crate::course::fragment`].
use std::path::PathBuf;
@@ -35,11 +40,48 @@ impl Layout {
self.root.join(COURSE_FILE)
}
/// Path to `references.yaml`, the bibliography when it is kept out of
/// `course.yaml`.
///
/// Optional: absent means the course keeps its `references:` section in the
/// course file, which is how an unsplit course is arranged.
pub fn references_file(&self) -> PathBuf {
self.root.join(crate::course::fragment::REFERENCES_FILE)
}
/// Directory holding one file per lecture.
pub fn lectures(&self) -> PathBuf {
self.root.join("lectures")
}
/// Directory holding one file per learning objective.
pub fn objectives(&self) -> PathBuf {
self.root.join("objectives")
}
/// Directory holding item bank YAML files.
pub fn banks(&self) -> PathBuf {
self.root.join("banks")
}
/// Directory holding the statistics, which are kept out of the banks.
///
/// Committed, unlike [`Layout::data`]: everything under here is a cohort
/// aggregate with no student in it. See [`crate::calibration`].
pub fn analysis(&self) -> PathBuf {
self.root.join("analysis")
}
/// The pooled per-item calibration store.
pub fn calibration_file(&self) -> PathBuf {
self.analysis().join(crate::calibration::CALIBRATION_FILE)
}
/// Directory holding one immutable record per administration.
pub fn measurements(&self) -> PathBuf {
self.analysis().join("administrations")
}
/// Directory holding assessment records.
pub fn assessments(&self) -> PathBuf {
self.root.join("assessments")
@@ -93,6 +135,10 @@ impl Layout {
pub fn create_all(&self) -> Result<()> {
for dir in [
self.root.clone(),
self.analysis(),
self.measurements(),
self.lectures(),
self.objectives(),
self.banks(),
self.assessments(),
self.data(),
+128 -4
View File
@@ -124,6 +124,26 @@ pub struct SealMeta {
/// from the file's own contents; [`SealFile::verify`] rebuilds a seal from the
/// live course and compares this against it.
pub digest: String,
/// The digests this file had before it was rewritten, oldest first.
///
/// A seal is meant to be tamper-evident, which puts a migration that
/// renames an id inside it in an awkward position: the id is part of
/// [`SealFile::digest_input`], so rewriting the file invalidates the digest
/// that vouched for it, and recomputing it quietly would produce a record
/// that looks untouched and is not.
///
/// So a migration does both — recomputes the digest and leaves the old one
/// here. What the seal then says is the honest thing: these are the
/// contents, this is what they hash to, and here is what the file hashed to
/// before each rewrite. Anyone holding an earlier copy, a backup or a git
/// revision, can check it against the right entry.
///
/// Deliberately outside [`SealFile::digest_input`]: a field that recorded
/// past digests and was itself covered by the current one could not be
/// appended to without invalidating what it describes.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub superseded_digests: Vec<String>,
}
/// What wrote a seal.
@@ -145,8 +165,16 @@ pub struct SealedItem {
/// The item's global id.
pub item: String,
/// The item version as administered.
/// Retained only so a pre-2.0 seal still loads. Ignored when reasoning
/// about the item, but still written: [`SealFile::digest_input`] covers it,
/// so dropping it on a round trip would invalidate the digest of every
/// administration sealed before 2.0.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub version: Option<u32>,
/// The stem's digest as administered.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub stem_digest: Option<String>,
/// The item's content fingerprint, the same one
/// [`crate::item::Item::fingerprint`] computes, so a seal and a bank can be
/// compared without re-hashing either by hand.
@@ -450,6 +478,7 @@ pub fn build(catalog: &Catalog, record: &AssessmentFile, opts: &Options) -> Resu
content: opts.content,
dropped,
digest: String::new(),
superseded_digests: Vec::new(),
},
items,
forms: sealed_forms,
@@ -484,7 +513,8 @@ fn sealed_item(
SealedItem {
number: placement.number,
item: placement.item.clone(),
version: placement.version.or(Some(item.version)),
version: None,
stem_digest: Some(item.stem_digest()),
fingerprint: item.fingerprint(),
points: placement
.points
@@ -530,7 +560,9 @@ fn sealed_form(catalog: &Catalog, record: &AssessmentFile, form: &Form) -> Resul
for (index, placement) in printed.iter().enumerate() {
let entry = catalog.require(&placement.item)?;
let item = &entry.item;
let n = item.options.len();
// The pool is not the paper: seal what this placement administered.
let shown = item.administered(&placement.key, &placement.distractors);
let n = shown.len();
let order = select::option_order(form, &placement.item, n);
let canonical_key: BTreeSet<String> = if placement.key.is_empty() {
@@ -542,8 +574,7 @@ fn sealed_form(catalog: &Catalog, record: &AssessmentFile, form: &Form) -> Resul
let mut options = Vec::with_capacity(n);
let mut printed_key = Vec::new();
for (position, source_index) in order.iter().enumerate() {
let canonical = item
.options
let canonical = shown
.get(*source_index)
.map(|c| c.id.clone())
.unwrap_or_else(|| printed_letter(*source_index));
@@ -638,6 +669,33 @@ impl SealFile {
yaml::read(path)
}
/// Loads every seal in a directory, oldest administration first.
///
/// # Arguments
///
/// * `dir` - the seals directory.
///
/// # Returns
///
/// The seals, empty when the directory does not exist.
///
/// # Errors
///
/// Propagates load failures, including a seal that does not parse.
pub fn load_all(dir: &Path) -> Result<Vec<SealFile>> {
let mut out = Vec::new();
for path in crate::yaml::list_yaml(dir)? {
out.push(SealFile::load(&path)?);
}
out.sort_by(|a, b| {
a.seal
.date
.cmp(&b.seal.date)
.then(a.seal.assessment.cmp(&b.seal.assessment))
});
Ok(out)
}
/// Loads the seal for an assessment, if one has been written.
///
/// Absence is not an error. A course that has never sealed anything should
@@ -719,6 +777,12 @@ impl SealFile {
self.schema_version, self.seal.assessment, self.seal.course, self.seal.term
));
for item in &self.items {
// Appended only when present: a seal written before 2.0 has to keep
// producing the input it was digested from, or every older
// administration fails verification.
if let Some(stem) = &item.stem_digest {
buf.push_str(&format!("stem\u{1f}{}\u{1f}{stem}\n", item.number));
}
buf.push_str(&format!(
"item\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}",
item.number,
@@ -744,6 +808,39 @@ impl SealFile {
buf
}
/// Renames the items this seal froze, keeping the digest honest.
///
/// The only sanctioned way to rewrite a seal. It recomputes the digest,
/// because the ids are part of what the digest covers, and records the
/// previous one in [`SealMeta::superseded_digests`], because a recomputed
/// digest with no trace of the recompute is a record that claims never to
/// have been touched.
///
/// # Arguments
///
/// * `rename` - a map from old item id to new.
///
/// # Returns
///
/// How many placements were renamed. Zero leaves the file alone, digest
/// included.
pub fn rename_items(&mut self, rename: &BTreeMap<String, String>) -> usize {
let mut renamed = 0;
for item in &mut self.items {
if let Some(new) = rename.get(&item.item) {
item.item = new.clone();
renamed += 1;
}
}
if renamed == 0 {
return 0;
}
let previous = std::mem::take(&mut self.seal.digest);
self.seal.superseded_digests.push(previous);
self.seal.digest = self.recompute_digest();
renamed
}
/// Recomputes this file's digest from its own contents.
///
/// # Returns
@@ -1152,11 +1249,13 @@ mod tests {
content: true,
dropped: Vec::new(),
digest: String::new(),
superseded_digests: Vec::new(),
},
items: vec![SealedItem {
number: 1,
item: "b::q-1".into(),
version: Some(1),
stem_digest: None,
fingerprint: "abc".into(),
points: 1.0,
bonus: false,
@@ -1299,6 +1398,31 @@ mod tests {
assert_eq!(short("sha256:0123456789abcdef"), "sha256:01234567");
}
#[test]
fn renaming_an_item_recomputes_the_digest_and_says_it_did() {
let mut seal = sample();
seal.seal.digest = seal.recompute_digest();
let original = seal.seal.digest.clone();
assert!(seal.check_self().is_empty());
let mut rename = BTreeMap::new();
rename.insert(seal.items[0].item.clone(), "q-renamed".to_string());
assert_eq!(seal.rename_items(&rename), 1);
// The digest covers the ids, so it has to move — and the file has to
// admit that it moved rather than looking untouched.
assert_eq!(seal.items[0].item, "q-renamed");
assert_ne!(seal.seal.digest, original);
assert_eq!(seal.seal.superseded_digests, vec![original]);
assert!(seal.check_self().is_empty(), "{:?}", seal.check_self());
// A rename that matches nothing leaves the file entirely alone.
let steady = seal.seal.digest.clone();
assert_eq!(seal.rename_items(&BTreeMap::new()), 0);
assert_eq!(seal.seal.digest, steady);
assert_eq!(seal.seal.superseded_digests.len(), 1);
}
#[test]
fn round_trips_through_yaml() {
let file = sample();
+1
View File
@@ -9,6 +9,7 @@
//! replaces a dependency that would otherwise need to keep working for as long as
//! a course repository needs to stay readable.
pub mod citation;
pub mod date;
pub mod hash;
pub mod markup;
+330
View File
@@ -0,0 +1,330 @@
// SPDX-License-Identifier: Prosperity-3.0.0
// Copyright Scientific Computing Studio
// Source: https://git.scient.ing/education/coursebank
//! Pulling a citation apart when it was written as prose.
//!
//! A bibliography assembled by hand tends to collect entries like
//!
//! ```text
//! note: 'Nucleic Acids Res 25:3389-3402. doi:10.1093/nar/25.17.3389'
//! ```
//!
//! which is a complete citation in a field that means "anything else worth
//! saying". Nothing can use it: a reading list cannot link the DOI, an export to
//! Hayagriva or BibTeX has no journal to put in `parent` or `journal`, and the
//! `container`, `volume`, `pages`, and `doi` fields sit empty beside it.
//!
//! [`parse`] takes such a note apart. What it cannot account for it leaves in
//! the note, which is the important half of the contract: a note reading
//! `'Bioinformatics 18:440-445. Origin of spaced seeds.'` yields the journal,
//! the volume, the pages, and a note that still says where spaced seeds came
//! from. Nothing is discarded and nothing is invented — an issue number that was
//! never written down stays absent, even when a publisher's DOI happens to
//! encode one.
/// The parts of a citation recovered from a note.
#[derive(Debug, Clone, Default, PartialEq, Eq)]
pub struct Parsed {
/// The journal, proceedings, or book the work appeared in.
pub container: Option<String>,
/// The volume.
pub volume: Option<String>,
/// The page range, as `first-last`.
pub pages: Option<String>,
/// The DOI, bare.
pub doi: Option<String>,
/// What the note still says after the citation is removed.
pub note: Option<String>,
}
/// Takes a citation apart, leaving the rest of the note alone.
///
/// # Arguments
///
/// * `note` - the note as written.
/// * `year` - the record's year, which is how a trailing year is recognized as
/// part of a conference name rather than part of the title of the venue.
///
/// # Returns
///
/// The parts found. Every field is independently optional: a note that carries
/// only a DOI yields only a DOI.
pub fn parse(note: &str, year: Option<u32>) -> Parsed {
let mut out = Parsed::default();
let mut rest = note.trim().to_string();
if let Some((container, volume, pages, tail)) = citation(&rest) {
out.container = Some(container);
out.volume = Some(volume);
out.pages = Some(pages);
rest = tail;
}
if let Some((doi, tail)) = doi(&rest) {
out.doi = Some(doi);
rest = tail;
}
// A venue with no volume or pages — a conference, usually — is named by the
// clause that ends in the year the work was published.
if out.container.is_none() {
if let Some(y) = year {
if let Some((container, tail)) = venue(&rest, y) {
out.container = Some(container);
rest = tail;
}
}
}
// The year belongs to the record, not to the name of the venue.
if let (Some(container), Some(y)) = (&out.container, year) {
let suffix = format!(" {y}");
if let Some(trimmed) = container.strip_suffix(&suffix) {
out.container = Some(trimmed.trim_end().to_string());
}
}
let rest = rest.trim().trim_start_matches('.').trim().to_string();
out.note = (!rest.is_empty()).then_some(rest);
out
}
/// Finds `Journal 25:3389-3402` or `Journal 48, 443-453` at the start.
///
/// The volume is the first digit run that follows a space and is followed by a
/// separator and a page range. Requiring the whole shape is what keeps a year in
/// a conference name (`Proc. FOCS 2000.`) from being read as a volume.
///
/// # Returns
///
/// The container, volume, page range, and whatever followed.
fn citation(text: &str) -> Option<(String, String, String, String)> {
let bytes = text.as_bytes();
let mut at = 0;
while at < bytes.len() {
// A volume follows a space, so that a digit inside a name is not one.
if !(bytes[at].is_ascii_digit() && at > 0 && bytes[at - 1] == b' ') {
at += 1;
continue;
}
let volume_start = at;
let volume_end = digits(bytes, volume_start);
let mut cursor = spaces(bytes, volume_end);
// The separator between volume and pages is a colon or a comma.
if cursor < bytes.len() && (bytes[cursor] == b':' || bytes[cursor] == b',') {
cursor = spaces(bytes, cursor + 1);
let first_start = cursor;
let first_end = digits(bytes, first_start);
if first_end > first_start {
let dash = text[first_end..]
.strip_prefix('-')
.or_else(|| text[first_end..].strip_prefix('\u{2013}'));
if let Some(after_dash) = dash {
let last_offset = text.len() - after_dash.len();
let last_end = digits(bytes, last_offset);
if last_end > last_offset {
let container = text[..volume_start].trim_end_matches([' ', ',']);
if !container.is_empty() {
return Some((
container.to_string(),
text[volume_start..volume_end].to_string(),
format!(
"{}-{}",
&text[first_start..first_end],
&text[last_offset..last_end]
),
text[last_end..].to_string(),
));
}
}
}
}
}
at = volume_end;
}
None
}
/// Finds a `doi:10.…` anywhere in the text.
///
/// # Returns
///
/// The DOI and the text with it removed.
fn doi(text: &str) -> Option<(String, String)> {
let lower = text.to_ascii_lowercase();
let at = lower.find("doi:")?;
let after = text[at + 4..].trim_start();
let offset = text.len() - after.len();
let end = after
.find(char::is_whitespace)
.map(|n| offset + n)
.unwrap_or(text.len());
let doi = text[offset..end].trim_end_matches('.');
if !doi.starts_with("10.") {
return None;
}
let mut remainder = String::from(text[..at].trim_end());
let tail = text[end..].trim();
if !tail.is_empty() {
if !remainder.is_empty() {
remainder.push(' ');
}
remainder.push_str(tail);
}
Some((doi.to_string(), remainder))
}
/// Finds a leading clause ending in the publication year: `Proc. FOCS 2000.`
///
/// # Returns
///
/// The clause without its trailing period, and whatever followed.
fn venue(text: &str, year: u32) -> Option<(String, String)> {
let needle = format!("{year}.");
let at = text.find(&needle)?;
let clause = text[..at + needle.len() - 1].trim();
if clause.is_empty() {
return None;
}
Some((
clause.to_string(),
text[at + needle.len()..].trim().to_string(),
))
}
/// The end of a run of ASCII digits starting at `from`.
fn digits(bytes: &[u8], from: usize) -> usize {
let mut at = from;
while at < bytes.len() && bytes[at].is_ascii_digit() {
at += 1;
}
at
}
/// The end of a run of spaces starting at `from`.
fn spaces(bytes: &[u8], from: usize) -> usize {
let mut at = from;
while at < bytes.len() && bytes[at] == b' ' {
at += 1;
}
at
}
/// Whether a note still looks like it is carrying a citation.
///
/// Used by validation to say so, rather than leaving a note that a reading list
/// cannot link and an export cannot use.
///
/// # Arguments
///
/// * `note` - the note as written.
///
/// # Returns
///
/// `true` when a volume and page range, or a DOI, can be found in it.
pub fn looks_like_a_citation(note: &str) -> bool {
citation(note.trim()).is_some() || doi(note.trim()).is_some()
}
#[cfg(test)]
mod tests {
use super::*;
/// Every article note in a real course bibliography, which is where the
/// shapes below come from. Two separator styles, DOIs in three positions,
/// a conference with no volume, and prose that has to survive.
#[test]
fn a_journal_citation_comes_apart() {
let p = parse(
"Nucleic Acids Res 25:3389-3402. doi:10.1093/nar/25.17.3389",
Some(1997),
);
assert_eq!(p.container.as_deref(), Some("Nucleic Acids Res"));
assert_eq!(p.volume.as_deref(), Some("25"));
assert_eq!(p.pages.as_deref(), Some("3389-3402"));
assert_eq!(p.doi.as_deref(), Some("10.1093/nar/25.17.3389"));
assert_eq!(p.note, None);
// The DOI encodes volume 25, issue 17. Nothing infers the issue from
// it: a field nobody wrote down stays empty.
}
#[test]
fn an_abbreviation_keeps_its_final_period() {
let p = parse("J. Mol. Biol. 48, 443-453.", Some(1970));
assert_eq!(p.container.as_deref(), Some("J. Mol. Biol."));
assert_eq!(p.volume.as_deref(), Some("48"));
assert_eq!(p.pages.as_deref(), Some("443-453"));
assert_eq!(p.note, None);
}
#[test]
fn prose_after_a_citation_stays_in_the_note() {
let p = parse(
"Bioinformatics 18:440-445. Origin of spaced seeds.",
Some(2002),
);
assert_eq!(p.container.as_deref(), Some("Bioinformatics"));
assert_eq!(p.note.as_deref(), Some("Origin of spaced seeds."));
// A caveat the author wrote is the last thing to throw away.
let p = parse("J Mol Biol 215:403-410. Verify before use.", Some(1990));
assert_eq!(p.note.as_deref(), Some("Verify before use."));
let p = parse(
"Bioinformatics 25:2078-2079. doi:10.1093/bioinformatics/btp352. Author list is the \
core set plus the 1000 Genomes Data Processing Subgroup; verify.",
Some(2009),
);
assert_eq!(p.doi.as_deref(), Some("10.1093/bioinformatics/btp352"));
assert_eq!(
p.note.as_deref(),
Some(
"Author list is the core set plus the 1000 Genomes Data Processing Subgroup; \
verify."
)
);
}
#[test]
fn a_conference_has_a_year_where_a_volume_would_be() {
let p = parse(
"Proc. FOCS 2000. doi:10.1109/SFCS.2000.892127. The FM-index. Theory background.",
Some(2000),
);
assert_eq!(p.container.as_deref(), Some("Proc. FOCS"));
// No volume and no pages were written, so none are invented — and the
// 2000 in the DOI is not mistaken for either.
assert_eq!(p.volume, None);
assert_eq!(p.pages, None);
assert_eq!(p.doi.as_deref(), Some("10.1109/SFCS.2000.892127"));
assert_eq!(p.note.as_deref(), Some("The FM-index. Theory background."));
}
#[test]
fn a_note_with_nothing_to_find_is_left_whole() {
let p = parse("Origin of the MAPQ score.", Some(2008));
assert_eq!(p.container, None);
assert_eq!(p.note.as_deref(), Some("Origin of the MAPQ score."));
}
#[test]
fn a_multi_word_journal_is_not_cut_at_a_number() {
let p = parse("Advances in Mathematics 20, 367-387.", Some(1976));
assert_eq!(p.container.as_deref(), Some("Advances in Mathematics"));
assert_eq!(p.volume.as_deref(), Some("20"));
}
#[test]
fn what_validation_looks_for() {
assert!(looks_like_a_citation("Nat Methods 12:59-60."));
assert!(looks_like_a_citation("doi:10.1038/nmeth.3176"));
assert!(!looks_like_a_citation("Origin of minimizers."));
assert!(!looks_like_a_citation("Verify before use."));
// A page range with no volume is not a citation shape.
assert!(!looks_like_a_citation("see pages 12-14"));
}
}
+48 -1
View File
@@ -16,7 +16,7 @@ use std::fs;
use std::path::Path;
use serde::de::{self, DeserializeOwned, Visitor};
use serde::{Deserializer, Serialize};
use serde::{Deserialize, Deserializer, Serialize};
use crate::error::{Error, Result};
@@ -61,6 +61,26 @@ pub fn write<T: Serialize>(path: &Path, value: &T) -> Result<()> {
fs::write(path, text).map_err(|e| Error::io(path, e))
}
/// Serializes a value to a YAML string.
///
/// Used where the caller needs to put something in front of the document, such
/// as the banner on a generated file.
///
/// # Arguments
///
/// * `value` - the value to serialize.
///
/// # Returns
///
/// The YAML text.
///
/// # Errors
///
/// Returns [`Error::Other`] if the value cannot be represented as YAML.
pub fn to_string<T: Serialize>(value: &T) -> Result<String> {
serde_yaml_ng::to_string(value).map_err(Error::other)
}
/// Deserializes a JSON file into any type.
///
/// Used only for importing legacy banks and for reading emitted schemas back in
@@ -202,6 +222,33 @@ where
d.deserialize_any(V)
}
/// Deserializes an optional scalar as a string, quoted or not.
///
/// The [`flexible_string`] of a field that may be absent, which is what a
/// fragment's `schema_version` is: one file in a course declares it and the
/// rest inherit.
///
/// # Arguments
///
/// * `d` - the deserializer.
///
/// # Returns
///
/// The value as a string, or `None`.
///
/// # Errors
///
/// Returns a deserialization error for non-scalar input.
pub fn flexible_string_opt<'de, D>(d: D) -> std::result::Result<Option<String>, D::Error>
where
D: Deserializer<'de>,
{
#[derive(serde::Deserialize)]
struct Wrapper(#[serde(deserialize_with = "flexible_string")] String);
Ok(Option::<Wrapper>::deserialize(d)?.map(|w| w.0))
}
#[cfg(test)]
mod tests {
use super::*;