Compare commits

...
2 Commits
Author SHA1 Message Date
alexm 5ac1e317c0 feat: cooked up something fierce
Pipeline / check (pull_request) Successful in 3m9s
Pipeline / docs (pull_request) Skipped
Pipeline / nightly (pull_request) Skipped
Pipeline / release (pull_request) Skipped
2026-09-26 18:22:25 -04:00
alexm eabc98ad31 feat: splitting 2026-09-26 01:15:12 -04:00
37 changed files with 6328 additions and 278 deletions
+279 -1
View File
@@ -35,7 +35,7 @@ use crate::classical::{self, Analysis, ItemAnalysis, Thresholds};
use crate::date::Date;
use crate::error::{Error, Result};
use crate::irt::{self, Fit};
use crate::item::{Calibration, IrtParams, OptionStat};
use crate::item::{Calibration, IrtParams, Item, OptionStat, VariantCalibration};
use crate::responses::ResponseSet;
use crate::store::Store;
use crate::taxonomy::Flag;
@@ -267,6 +267,35 @@ pub fn plan(catalog: &Catalog, store: &Store, opts: &Options) -> Result<Plan> {
));
}
// Splitting by option set means an item administered three times with three
// different sets has three cells of 24 rather than one of 72. That is the
// honest picture, and it is worth saying out loud rather than leaving
// someone to read an IRT fit that was never possible.
for (uid, appearances) in &by_item {
let mut sizes: Vec<usize> = Vec::new();
for (_, analysis) in appearances {
if analysis.variant.is_some() {
sizes.push(analysis.n);
}
}
if sizes.len() > 1 {
let distinct: BTreeSet<&str> = appearances
.iter()
.filter_map(|(_, a)| a.variant.as_deref())
.collect();
if distinct.len() > 1 {
warnings.push(format!(
"{uid}: {} option sets across {} administrations, largest n = {}. \
Statistics are kept per set, because a stem shown with different \
distractors is a different item.",
distinct.len(),
appearances.len(),
sizes.iter().copied().max().unwrap_or(0)
));
}
}
}
let mut changes = Vec::new();
for (uid, appearances) in &by_item {
@@ -304,6 +333,12 @@ pub fn plan(catalog: &Catalog, store: &Store, opts: &Options) -> Result<Plan> {
option_stats: pooled.option_stats.clone(),
irt: irt_params,
flags: pooled.flags.clone(),
variants: variant_records(&entry.item, appearances, previous_variants(entry)),
options: BTreeMap::new(),
};
let calibration = Calibration {
options: option_histories(&calibration.variants),
..calibration
};
let previous = entry.item.calibration.as_ref();
@@ -703,6 +738,16 @@ pub fn plan_from_analysis(
.collect(),
irt: irt_params,
flags: item.flags.clone(),
variants: variant_records(
&entry.item,
&[(admin.clone(), item.clone())],
previous_variants(entry),
),
options: BTreeMap::new(),
};
let calibration = Calibration {
options: option_histories(&calibration.variants),
..calibration
};
let previous = entry.item.calibration.as_ref();
@@ -730,6 +775,171 @@ pub fn plan_from_analysis(
}
}
/// The variant records an item already has, to be merged with the new ones.
fn previous_variants(entry: &crate::catalog::Entry) -> Vec<VariantCalibration> {
entry
.item
.calibration
.as_ref()
.map(|c| c.variants.clone())
.unwrap_or_default()
}
/// Builds one calibration record per option set the item was administered in.
///
/// The records this pass computes replace the stored ones for the same variant
/// and leave the rest alone. That is what makes a partial recalibration safe:
/// a pass given only this term's data must not silently discard the numbers for
/// an option set that was retired two terms ago.
///
/// Statistics come from the same [`pool`] used for the flat summary, so the two
/// agree for an item whose pool is its form — the case every pre-2.0 item is in.
///
/// # Arguments
///
/// * `item` - the bank item.
/// * `appearances` - the administration id and analysis of each appearance.
/// * `previous` - the records already stored.
///
/// # Returns
///
/// The merged records, in variant order.
fn variant_records(
item: &Item,
appearances: &[(String, ItemAnalysis)],
previous: Vec<VariantCalibration>,
) -> Vec<VariantCalibration> {
// Appearances whose administration mixed two option sets under one question
// number carry no variant, and there is no set for them to describe.
let mut by_variant: BTreeMap<String, Vec<(String, ItemAnalysis)>> = BTreeMap::new();
for (admin, analysis) in appearances {
if let Some(variant) = &analysis.variant {
by_variant
.entry(variant.clone())
.or_default()
.push((admin.clone(), analysis.clone()));
}
}
let mut merged: BTreeMap<String, VariantCalibration> = previous
.into_iter()
.map(|v| (v.variant.clone(), v))
.collect();
for (variant, group) in by_variant {
let pooled = pool(&group);
// The option set is recovered from the digest's own record when the
// stored one has it, and from the options that were actually chosen
// otherwise, so a record written from data alone still says what it
// describes.
let (key, distractors) = describe(item, &variant, &group, merged.get(&variant));
merged.insert(
variant.clone(),
VariantCalibration {
variant,
key,
distractors,
administrations: group.iter().map(|(a, _)| a.clone()).collect(),
n_examinees: Some(pooled.n),
p_value: Some(round4(pooled.p_value)),
point_biserial: pooled.point_biserial.map(round4),
discrimination_index: pooled.discrimination_index.map(round4),
option_stats: pooled.option_stats.clone(),
irt: None,
flags: pooled.flags.clone(),
},
);
}
merged.into_values().collect()
}
/// Which options a variant administered.
///
/// Prefers what a stored record already says. Failing that, the options that
/// appear in the statistics are the ones students saw, and the item says which
/// of those are keyed.
fn describe(
item: &Item,
variant: &str,
group: &[(String, ItemAnalysis)],
stored: Option<&VariantCalibration>,
) -> (Vec<String>, Vec<String>) {
if let Some(stored) = stored {
if !stored.key.is_empty()
&& item.variant_digest(&stored.key, &stored.distractors) == variant
{
return (stored.key.clone(), stored.distractors.clone());
}
}
let mut seen: BTreeSet<String> = BTreeSet::new();
for (_, analysis) in group {
seen.extend(analysis.options.keys().cloned());
}
let keyed: BTreeSet<String> = group
.iter()
.flat_map(|(_, a)| a.key.iter().cloned())
.collect();
let key: Vec<String> = seen
.iter()
.filter(|o| keyed.contains(*o))
.cloned()
.collect();
let distractors: Vec<String> = seen
.iter()
.filter(|o| !keyed.contains(*o))
.cloned()
.collect();
(key, distractors)
}
/// Summarizes what each option has done across every set it appeared in.
///
/// Deliberately coarse, because selection rates are shares of a fixed set and
/// averaging them across different sets is not a statistic. The one claim this
/// view supports is the one worth having: an option that draws nobody in any
/// set it has appeared in is not doing anything, and that is the evidence for
/// retiring it — evidence a single administration cannot provide.
///
/// # Arguments
///
/// * `variants` - the per-variant records.
///
/// # Returns
///
/// One history per option id.
fn option_histories(
variants: &[VariantCalibration],
) -> BTreeMap<String, crate::item::OptionHistory> {
let mut out: BTreeMap<String, crate::item::OptionHistory> = BTreeMap::new();
let mut rates: BTreeMap<String, Vec<f64>> = BTreeMap::new();
for variant in variants {
let n = variant.n_examinees.unwrap_or(0);
for (option, stat) in &variant.option_stats {
let entry = out.entry(option.clone()).or_default();
entry.appearances += 1;
entry.n_examinees += n;
if let Some(rate) = stat.selection_rate {
rates.entry(option.clone()).or_default().push(rate);
}
}
}
for (option, entry) in &mut out {
if let Some(seen) = rates.get(option) {
if !seen.is_empty() {
entry.mean_selection_rate =
Some(round4(seen.iter().sum::<f64>() / seen.len() as f64));
entry.never_chosen = seen.iter().all(|r| *r <= f64::EPSILON);
}
}
}
out
}
/// Rounds to four decimals.
fn round4(x: f64) -> f64 {
(x * 1e4).round() / 1e4
@@ -754,9 +964,77 @@ mod tests {
option_stats: BTreeMap::new(),
irt: None,
flags: Vec::new(),
variants: Vec::new(),
options: BTreeMap::new(),
}
}
fn stat(rate: f64) -> OptionStat {
OptionStat {
selection_rate: Some(rate),
point_biserial: None,
upper_group_rate: None,
lower_group_rate: None,
}
}
fn variant(id: &str, n: usize, dead_rate: f64) -> VariantCalibration {
VariantCalibration {
variant: id.into(),
n_examinees: Some(n),
option_stats: [
("o-key".to_string(), stat(0.8)),
("o-dead".to_string(), stat(dead_rate)),
]
.into_iter()
.collect(),
..VariantCalibration::default()
}
}
#[test]
fn an_option_that_draws_nobody_is_only_visible_across_sets() {
let histories = option_histories(&[variant("v1", 50, 0.0), variant("v2", 46, 0.0)]);
let dead = &histories["o-dead"];
assert_eq!(dead.appearances, 2);
assert_eq!(dead.n_examinees, 96);
// The claim the cross-variant view exists to support, and the one a
// single administration cannot make.
assert!(dead.never_chosen);
assert!(!histories["o-key"].never_chosen);
assert_eq!(histories["o-key"].mean_selection_rate, Some(0.8));
// One set where it drew is enough to stop the claim.
let mixed = option_histories(&[variant("v1", 50, 0.0), variant("v2", 46, 0.04)]);
assert!(!mixed["o-dead"].never_chosen);
}
#[test]
fn a_pass_with_no_data_for_an_item_keeps_the_records_it_has() {
let item: Item = serde_yaml_ng::from_str(
r#"id: q-x
status: approved
level: 1
stem: s
options:
- { id: o-key, text: right, correct: true }
- { id: o-one, text: wrong }
"#,
)
.expect("item parses");
let kept = variant_records(&item, &[], vec![variant("v-old", 96, 0.0)]);
assert_eq!(
kept.len(),
1,
"a pass given nothing must not delete history"
);
assert_eq!(kept[0].variant, "v-old");
assert_eq!(kept[0].n_examinees, Some(96));
}
#[test]
fn a_first_calibration_is_all_new() {
let next = calibration(0.7, Some(0.3), 24, "abc");
+29
View File
@@ -134,6 +134,14 @@ pub struct ItemAnalysis {
pub number: u32,
/// The item's global id, when known.
pub item_ref: Option<String>,
/// The option set administered, when the rows agree on one.
///
/// Within one administration an item has one variant, because a placement's
/// distractors are drawn once and shared by every form — only the printed
/// order differs. `None` means the rows disagreed, which happens when a
/// course opts into drawing distractors per form; the statistics below then
/// describe a mixture and cannot be pooled by option set.
pub variant: Option<String>,
/// How many students the item was administered to.
pub n: usize,
/// How many gave a non-blank response.
@@ -553,6 +561,7 @@ pub fn analyze(
item_ref: record
.and_then(|r| r.placement(*number))
.map(|p| p.item.clone()),
variant: one_variant(set, *number),
n,
n_answered,
blank_rate: blank as f64 / responded as f64,
@@ -906,6 +915,25 @@ fn reliability(coded: &[Vec<f64>], totals: &[f64], p_values: &[f64], rpbs: &[f64
/// # Returns
///
/// The letters that appear on full-credit responses.
/// The single variant every row for one question names, if they agree.
///
/// Disagreement is not an error, it is a fact about the administration: a course
/// that draws distractors per form has two option sets under one question
/// number, and no pooled statistic describes both. Returning `None` is what
/// keeps the per-variant records from claiming otherwise.
fn one_variant(set: &ResponseSet, number: u32) -> Option<String> {
let mut seen: Option<&str> = None;
for row in set.rows.iter().filter(|r| r.item_number == number) {
let variant = row.variant.as_deref()?;
match seen {
None => seen = Some(variant),
Some(first) if first == variant => {}
Some(_) => return None,
}
}
seen.map(str::to_string)
}
fn infer_key(rows: &[&crate::responses::Response]) -> Vec<String> {
let mut out: BTreeSet<String> = BTreeSet::new();
for r in rows {
@@ -1001,6 +1029,7 @@ mod tests {
item_number: number,
item_ref: None,
item_version: None,
variant: None,
selected: if letter.is_empty() {
vec![]
} else {
+4 -18
View File
@@ -71,7 +71,7 @@ use serde::Serialize;
use crate::assessment::AssessmentFile;
use crate::catalog::Catalog;
use crate::classical::Analysis;
use crate::course::{CourseFile, ReadingRole, Reference};
use crate::course::{CourseFile, ReadingRole};
use crate::irt::Fit;
use crate::item::Citation;
use crate::responses::{Response, ResponseSet};
@@ -797,9 +797,9 @@ fn item_readings(course: &CourseFile, citations: &[Citation]) -> Vec<ItemReading
let (label, title, url) = match reference {
Some((key, reference)) => (
reference.label.as_deref().unwrap_or(key).to_string(),
reference.label_or(key).to_string(),
Some(reference.title.clone()),
resolve_citation_url(citation, reference),
citation.href(reference),
),
None => (citation.display(), None, citation.url.clone()),
};
@@ -821,20 +821,6 @@ fn item_readings(course: &CourseFile, citations: &[Citation]) -> Vec<ItemReading
out
}
/// A citation's own URL, else the reference's `base_url` joined with its `path`.
fn resolve_citation_url(citation: &Citation, reference: &Reference) -> Option<String> {
if let Some(url) = &citation.url {
return Some(url.clone());
}
let path = citation.path.as_deref()?;
let base = reference.base_url.as_deref()?;
Some(match (base.ends_with('/'), path.starts_with('/')) {
(true, true) => format!("{base}{}", &path[1..]),
(false, false) => format!("{base}/{path}"),
_ => format!("{base}{path}"),
})
}
/// Ranks the lectures behind a student's missed questions.
///
/// A lecture earns its place by how many distinct objectives went wrong in it,
@@ -1694,7 +1680,7 @@ pub fn cohort(
.map(|option| {
let count = responses
.iter()
.filter(|r| r.chosen().iter().any(|l| *l == option.id))
.filter(|r| r.chosen().contains(&option.id))
.count();
OptionRow {
text: Some(option.text.clone()),
+1
View File
@@ -1363,6 +1363,7 @@ learning_objectives:
item_number: number,
item_ref: None,
item_version: None,
variant: None,
selected: vec!["A".into()],
selected_source: vec![],
eliminated: vec![],
+231 -11
View File
@@ -31,6 +31,12 @@ const BASE: &str = "https://coursebank.dev/schema";
pub enum Kind {
/// `course.yaml`.
Course,
/// `references.yaml`.
References,
/// `lectures/*.yaml`.
Lecture,
/// `objectives/*.yaml`.
Objective,
/// `banks/*.yaml`.
Bank,
/// `assessments/*.yaml`.
@@ -38,13 +44,23 @@ pub enum Kind {
}
impl Kind {
/// All three kinds.
pub const ALL: [Kind; 3] = [Kind::Course, Kind::Bank, Kind::Assessment];
/// Every kind.
pub const ALL: [Kind; 6] = [
Kind::Course,
Kind::References,
Kind::Lecture,
Kind::Objective,
Kind::Bank,
Kind::Assessment,
];
/// The file name a schema is written to.
pub fn filename(self) -> &'static str {
match self {
Kind::Course => "course.schema.json",
Kind::References => "references.schema.json",
Kind::Lecture => "lecture.schema.json",
Kind::Objective => "objective.schema.json",
Kind::Bank => "bank.schema.json",
Kind::Assessment => "assessment.schema.json",
}
@@ -80,12 +96,15 @@ impl Kind {
pub fn schema(kind: Kind) -> Value {
match kind {
Kind::Course => course_schema(),
Kind::References => references_schema(),
Kind::Lecture => lecture_fragment_schema(),
Kind::Objective => objective_fragment_schema(),
Kind::Bank => bank_schema(),
Kind::Assessment => assessment_schema(),
}
}
/// Writes all three schemas to a directory.
/// Writes every schema to a directory.
///
/// # Arguments
///
@@ -328,6 +347,13 @@ fn lecture_schema() -> Value {
"date": date("Date delivered."),
"unit": { "type": "string", "description": "Unit id." },
"slides_url": { "type": "string" },
"teaches": {
"type": "array",
"description": "The objectives this session develops. Each named objective gains \
this lecture in its `lectures` list when the course is loaded, \
so the pair is declared once, here, while planning the lecture.",
"items": { "type": "string" }
},
"readings": {
"type": "array",
"description": "Readings assigned with this lecture, in the order you assign \
@@ -434,7 +460,23 @@ fn reference_schema() -> Value {
"volume": { "type": "string" },
"issue": { "type": "string" },
"pages": { "type": "string", "description": "Pages of the work, not of a reading." },
"doi": { "type": "string", "description": "Bare DOI: 10.1038/nature12373." },
"doi": {
"type": "string",
"pattern": "^(doi:|https?://(dx\\.)?doi\\.org/)?10\\.",
"description": "Bare DOI: 10.1038/nature12373. For a manuscript this is \
usually the only link worth storing, since a reading list \
resolves it to doi.org."
},
"arxiv": { "type": "string", "description": "Bare arXiv id: 2301.00001." },
"pmcid": {
"type": "string",
"description": "PubMed Central id, which hosts the full text: PMC3084216."
},
"pmid": {
"type": "string",
"pattern": "^[0-9]+$",
"description": "PubMed id, which hosts a record about the work: 21471563."
},
"isbn": { "type": "string" },
"url": { "type": "string", "description": "Canonical URL for the whole work." },
"base_url": {
@@ -568,6 +610,94 @@ fn course_schema() -> Value {
})
}
/// The schema for `references.yaml`.
fn references_schema() -> Value {
json!({
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": format!("{BASE}/references.schema.json"),
"title": "coursebank references file",
"description": "The works the course cites, by citation key. One fragment of the \
course file; see course.schema.json for the whole.",
"type": "object",
"additionalProperties": false,
"properties": {
"schema_version": {
"type": ["string", "number"],
"description": format!("Format version; currently {SCHEMA_VERSION}. Declared in \
course.yaml; fragments inherit it.")
},
"references": {
"type": "object",
"description": "Works by citation key.",
"additionalProperties": reference_schema()
}
}
})
}
/// The schema for one file under `lectures/`.
fn lecture_fragment_schema() -> Value {
json!({
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": format!("{BASE}/lecture.schema.json"),
"title": "coursebank lecture file",
"description": "One session: its readings, and the objectives it develops. A fragment \
of the course file, merged on load.",
"type": "object",
"required": ["lectures"],
"additionalProperties": false,
"properties": {
"schema_version": {
"type": ["string", "number"],
"description": "Declared in course.yaml; fragments inherit it."
},
"lectures": {
"type": "object",
"description": "Keyed by lecture id, conventionally one entry per file.",
"additionalProperties": lecture_schema()
}
}
})
}
/// The schema for one file under `objectives/`.
fn objective_fragment_schema() -> Value {
json!({
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": format!("{BASE}/objective.schema.json"),
"title": "coursebank objective file",
"description": "One learning objective and the learning targets it decomposes into. A \
fragment of the course file, merged on load.",
"type": "object",
"required": ["learning_objectives"],
"additionalProperties": false,
"properties": {
"schema_version": {
"type": ["string", "number"],
"description": "Declared in course.yaml; fragments inherit it."
},
"learning_objectives": {
"type": "object",
"description": "Keyed by objective id, conventionally one entry per file. Its \
`lectures` list is derived from each lecture's `teaches`, so \
leave it out unless you prefer to declare it here.",
"additionalProperties": objective_schema()
},
"learning_targets": {
"type": "object",
"description": "The targets of this file's objective, by id. A target with no \
`lectures` of its own inherits its objective's.",
"additionalProperties": target_schema()
},
"stimuli": {
"type": "object",
"description": "Shared passages, figures, or data that several items refer to.",
"additionalProperties": stimulus_schema()
}
}
})
}
/// One option's schema.
///
/// Split out from [`item_schema`] rather than inlined, because `serde_json`'s
@@ -583,9 +713,11 @@ fn option_schema() -> Value {
"properties": {
"id": {
"type": "string",
"pattern": "^[A-H]$",
"description": "Option letter. Identity, not print position — shuffled forms \
relabel on the way out."
"pattern": "^(o-[a-z0-9]+(-[a-z0-9]+)*|[A-H])$",
"description": "Option id, unique within the item: `o-fourth-line`. An \
identity, not a print position — shuffled forms relabel on the \
way out. A single letter A-H is the pre-2.0 form; \
`coursebank migrate options` renames it."
},
"text": text("The option as a student reads it."),
"correct": { "type": "boolean" },
@@ -617,7 +749,8 @@ fn option_schema() -> Value {
},
"selection_rate_expected": proportion(
"How often you expect this to be chosen. Compared against reality."
)
),
"retired": retirement_schema()
}
})
}
@@ -812,6 +945,21 @@ fn calibration_schema() -> Value {
"additionalProperties": option_stat_schema()
},
"irt": irt_schema(),
"variants": {
"type": "array",
"description": "One record per option set ever administered. A stem shown with \
different distractors is a different item, so a p-value pooled \
across both would average two questions.",
"items": variant_calibration_schema()
},
"options": {
"type": "object",
"description": "One record per option, pooled across every set it appeared in. \
Supports one claim — this option draws nobody, anywhere — which \
is what retires a distractor and what one administration cannot \
show.",
"additionalProperties": option_history_schema()
},
"flags": {
"type": "array",
"items": { "type": "string", "enum": strings(&flags) }
@@ -853,6 +1001,63 @@ fn history_schema() -> Value {
})
}
/// The schema for one option set's statistics.
fn variant_calibration_schema() -> Value {
json!({
"type": "object",
"required": ["variant"],
"additionalProperties": false,
"properties": {
"variant": {
"type": "string",
"description": "Digest of the stem, the administered options, and which was keyed."
},
"key": string_array("The option ids keyed correct in this set."),
"distractors": string_array("The option ids offered alongside them."),
"administrations": string_array("The administrations pooled into these numbers."),
"n_examinees": { "type": "integer", "minimum": 0 },
"p_value": proportion("Proportion correct, for this option set only."),
"point_biserial": { "type": "number", "minimum": -1.0, "maximum": 1.0 },
"discrimination_index": { "type": "number", "minimum": -1.0, "maximum": 1.0 },
"option_stats": {
"type": "object",
"description": "Per-option behaviour within this set, by option id.",
"additionalProperties": option_stat_schema()
},
"irt": irt_schema(),
"flags": {
"type": "array",
"items": {
"type": "string",
"enum": strings(
&Flag::ALL.iter().map(|f| f.as_str()).collect::<Vec<&str>>()
)
}
}
}
})
}
/// The schema for one option's cross-variant history.
fn option_history_schema() -> Value {
json!({
"type": "object",
"additionalProperties": false,
"properties": {
"appearances": { "type": "integer", "minimum": 0 },
"n_examinees": { "type": "integer", "minimum": 0 },
"mean_selection_rate": proportion(
"Mean of the within-variant rates. For reading, not for acting on: each rate is \
a share of a different set."
),
"never_chosen": {
"type": "boolean",
"description": "Never chosen, anywhere. The claim that justifies retiring it."
}
}
})
}
/// The schema for a retirement record.
fn retirement_schema() -> Value {
json!({
@@ -1157,7 +1362,21 @@ fn placement_schema() -> Value {
},
"points": { "type": "number", "minimum": 0.0 },
"bonus": { "type": "boolean" },
"key": string_array("Keyed option letters as administered."),
"key": string_array(
"The option ids keyed correct for this administration. One for a \
single_best_answer, chosen from the item's pool of defensible keys."
),
"distractors": string_array(
"The option ids offered alongside the key. Resolved when the assessment is \
assembled and written out explicitly, so a later bank edit cannot change the \
paper. Empty means the whole pool."
),
"variant": {
"type": "string",
"description": "Digest of the item as this administration showed it: stem, \
administered options, and which was keyed. The key statistics \
pool on."
},
"level": level(),
"learning_targets": string_array("Targets as administered."),
"credit_overrides": {
@@ -1260,7 +1479,8 @@ mod tests {
assert_eq!(props["options"]["maxItems"], 8);
assert_eq!(
props["options"]["items"]["properties"]["id"]["pattern"],
"^[A-H]$"
// Either form: the 2.0 name, or the letter it replaces.
"^(o-[a-z0-9]+(-[a-z0-9]+)*|[A-H])$"
);
}
@@ -1283,7 +1503,7 @@ mod tests {
let dir = std::env::temp_dir().join(format!("cb-schema-{}", std::process::id()));
std::fs::remove_dir_all(&dir).ok();
let written = write_all(&dir).unwrap();
assert_eq!(written.len(), 3);
assert_eq!(written.len(), Kind::ALL.len());
for path in &written {
assert!(path.exists());
let text = std::fs::read_to_string(path).unwrap();
+162 -2
View File
@@ -88,6 +88,16 @@ pub enum Rule {
/// An expectation of low discrimination on a higher-level item.
ContradictoryDesign,
// --- the option pool ---
/// Fewer usable distractors than a form shows.
ThinOptionPool,
/// An option that has never been administered and has not been retired.
UnusedOption,
/// An option retired without saying what it did.
UnjustifiedRetirement,
/// A distractor that has never been chosen, in any set it appeared in.
NonfunctioningDistractor,
// --- evidence ---
/// Statistics describe an older version of the item.
StaleCalibration,
@@ -104,7 +114,7 @@ impl Rule {
///
/// Used by `--list-rules`, and by the test that keeps this list in step with
/// the enum.
pub const ALL: [Rule; 26] = [
pub const ALL: [Rule; 30] = [
Rule::KeyIsLongest,
Rule::UnevenOptionLength,
Rule::WordRepeatCue,
@@ -127,6 +137,10 @@ impl Rule {
Rule::WeakFormatForLevel,
Rule::ScoredBonusLevel,
Rule::ContradictoryDesign,
Rule::ThinOptionPool,
Rule::UnusedOption,
Rule::UnjustifiedRetirement,
Rule::NonfunctioningDistractor,
Rule::StaleCalibration,
Rule::DifficultyMissed,
Rule::DiscriminationMissed,
@@ -152,6 +166,9 @@ impl Rule {
// Statistics attached to text that has since changed are actively
// misleading, which is worse than absent.
R::StaleCalibration => Severity::High,
// An item that cannot fill a form is an item `assemble` will put on
// a paper short an option.
R::ThinOptionPool => Severity::High,
// An unanswerable question for a screen-reader user.
R::AssetWithoutAltText => Severity::High,
@@ -178,6 +195,13 @@ impl Rule {
| R::NoStudentFeedback
| R::DifficultyMissed
| R::DiscriminationMissed => Severity::Low,
// A distractor that draws nobody across several administrations is
// evidence to act on, not a style note.
R::NonfunctioningDistractor => Severity::Medium,
// Both are tidiness: the item still works, but its pool is
// carrying something nobody has accounted for.
R::UnusedOption | R::UnjustifiedRetirement => Severity::Low,
}
}
@@ -206,6 +230,10 @@ impl Rule {
Rule::WeakFormatForLevel => "complete-format-level",
Rule::ScoredBonusLevel => "complete-bonus-policy",
Rule::ContradictoryDesign => "complete-design-conflict",
Rule::ThinOptionPool => "pool-thin",
Rule::UnusedOption => "pool-unused",
Rule::UnjustifiedRetirement => "pool-unjustified-retirement",
Rule::NonfunctioningDistractor => "evidence-nonfunctioning",
Rule::StaleCalibration => "evidence-stale",
Rule::DifficultyMissed => "evidence-difficulty",
Rule::DiscriminationMissed => "evidence-discrimination",
@@ -238,9 +266,11 @@ impl Rule {
| Rule::WeakFormatForLevel
| Rule::ScoredBonusLevel
| Rule::ContradictoryDesign => "completeness",
Rule::ThinOptionPool | Rule::UnusedOption | Rule::UnjustifiedRetirement => "pool",
Rule::StaleCalibration
| Rule::DifficultyMissed
| Rule::DiscriminationMissed
| Rule::NonfunctioningDistractor
| Rule::DuplicateStem => "evidence",
}
}
@@ -270,6 +300,10 @@ impl Rule {
Rule::WeakFormatForLevel,
Rule::ScoredBonusLevel,
Rule::ContradictoryDesign,
Rule::ThinOptionPool,
Rule::UnusedOption,
Rule::UnjustifiedRetirement,
Rule::NonfunctioningDistractor,
Rule::StaleCalibration,
Rule::DifficultyMissed,
Rule::DiscriminationMissed,
@@ -302,6 +336,12 @@ impl Rule {
Rule::WeakFormatForLevel => "true/false at an analytic level",
Rule::ScoredBonusLevel => "a level the policy reserves for bonus is scored",
Rule::ContradictoryDesign => "low expected discrimination on a higher-level item",
Rule::ThinOptionPool => "fewer usable distractors than a form shows",
Rule::UnusedOption => "an option has never been administered and is not retired",
Rule::UnjustifiedRetirement => "an option was retired without saying what it did",
Rule::NonfunctioningDistractor => {
"a distractor has never been chosen in any set it appeared in"
}
Rule::StaleCalibration => "statistics describe an older version of the item",
Rule::DifficultyMissed => "observed difficulty was far from predicted",
Rule::DiscriminationMissed => "observed discrimination contradicted the prediction",
@@ -749,6 +789,87 @@ pub fn lint_item(entry: &Entry, course: &CourseFile, t: &Thresholds) -> Vec<Find
}
}
// --- the option pool
// These only make sense once options are a pool, and the pool is where an
// item's spare parts sit. A bank that never draws from it will not trip any
// of them.
if it.format.has_options() {
let (keys, distractors) = it.pool();
let wanted = course.policy.options_per_item.saturating_sub(1);
if distractors.len() < wanted {
push(
Rule::ThinOptionPool,
Severity::High,
format!(
"has {} usable distractor(s) but a form shows {}, so `assemble` will put \
this on a paper an option short",
distractors.len(),
course.policy.options_per_item
),
);
}
for option in &it.options {
match &option.retired {
Some(retirement) => {
if retirement.reason.trim().len() < 12 {
push(
Rule::UnjustifiedRetirement,
Severity::Low,
format!(
"option `{}` is retired with no real reason. The reason is the \
finding — what it drew, or failed to draw — and it is the only \
part of a retirement worth anything in two years",
option.id
),
);
}
}
None => {
// An option nobody has been shown is a draft, and a draft
// sitting in an approved item's pool will eventually be
// drawn onto a paper without ever having been reviewed
// against data.
if let Some(history) = it
.calibration
.as_ref()
.filter(|c| !c.options.is_empty())
.and_then(|c| c.options.get(&option.id))
{
if history.never_chosen && history.appearances > 1 {
push(
Rule::NonfunctioningDistractor,
Severity::Medium,
format!(
"option `{}` has appeared in {} option set(s) across {} \
examinees and has never been chosen. One administration \
would not show this; several do.",
option.id, history.appearances, history.n_examinees
),
);
}
} else if it
.calibration
.as_ref()
.is_some_and(|c| !c.options.is_empty())
{
push(
Rule::UnusedOption,
Severity::Low,
format!(
"option `{}` has never been administered. Either it is waiting \
its turn, or it was drafted and forgotten — retire it and say \
which.",
option.id
),
);
}
}
}
}
let _ = keys;
}
// --- evidence
if !it.calibration_is_current() {
push(
@@ -1196,7 +1317,7 @@ mod tests {
fn entry(yaml: &str) -> Entry {
let item: Item = serde_yaml_ng::from_str(yaml).expect("item parses");
Entry {
uid: format!("b::{}", item.id),
uid: item.id.clone(),
bank: "b".into(),
path: PathBuf::from("b.yaml"),
index: 0,
@@ -1215,6 +1336,45 @@ mod tests {
c
}
#[test]
fn a_pool_too_thin_to_fill_a_form_is_flagged() {
let c = codes(
r#"
id: q-a-001
status: draft
level: 2
stem: Which mechanism best explains the sigmoidal binding curve?
options:
- { id: o-shift, text: Ligand binding shifts the tetramer to a higher-affinity state, correct: true }
- { id: o-fixed, text: Each subunit binds with the same fixed affinity throughout }
"#,
);
// Policy shows four options and the pool can supply two, so `assemble`
// would put this on a paper two short.
assert!(c.contains(&"pool-thin"), "{c:?}");
}
#[test]
fn a_retirement_with_no_finding_is_flagged() {
let c = codes(
r#"
id: q-a-001
status: draft
level: 2
stem: Which mechanism best explains the sigmoidal binding curve?
options:
- { id: o-shift, text: Ligand binding shifts the tetramer to a higher-affinity state, correct: true }
- { id: o-fixed, text: Each subunit binds with the same fixed affinity throughout }
- { id: o-consumed, text: "Ligand is consumed as it binds, depleting the available pool" }
- { id: o-oxidation, text: The heme iron changes oxidation state upon binding }
- { id: o-cooperative, text: Subunits bind independently of one another, retired: { 'on': 2026-09-20, reason: bad } }
"#,
);
assert!(c.contains(&"pool-unjustified-retirement"), "{c:?}");
// Four live distractors is enough for a four-option form.
assert!(!c.contains(&"pool-thin"), "{c:?}");
}
#[test]
fn clean_item_passes() {
let c = codes(
+154 -3
View File
@@ -33,8 +33,9 @@ use crate::course::{CourseFile, SCHEMA_VERSION};
use crate::date::Date;
use crate::error::{Error, Result};
use crate::history::History;
use crate::item::{Choice, Item};
use crate::rng::Rng;
use crate::taxonomy::Level;
use crate::taxonomy::{Format, Level};
/// The result of a draw.
#[derive(Debug, Clone)]
@@ -466,14 +467,23 @@ pub fn to_record(
.chain(selection.bonus.iter().map(|u| (u, true))),
) {
let e = catalog.require(uid)?;
let (key, distractors) = draw_options(
&e.item,
catalog.course.policy.options_per_item,
blueprint.seed.unwrap_or(0),
uid,
);
items.push(Placement {
number,
item: uid.clone(),
version: Some(e.item.version),
version: None,
stem_digest: Some(e.item.stem_digest()),
variant: Some(e.item.variant_digest(&key, &distractors)),
fingerprint: Some(e.item.fingerprint()),
points: Some(e.item.points(default_points)),
bonus: is_bonus || e.item.bonus,
key: e.item.key_letters(),
distractors,
key,
level: Some(e.item.level),
learning_targets: e.item.learning_targets.clone(),
credit_overrides: BTreeMap::new(),
@@ -570,6 +580,85 @@ pub fn layout(record: &AssessmentFile, form: &Form) -> Vec<Placement> {
scored.into_iter().chain(bonus).collect()
}
/// Draws the key and the distractors one placement administers.
///
/// Resolved here, at assembly, and written into the record as explicit lists.
/// Nothing downstream samples: an export that drew its own options would print
/// a different paper every time the bank was touched.
///
/// The draw is seeded on the blueprint and the item, so re-running `assemble`
/// with the same seed produces the same paper, and two items in one assessment
/// draw independently.
///
/// # Arguments
///
/// * `item` - the item, whose options are a pool.
/// * `per_item` - how many options a form shows, from course policy.
/// * `seed` - the blueprint seed.
/// * `uid` - the item id, salting the draw.
///
/// # Returns
///
/// The keyed ids and the distractor ids, each sorted, naming options of `item`.
/// Both empty for an item with no options, which is an open response.
pub fn draw_options(
item: &Item,
per_item: usize,
seed: u64,
uid: &str,
) -> (Vec<String>, Vec<String>) {
let (keys, distractors) = item.pool();
if keys.is_empty() && distractors.is_empty() {
return (Vec::new(), Vec::new());
}
// Multiple response keys every correct option; anything else keys one, and
// when the pool offers several defensible keys the draw picks one so that
// the record says which.
let wanted_keys = match item.format {
Format::MultipleResponse => keys.len(),
_ => 1.min(keys.len()),
};
let mut rng = Rng::from_label(&format!("{seed}/{uid}/options"));
let mut key_ids = pick(&keys, wanted_keys, &mut rng);
key_ids.sort();
// A pool with fewer usable distractors than the policy asks for is a
// finding, not a failure: the form comes out short and `lint` says so,
// rather than `assemble` refusing to build the assessment at all.
let wanted = per_item.saturating_sub(key_ids.len());
let mut distractor_ids = pick(&distractors, wanted.min(distractors.len()), &mut rng);
distractor_ids.sort();
(key_ids, distractor_ids)
}
/// Takes `n` options, preferring the ones that were designed rather than merely
/// written.
///
/// A distractor carrying a misconception and an error type is one you thought
/// about; one carrying neither is filler. When the pool is larger than the form,
/// the thought-about ones go on the paper. The shuffle comes first so that
/// options of equal standing are drawn by seed rather than by declaration
/// order.
fn pick(options: &[&Choice], n: usize, rng: &mut Rng) -> Vec<String> {
if n >= options.len() {
return options.iter().map(|o| o.id.clone()).collect();
}
let mut order: Vec<usize> = (0..options.len()).collect();
rng.shuffle(&mut order);
order.sort_by_key(|&i| {
let o = options[i];
u8::from(o.misconception.is_none()) + u8::from(o.error_type.is_none())
});
order
.into_iter()
.take(n)
.map(|i| options[i].id.clone())
.collect()
}
/// The option order for one item on one form.
///
/// # Arguments
@@ -692,6 +781,68 @@ mod tests {
assert_eq!(form_label(27), "AB");
}
/// An item whose options are given as YAML, so the test needs no literal.
fn pool_item(options: &str) -> Item {
let src = format!(
r#"id: q-x
status: approved
level: 1
cognitive_process: recall
stem: Which line holds the quality scores?
learning_targets: [t-x]
sources: [{{ lecture: L1 }}]
options:
{options}"#
);
serde_yaml_ng::from_str(&src).expect("item parses")
}
const DESIGNED: &str = r#" - { id: o-key, text: right, correct: true }
- { id: o-designed-a, text: a, misconception: mistakes the separator, error_type: recall_confusion }
- { id: o-designed-b, text: b, misconception: confuses the two, error_type: recall_confusion }
- { id: o-filler-a, text: c }
- { id: o-filler-b, text: d }
"#;
#[test]
fn a_draw_prefers_designed_distractors_and_is_reproducible() {
let item = pool_item(DESIGNED);
let (key, distractors) = draw_options(&item, 3, 1103, "q-x");
assert_eq!(key, vec!["o-key".to_string()]);
assert_eq!(distractors.len(), 2);
// Thought-about distractors go on the paper before filler does.
assert!(
distractors.iter().all(|d| d.starts_with("o-designed")),
"{distractors:?}"
);
// Same seed, same paper.
assert_eq!(draw_options(&item, 3, 1103, "q-x"), (key, distractors));
// A retired option is not drawn, and the form comes out of the rest.
let retired = pool_item(&DESIGNED.replace(
"{ id: o-designed-a, text: a,",
"{ id: o-designed-a, text: a, retired: { 'on': 2026-09-20, reason: nonfunctioning },",
));
let (_, after) = draw_options(&retired, 3, 1103, "q-x");
assert!(!after.iter().any(|d| d == "o-designed-a"), "{after:?}");
}
#[test]
fn a_thin_pool_comes_out_short_rather_than_refusing_to_build() {
let item = pool_item(
" - { id: o-key, text: right, correct: true }\n - { id: o-one, text: wrong }\n",
);
let (key, distractors) = draw_options(&item, 4, 7, "q-y");
assert_eq!(key.len(), 1);
assert_eq!(
distractors.len(),
1,
"one usable distractor, so one is drawn"
);
}
#[test]
fn option_order_is_a_reproducible_permutation() {
let form = Form {
+131
View File
@@ -24,6 +24,7 @@ use coursebank::assessment::{Kind as AssessmentKind, Platform};
use coursebank::catalog::Severity;
use coursebank::item::IrtModel;
use coursebank::lecture::Style as PageStyle;
use coursebank::references;
use coursebank::store;
/// Manage course item banks, assessments, and the analysis that comes back.
@@ -48,6 +49,15 @@ pub(crate) struct Cli {
pub(crate) enum Command {
/// Create a new course directory.
Init(InitArgs),
/// Inspect the course file.
#[command(subcommand)]
Course(CourseCommand),
/// One-time conversions from an older layout.
#[command(subcommand)]
Migrate(MigrateCommand),
/// Work with the bibliography.
#[command(subcommand)]
References(ReferencesCommand),
/// Write JSON Schemas so your editor can validate the YAML as you type.
Schema,
/// Check every file for problems that must be fixed.
@@ -93,6 +103,127 @@ pub(crate) enum Command {
Data,
}
/// `course`: the course file itself, which may be one file or many.
#[derive(Debug, Subcommand)]
pub(crate) enum CourseCommand {
/// List the files the course is assembled from.
Files,
/// Print the merged course, or write it to a file.
///
/// Nothing reads what this writes. It exists so you can see what the
/// fragments add up to, and diff two revisions of a course that no longer
/// lives in one file.
Build {
/// Output path; prints to stdout when omitted.
#[arg(long)]
out: Option<PathBuf>,
},
/// Say which file defines an id.
Where {
/// A unit, lecture, objective, target, reference, or stimulus id.
id: String,
},
}
/// `migrate`: the one-time conversions, grouped so they are findable together.
#[derive(Debug, Subcommand)]
pub(crate) enum MigrateCommand {
/// Split one course.yaml into references.yaml, lectures/, and objectives/.
///
/// The original is kept as course.yaml.bak, and the result is reassembled
/// and compared against it before the command reports success.
Split {
/// Show what would be written, and write nothing.
#[arg(long)]
dry_run: bool,
},
/// Fill in the stored `variant` column from the assessment records.
///
/// Nothing in the tool needs it — a variant is derived from the placement
/// when a row has none. It is for pandas, DuckDB, and R, which see only
/// what is in the column and will otherwise average two option sets of one
/// stem into an item that never existed.
Variants {
/// Show what would change, and write nothing.
#[arg(long)]
dry_run: bool,
},
/// Drop `version:` and `history:`, which 2.0 ignores.
///
/// A stem's text is its identity: reword it and it is a new item with a new
/// id and `supersedes:` pointing back. `validate` enforces that against
/// every seal, so what a version number used to hint at is now checked.
Stems {
/// Show what would change, and write nothing.
#[arg(long)]
dry_run: bool,
},
/// Rewrite option letters as names derived from the option text.
///
/// A letter is a position, and a position in a field that pooled
/// statistics and `credit_overrides` join on is a bug waiting for someone
/// to reorder a YAML block. Run with --dry-run first: the names land in
/// the response store, so they are as permanent as an item id.
Options {
/// Show the derived names, and write nothing.
#[arg(long)]
dry_run: bool,
},
/// Rewrite pre-2.0 `bank::item` ids as the item ids they name.
///
/// Touches assessment records, seals, and the response store. Everything
/// keeps working unmigrated — an old id still resolves — but a store
/// holding both forms groups one question into two for anything reading the
/// Parquet without this tool.
Ids {
/// Show what would change, and write nothing.
#[arg(long)]
dry_run: bool,
},
}
/// `references`: the bibliography, and the formats other tools read it in.
#[derive(Debug, Subcommand)]
pub(crate) enum ReferencesCommand {
/// List every work, with the link a reading list would use.
///
/// The column that matters is the last one: a work with no link is one a
/// student cannot reach from a report, which for a manuscript usually means
/// its DOI is missing.
List,
/// Write the bibliography in a citation format.
Export {
/// Which format to write.
#[arg(long, value_enum, default_value = "hayagriva")]
format: ReferenceFormat,
/// Output path; prints to stdout when omitted.
#[arg(long)]
out: Option<PathBuf>,
},
}
/// The citation formats `references export` can write.
#[derive(Debug, Clone, Copy, ValueEnum)]
pub(crate) enum ReferenceFormat {
/// Hayagriva YAML, which Typst reads natively.
Hayagriva,
/// CSL-JSON, for Zotero, Pandoc, and CSL processors.
CslJson,
/// BibTeX.
Bibtex,
}
impl ReferenceFormat {
/// The library-side format.
pub(crate) fn as_format(self) -> references::Format {
match self {
ReferenceFormat::Hayagriva => references::Format::Hayagriva,
ReferenceFormat::CslJson => references::Format::CslJson,
ReferenceFormat::Bibtex => references::Format::Bibtex,
}
}
}
#[derive(Debug, Args)]
pub(crate) struct InitArgs {
/// Course code, e.g. "BIOSC 1540".
+5 -2
View File
@@ -8,8 +8,8 @@
//! calls the matching handler. The handlers themselves live in submodules that
//! follow the workflow described in the crate documentation:
//!
//! - [`project`] — set up and check a course: `init`, `schema`, `validate`,
//! `lint`, `catalog`.
//! - [`project`] — set up and check a course: `init`, `course`, `schema`,
//! `validate`, `lint`, `catalog`.
//! - [`lectures`] — render a lecture's reading list and check what backs each
//! objective: `lecture`.
//! - [`banks`] — manage items and build assessments: `bank`, `assessment`,
@@ -55,6 +55,9 @@ pub(crate) enum Outcome {
pub(crate) fn run(cli: &Cli) -> Result<Outcome> {
match &cli.command {
Command::Init(args) => project::init(cli, args),
Command::Course(sub) => project::course(cli, sub),
Command::References(sub) => project::references(cli, sub),
Command::Migrate(sub) => project::migrate(cli, sub),
Command::Schema => project::schema(cli),
Command::Validate => project::validate(cli),
Command::Lint(args) => project::lint(cli, args),
+337 -7
View File
@@ -5,9 +5,10 @@
//! Setting up a course and checking it stays well-formed.
//!
//! These are the commands you reach for before and around authoring: create the
//! directory (`init`), write editor schemas (`schema`), and run the two kinds of
//! checking — [`validate`] for problems that must be fixed and [`lint`] for
//! item-writing guidance. [`catalog`] summarizes the pool that results.
//! directory (`init`), see and split the course file (`course`), export the
//! bibliography (`references`), write editor schemas (`schema`), and run the two
//! kinds of checking — [`validate`] for problems that must be fixed and [`lint`]
//! for item-writing guidance. [`catalog`] summarizes the pool that results.
use std::collections::{BTreeMap, BTreeSet};
use std::fs;
@@ -15,15 +16,20 @@ use std::path::Path;
use coursebank::assessment::AssessmentFile;
use coursebank::bank::BankFile;
use coursebank::course::fragment::{self, Section};
use coursebank::course::{COURSE_FILE, CourseFile};
use coursebank::error::{Error, Result};
use coursebank::jsonschema;
use coursebank::layout::Layout;
use coursebank::lint::{self, Rule};
use coursebank::migrate;
use coursebank::references;
use coursebank::taxonomy::{Level, Tier};
use coursebank::yaml;
use crate::cli::{CatalogArgs, Cli, InitArgs, LintArgs};
use crate::cli::{
CatalogArgs, Cli, CourseCommand, InitArgs, LintArgs, MigrateCommand, ReferencesCommand,
};
use crate::commands::Outcome;
use crate::helpers::{load, truncate};
@@ -77,13 +83,334 @@ pub(crate) fn init(cli: &Cli, args: &InitArgs) -> Result<Outcome> {
write_gitignore(&cli.course.join(".gitignore"))?;
println!(
"\nNext: edit {} to add your learning objectives, their targets, and your\n lectures, then\n \
coursebank bank new unit-1 --title \"Unit 1\"\n coursebank validate",
COURSE_FILE
"\nNext: edit {COURSE_FILE} to add your learning objectives, their targets, and\n \
your lectures, then\n coursebank bank new unit-1 --title \"Unit 1\"\n \
coursebank validate\n\nOnce {COURSE_FILE} is more than you want to scroll, \
`coursebank migrate split`\n moves each lecture and objective into its own file under \
lectures/ and\n objectives/, and every command goes on reading the course as one."
);
Ok(Outcome::Ok)
}
/// `course`: inspect the course file, or split it into fragments.
pub(crate) fn course(cli: &Cli, sub: &CourseCommand) -> Result<Outcome> {
match sub {
CourseCommand::Files => course_files(cli),
CourseCommand::Build { out } => course_build(cli, out.as_deref()),
CourseCommand::Where { id } => course_where(cli, id),
}
}
/// `migrate`: the one-time layout conversions.
pub(crate) fn migrate(cli: &Cli, sub: &MigrateCommand) -> Result<Outcome> {
match sub {
MigrateCommand::Split { dry_run } => course_split(cli, *dry_run),
MigrateCommand::Ids { dry_run } => migrate_ids(cli, *dry_run),
MigrateCommand::Options { dry_run } => migrate_options(cli, *dry_run),
MigrateCommand::Stems { dry_run } => migrate_stems(cli, *dry_run),
MigrateCommand::Variants { dry_run } => migrate_variants(cli, *dry_run),
}
}
/// Fills in the stored variant column.
fn migrate_variants(cli: &Cli, dry_run: bool) -> Result<Outcome> {
let touched = migrate::store_variants(&cli.course, !dry_run)?;
if touched.is_empty() {
println!("nothing to fill in: every stored row already names its variant");
return Ok(Outcome::Ok);
}
for (path, n) in &touched {
println!(" {:<44} {n:>5} row(s)", path.display());
}
if dry_run {
println!("\nnothing written");
return Ok(Outcome::Ok);
}
println!("\nrewrote {} data file(s)", touched.len());
Ok(Outcome::Ok)
}
/// Drops the version fields 2.0 ignores.
fn migrate_stems(cli: &Cli, dry_run: bool) -> Result<Outcome> {
let touched = migrate::stems(&cli.course, !dry_run)?;
if touched.is_empty() {
println!("nothing to migrate: no `version:` or `history:` left to drop");
return Ok(Outcome::Ok);
}
for (path, n) in &touched {
println!(" {:<44} {n:>5} line(s) dropped", path.display());
}
if dry_run {
println!("\nnothing written");
return Ok(Outcome::Ok);
}
println!(
"\nrewrote {} file(s)\n\nNext:\n coursebank validate\n\nFrom here, rewording a stem \
is an error rather than a version bump: give the new\n wording a new id and \
`supersedes:` the old one.",
touched.len()
);
Ok(Outcome::Ok)
}
/// Rewrites option letters as names, showing every name before writing.
fn migrate_options(cli: &Cli, dry_run: bool) -> Result<Outcome> {
let (map, problems) = migrate::options_plan(&cli.course)?;
for (item, options) in &map {
println!("{item}");
for (letter, name) in options {
println!(" {letter} -> {name}");
}
}
if !problems.is_empty() {
println!("\n{} item(s) need naming by hand:", problems.len());
for problem in &problems {
println!(" - {problem}");
}
}
if map.is_empty() {
println!("nothing to migrate: every option is already named");
return Ok(Outcome::Ok);
}
let touched = migrate::apply_options(&cli.course, &map, !dry_run)?;
println!();
for (path, n) in &touched {
println!(" {:<44} {n:>5} rename(s)", path.display());
}
if dry_run {
println!("\nnothing written");
return Ok(if problems.is_empty() {
Outcome::Ok
} else {
Outcome::Findings
});
}
println!(
"\nrewrote {} file(s)\n\nSeals keep their letters on purpose; see `coursebank migrate \
--help`.\nNext:\n coursebank validate\n coursebank lint",
touched.len()
);
Ok(if problems.is_empty() {
Outcome::Ok
} else {
Outcome::Findings
})
}
/// Rewrites pre-2.0 bank-qualified item ids everywhere they are stored.
fn migrate_ids(cli: &Cli, dry_run: bool) -> Result<Outcome> {
let files = migrate::qualified_ids(&cli.course)?;
for (path, n, _) in &files {
println!(" {:<44} {n:>5} id(s)", path.display());
}
if !dry_run {
migrate::apply_ids(&cli.course, &files)?;
}
let data = migrate::store_ids(&cli.course, !dry_run)?;
for (path, n) in &data {
println!(" {:<44} {n:>5} row(s)", path.display());
}
if files.is_empty() && data.is_empty() {
println!("nothing to migrate: every item id already names the item course-wide");
return Ok(Outcome::Ok);
}
if dry_run {
println!("\nnothing written");
return Ok(Outcome::Ok);
}
println!(
"\nrewrote {} file(s) and {} data file(s)\n\nNext:\n coursebank validate\n \
coursebank analyze items --all",
files.len(),
data.len()
);
Ok(Outcome::Ok)
}
/// Lists the fragments a course is assembled from, with what each defines.
fn course_files(cli: &Cli) -> Result<Outcome> {
let layout = Layout::new(&cli.course);
let course = CourseFile::load_dir(&cli.course)?;
for (path, role) in fragment::files(&layout)? {
if !path.exists() {
continue;
}
let shown = path.strip_prefix(&cli.course).unwrap_or(&path);
let mut defines: Vec<String> = Vec::new();
for section in Section::ALL {
let n = course
.origins
.iter()
.filter(|((s, _), p)| *s == section && p.as_path() == shown)
.count();
if n == 0 {
continue;
}
defines.push(match section {
// These are declared once for the whole course, so a count
// would always be 1 and would read as though it could be more.
Section::Course | Section::Policy => section.key().to_string(),
_ => format!("{n} {}", section.key()),
});
}
println!(
"{:<40} {:<11} {}",
shown.display(),
role.label(),
defines.join(", ")
);
}
Ok(Outcome::Ok)
}
/// Prints or writes the merged course.
fn course_build(cli: &Cli, out: Option<&Path>) -> Result<Outcome> {
let course = CourseFile::load_dir(&cli.course)?;
match out {
Some(path) => {
course.write_resolved(path)?;
println!(
"wrote {} from {} file(s)",
path.display(),
course.fragment_paths().len()
);
}
None => print!("{}", yaml::to_string(&course)?),
}
Ok(Outcome::Ok)
}
/// Says which file defines an id.
fn course_where(cli: &Cli, id: &str) -> Result<Outcome> {
let course = CourseFile::load_dir(&cli.course)?;
match course.origin(id) {
Some((section, path)) => {
println!(
"{} defines `{id}` under `{}`",
path.display(),
section.key()
);
Ok(Outcome::Ok)
}
None => Err(Error::Unresolved {
kind: "id",
id: id.to_string(),
context: Some(cli.course.display().to_string()),
}),
}
}
/// Splits `course.yaml` into fragments, then checks the result reassembles.
fn course_split(cli: &Cli, dry_run: bool) -> Result<Outcome> {
let plan = migrate::split_plan(&cli.course)?;
println!("{} file(s):", plan.files.len());
for (path, lines) in plan.lines() {
let summary = plan
.files
.iter()
.find(|f| f.path == path)
.map(|f| f.summary.clone())
.unwrap_or_default();
println!(" {:<44} {lines:>5} lines {summary}", path.display());
}
for note in &plan.notes {
println!("\nnote: {note}");
}
if dry_run {
println!("\nnothing written");
return Ok(Outcome::Ok);
}
// Loaded before anything is written, since it is the thing the result is
// checked against.
let before = CourseFile::load(&Layout::new(&cli.course).course_file())?;
let written = migrate::apply(&cli.course, &plan)?;
println!("\nwrote {} file(s)", written.len());
let after = CourseFile::load_dir(&cli.course)?;
let diffs = migrate::differences(&before, &after)?;
if diffs.is_empty() {
println!(
"reassembled and compared against {COURSE_FILE}.bak: identical\n\nNext:\n \
coursebank validate\n coursebank schema\n git add -A && git diff --cached --stat"
);
return Ok(Outcome::Ok);
}
println!(
"\n{} difference(s) between the original and the reassembled course:",
diffs.len()
);
for diff in &diffs {
println!(" - {diff}");
}
println!(
"\nThe original is at {COURSE_FILE}.bak. Restore it with\n mv {COURSE_FILE}.bak \
{COURSE_FILE} && rm -r lectures objectives {}",
fragment::REFERENCES_FILE
);
Ok(Outcome::Findings)
}
/// `references`: list the bibliography, or export it in a citation format.
pub(crate) fn references(cli: &Cli, sub: &ReferencesCommand) -> Result<Outcome> {
let course = CourseFile::load_dir(&cli.course)?;
match sub {
ReferencesCommand::List => {
let mut unreachable = 0;
for (key, reference) in &course.references {
let link = match reference.href(None, None) {
Some(url) => url,
None => {
unreachable += 1;
"(no link)".to_string()
}
};
println!(
"{:<28} {:<10} {:<9} {:<6} {link}",
truncate(key, 27),
format!("{:?}", reference.kind).to_lowercase(),
format!("{:?}", reference.role).to_lowercase(),
reference.label_or("-"),
);
}
if unreachable > 0 && !cli.quiet {
println!(
"\n{unreachable} work(s) with no link. A student report can name one but \
cannot send anyone to it; for a manuscript, add its `doi`."
);
}
Ok(Outcome::Ok)
}
ReferencesCommand::Export { format, out } => {
let format = format.as_format();
let text = references::render(&course, format)?;
match out {
Some(path) => {
yaml::write_text(path, &text)?;
println!(
"wrote {} ({} work(s) as {})",
path.display(),
course.references.len(),
format.label()
);
}
None => print!("{text}"),
}
Ok(Outcome::Ok)
}
}
}
/// What reconciling [`GITIGNORE`] against a file already on disk would do.
struct GitignoreMerge {
/// The file to write. Identical to the input when nothing was missing.
@@ -260,6 +587,9 @@ pub(crate) fn validate(cli: &Cli) -> Result<Outcome> {
}
}
let seals = coursebank::seal::SealFile::load_all(&catalog.layout.seals())?;
all.extend(catalog.validate_seals(&seals));
if all.is_empty() {
if !cli.quiet {
println!(
+1
View File
@@ -337,6 +337,7 @@ pub fn ingest(
item_number: number,
item_ref,
item_version: None,
variant: None,
selected,
selected_source: Vec::new(),
eliminated: Vec::new(),
+4 -3
View File
@@ -248,7 +248,8 @@ impl FormDecoder {
for (index, placement) in printed.iter().enumerate() {
let entry = catalog.require(&placement.item)?;
let item = &entry.item;
let order = select::option_order(form, &placement.item, item.options.len());
let shown = item.administered(&placement.key, &placement.distractors);
let order = select::option_order(form, &placement.item, shown.len());
let canonical_key: BTreeSet<String> = if placement.key.is_empty() {
item.key_letters().into_iter().collect()
@@ -260,8 +261,7 @@ impl FormDecoder {
let mut to_printed = BTreeMap::new();
let mut printed_key = Vec::new();
for (position, source_index) in order.iter().enumerate() {
let canonical = item
.options
let canonical = shown
.get(*source_index)
.map(|c| c.id.clone())
.unwrap_or_else(|| printed_letter(*source_index));
@@ -738,6 +738,7 @@ mod tests {
form_position: None,
item_ref: None,
item_version: None,
variant: None,
selected: vec![selected.into()],
eliminated: Vec::new(),
selected_source: Vec::new(),
+1
View File
@@ -645,6 +645,7 @@ pub fn to_responses(questions: &[Question], ctx: &Context) -> Import {
item_number: q.number,
item_ref: None,
item_version: None,
variant: None,
selected,
selected_source: Vec::new(),
eliminated,
+93 -1
View File
@@ -75,6 +75,16 @@ pub struct Response {
/// The item version as administered.
pub item_version: Option<u32>,
/// The variant administered: which option set this row's student saw.
///
/// The grouping key for pooled statistics. Set at ingest from the
/// assessment record, and stored so that anything reading the Parquet
/// without this tool can group the same way — a store that only has
/// `item_ref` cannot tell two option sets of one stem apart, and will
/// average them.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub variant: Option<String>,
/// Option letters the student chose.
pub selected: Vec<String>,
/// The selected options in the bank's own lettering, written at ingest by
@@ -314,6 +324,49 @@ impl ResponseSet {
per_item.values().sum()
}
/// How many rows each option set of each item has.
///
/// What a calibration report needs to be honest about sample size: an item
/// administered three times with three different option sets has three
/// cells, not one, and reporting "n = 72" of it would be wrong three ways.
///
/// # Returns
///
/// Row counts keyed by item id and variant, with an empty variant for rows
/// that carry none.
pub fn variants(&self) -> BTreeMap<(String, String), usize> {
let mut out: BTreeMap<(String, String), usize> = BTreeMap::new();
for row in &self.rows {
let Some(item) = &row.item_ref else { continue };
let key = (item.clone(), row.variant.clone().unwrap_or_default());
*out.entry(key).or_insert(0) += 1;
}
out
}
/// The rows for one option set of one item.
///
/// # Arguments
///
/// * `item_ref` - the item id.
/// * `variant` - the variant digest, or `None` for rows carrying none.
///
/// # Returns
///
/// A set holding only those rows, keeping the warnings of the original.
pub fn for_variant(&self, item_ref: &str, variant: Option<&str>) -> ResponseSet {
ResponseSet {
rows: self
.rows
.iter()
.filter(|r| r.item_ref.as_deref() == Some(item_ref))
.filter(|r| r.variant.as_deref() == variant)
.cloned()
.collect(),
warnings: self.warnings.clone(),
}
}
/// Builds the response matrix for psychometrics.
///
/// # Arguments
@@ -416,6 +469,9 @@ impl ResponseSet {
};
r.item_ref = Some(p.item.clone());
r.item_version = p.version;
// Recorded when the record says so; derived from the option set
// otherwise, which is the case for every administration before 2.0.
r.variant = p.variant.clone();
r.bonus = r.bonus || p.bonus;
r.dropped = r.dropped || p.dropped;
r.dropped_full_credit = r.dropped_full_credit || p.dropped_with_credit();
@@ -434,6 +490,9 @@ impl ResponseSet {
if let Some(cat) = catalog {
if let Some(entry) = cat.get(&p.item) {
if r.variant.is_none() {
r.variant = Some(p.variant_of(&entry.item));
}
r.level = Some(entry.item.level);
r.learning_targets = if p.learning_targets.is_empty() {
entry.item.learning_targets.clone()
@@ -636,6 +695,8 @@ pub struct FlatResponse {
pub item_ref: String,
/// The item version, 0 when unknown.
pub item_version: u32,
/// The administered variant, empty when unknown.
pub variant: String,
/// Comma-joined selected letters.
pub selected: String,
/// Comma-joined eliminated letters.
@@ -710,6 +771,7 @@ impl FlatResponse {
item_number: r.item_number,
item_ref: r.item_ref.clone().unwrap_or_default(),
item_version: r.item_version.unwrap_or(0),
variant: r.variant.clone().unwrap_or_default(),
selected: r.selected.join(","),
selected_source: r.selected_source.join(","),
eliminated: r.eliminated.join(","),
@@ -767,7 +829,11 @@ impl FlatResponse {
email: none_if_empty(&self.email),
section: none_if_empty(&self.section),
item_number: self.item_number,
item_ref: none_if_empty(&self.item_ref),
// Canonicalized on read, so a term ingested before 2.0 pools with
// one ingested after it instead of splitting into two items.
item_ref: none_if_empty(&self.item_ref)
.map(|id| crate::item::canonical_id(&id).to_string()),
variant: none_if_empty(&self.variant),
item_version: if self.item_version == 0 {
None
} else {
@@ -825,6 +891,7 @@ mod tests {
item_number: number,
item_ref: None,
item_version: None,
variant: None,
selected: vec!["A".into()],
selected_source: vec![],
eliminated: vec![],
@@ -843,6 +910,31 @@ mod tests {
}
}
#[test]
fn rows_group_by_the_option_set_they_administered() {
let mut set = ResponseSet::new();
for (student, variant) in [("s1", "v1"), ("s2", "v1"), ("s3", "v2")] {
let mut r = row(student, 1, 1.0);
r.item_ref = Some("q-x".into());
r.variant = Some(variant.into());
set.rows.push(r);
}
// A row from before the column existed.
let mut old = row("s4", 1, 1.0);
old.item_ref = Some("q-x".into());
set.rows.push(old);
let counts = set.variants();
assert_eq!(counts[&("q-x".to_string(), "v1".to_string())], 2);
assert_eq!(counts[&("q-x".to_string(), "v2".to_string())], 1);
// Unknown groups on its own rather than joining either set.
assert_eq!(counts[&("q-x".to_string(), String::new())], 1);
assert_eq!(set.for_variant("q-x", Some("v1")).rows.len(), 2);
assert_eq!(set.for_variant("q-x", None).rows.len(), 1);
assert_eq!(set.for_variant("q-other", Some("v1")).rows.len(), 0);
}
#[test]
fn matrix_is_students_by_items() {
let mut set = ResponseSet::new();
+68 -1
View File
@@ -316,6 +316,70 @@ pub fn read_path(path: &Path) -> Result<ResponseSet> {
}
}
/// Reads a response file without turning its rows into [`Response`]s.
///
/// What a migration wants: the rows exactly as they sit on disk, so rewriting
/// one column cannot disturb another through a round trip.
///
/// # Arguments
///
/// * `path` - the file to read.
///
/// # Returns
///
/// The rows.
///
/// # Errors
///
/// Returns [`Error::Other`] for an unrecognized extension, [`Error::Csv`] or
/// [`Error::Other`] on a parse failure, and [`Error::FeatureDisabled`] for
/// Parquet without the feature.
pub fn read_flat(path: &Path) -> Result<Vec<FlatResponse>> {
match Format::from_path(path) {
Some(Format::Csv) => {
let mut r = csv::Reader::from_path(path).map_err(|e| Error::Csv {
path: path.to_path_buf(),
source: e,
})?;
let mut out = Vec::new();
for rec in r.deserialize::<FlatResponse>() {
out.push(rec.map_err(|e| Error::Csv {
path: path.to_path_buf(),
source: e,
})?);
}
Ok(out)
}
Some(Format::Parquet) => crate::store_parquet::read(path),
None => Err(Error::Other(format!(
"{} is not a response file; expected a .parquet or .csv",
path.display()
))),
}
}
/// Writes flat responses back to the file they came from.
///
/// # Arguments
///
/// * `path` - the destination, whose extension picks the format.
/// * `rows` - the rows.
///
/// # Errors
///
/// Returns [`Error::Other`] for an unrecognized extension and
/// [`Error::FeatureDisabled`] for Parquet without the feature.
pub fn write_flat(path: &Path, rows: &[FlatResponse]) -> Result<()> {
match Format::from_path(path) {
Some(Format::Csv) => write_csv(path, rows),
Some(Format::Parquet) => write_parquet(path, rows),
None => Err(Error::Other(format!(
"{} is not a response file; expected a .parquet or .csv",
path.display()
))),
}
}
/// Writes flat responses as CSV.
///
/// # Arguments
@@ -550,6 +614,7 @@ mod tests {
item_number: number,
item_ref: Some("bank::q-x-001".into()),
item_version: Some(2),
variant: None,
selected: vec!["C".into()],
selected_source: vec![],
eliminated: vec![],
@@ -591,7 +656,9 @@ mod tests {
let back = store.read("BIOSC1540/2026s/exam-4").unwrap();
assert_eq!(back.rows.len(), 2);
assert_eq!(back.rows[0].item_ref.as_deref(), Some("bank::q-x-001"));
// Canonicalized on the way in: the row was written with a pre-2.0
// `bank::item` key, and reading it yields the item it names.
assert_eq!(back.rows[0].item_ref.as_deref(), Some("q-x-001"));
assert_eq!(back.rows[0].selected, vec!["C".to_string()]);
assert_eq!(back.rows[0].learning_targets, vec!["lo-a".to_string()]);
+7
View File
@@ -49,6 +49,7 @@ pub fn schema() -> Schema {
Field::new("item_number", DataType::UInt32, false),
Field::new("item_ref", DataType::Utf8, false),
Field::new("item_version", DataType::UInt32, false),
Field::new("variant", DataType::Utf8, false),
Field::new("selected", DataType::Utf8, false),
Field::new("eliminated", DataType::Utf8, false),
Field::new("correct", DataType::Utf8, false),
@@ -114,6 +115,7 @@ fn to_batch(rows: &[FlatResponse]) -> Result<RecordBatch> {
u32c(|r| r.item_number),
s(|r| &r.item_ref),
u32c(|r| r.item_version),
s(|r| &r.variant),
s(|r| &r.selected),
s(|r| &r.eliminated),
s(|r| &r.correct),
@@ -279,6 +281,9 @@ fn from_batch(batch: &RecordBatch, path: &Path) -> Result<Vec<FlatResponse>> {
let item_number = uints("item_number")?;
let item_ref = strings("item_ref")?;
let item_version = uints("item_version")?;
// Added after the first stores were written, so absent rather than fatal in
// a file from before 2.0; `coursebank migrate variants` fills it in.
let variant = optional_strings("variant");
let selected = strings("selected")?;
let eliminated = strings("eliminated")?;
let correct = strings("correct")?;
@@ -314,6 +319,7 @@ fn from_batch(batch: &RecordBatch, path: &Path) -> Result<Vec<FlatResponse>> {
item_number: item_number.value(i),
item_ref: item_ref.value(i).to_string(),
item_version: item_version.value(i),
variant: variant.map(|c| c.value(i).to_string()).unwrap_or_default(),
selected: selected.value(i).to_string(),
selected_source: selected_source
.map(|a| a.value(i).to_string())
@@ -366,6 +372,7 @@ mod tests {
item_number: number,
item_ref: "bank::q-a-001".into(),
item_version: 3,
variant: "4c81fa".into(),
selected: "C".into(),
eliminated: String::new(),
correct: "1".into(),
+2
View File
@@ -12,6 +12,7 @@
//! | [`site`] | a Quarto partial and an encrypted bundle | a course page with password-gated solutions |
//! | [`report`] | Markdown and HTML | students, and yourself |
//! | [`lecture`] | Markdown | the reading list on the course website |
//! | [`references`] | Hayagriva, CSL-JSON, BibTeX | Typst, Zotero, LaTeX |
//!
//! [`qti`] and [`typst`] share one rule that is easy to get wrong: a form's answer
//! key must be generated from the same permutation that produced its question
@@ -26,6 +27,7 @@
pub mod lecture;
pub mod practice;
pub mod qti;
pub mod references;
pub mod report;
pub mod site;
pub mod typst;
+2 -1
View File
@@ -374,7 +374,7 @@ fn entry(
///
/// A linked citation when the location has a URL, and a plain one when it does not.
fn heading(reading: &Reading, key: &str, reference: &Reference, style: Style) -> String {
let label = reference.label.as_deref().unwrap_or(key);
let label = reference.label_or(key);
let locator = reading.locator.as_deref().unwrap_or("");
let linked = match reading.resolve_url(reference) {
Some(url) if !locator.is_empty() => format!("[{locator}]({url})"),
@@ -428,6 +428,7 @@ mod tests {
date: None,
unit: None,
slides_url: None,
teaches: Vec::new(),
readings: vec![
Reading {
reference: Some("kuriyan2013molecules".into()),
+15 -23
View File
@@ -30,7 +30,7 @@
use crate::assessment::{AssessmentFile, Form, Placement};
use crate::catalog::Catalog;
use crate::course::{CourseFile, Reference};
use crate::course::CourseFile;
use crate::error::Result;
use crate::item::{Choice, Citation, Item};
use crate::markup;
@@ -289,7 +289,7 @@ fn worksheet_question(
out.push_str("\n\n");
if item.has_options() {
let ordered = ordered_options(item, form, &placement.item);
let ordered = ordered_options(item, placement, form);
for (position, source) in ordered.iter().enumerate() {
out.push_str(&format!(
"{}. {}\n",
@@ -320,7 +320,7 @@ fn solution_question(
out.push_str("\n\n");
if item.has_options() {
let ordered = ordered_options(item, form, &placement.item);
let ordered = ordered_options(item, placement, form);
for (position, source) in ordered.iter().enumerate() {
let mark = if source.correct { " ✓" } else { "" };
let note = source
@@ -427,9 +427,9 @@ fn cite(course: &CourseFile, citation: &Citation) -> String {
let Some(reference) = course.references.get(key) else {
return citation.display();
};
let label = reference.label.as_deref().unwrap_or(key);
let label = reference.label_or(key);
let locator = citation.locator.as_deref().unwrap_or("");
match resolve_url(citation, reference) {
match citation.href(reference) {
Some(url) if !locator.is_empty() => format!("`{label}` [{locator}]({url})"),
Some(url) => format!("`{label}` [{}]({url})", reference.title),
None if !locator.is_empty() => format!("`{label}` {locator}"),
@@ -437,21 +437,6 @@ fn cite(course: &CourseFile, citation: &Citation) -> String {
}
}
/// The URL for a citation: its own `url`, else the reference `base_url` joined with
/// the citation `path`.
fn resolve_url(citation: &Citation, reference: &Reference) -> Option<String> {
if let Some(url) = &citation.url {
return Some(url.clone());
}
let path = citation.path.as_deref()?;
let base = reference.base_url.as_deref()?;
Some(match (base.ends_with('/'), path.starts_with('/')) {
(true, true) => format!("{base}{}", &path[1..]),
(false, false) => format!("{base}/{path}"),
_ => format!("{base}{path}"),
})
}
/// The `## Question N` heading, marking a bonus item.
fn heading(number: usize, placement: &Placement) -> String {
let bonus = if placement.bonus { " (bonus)" } else { "" };
@@ -473,10 +458,11 @@ fn meta_line(placement: &Placement, item: &Item) -> String {
/// Salted with the item's global id, the same value the Typst and QTI exports use,
/// so a worksheet built for form B lists options in the order that form's paper and
/// its Canvas quiz do.
fn ordered_options<'a>(item: &'a Item, form: &Form, uid: &str) -> Vec<&'a Choice> {
select::option_order(form, uid, item.options.len())
fn ordered_options<'a>(item: &'a Item, placement: &Placement, form: &Form) -> Vec<&'a Choice> {
let shown = item.administered(&placement.key, &placement.distractors);
select::option_order(form, &placement.item, shown.len())
.into_iter()
.map(|i| &item.options[i])
.map(|i| shown[i])
.collect()
}
@@ -593,6 +579,9 @@ items:
number: 1,
item: "l11::q-enthalpy-001".into(),
version: None,
stem_digest: None,
distractors: Vec::new(),
variant: None,
fingerprint: None,
points: Some(1.0),
bonus: false,
@@ -608,6 +597,9 @@ items:
number: 2,
item: "l11::q-enthalpy-op-001".into(),
version: None,
stem_digest: None,
distractors: Vec::new(),
variant: None,
fingerprint: None,
points: Some(2.0),
bonus: false,
+19 -3
View File
@@ -476,10 +476,12 @@ pub fn build(catalog: &Catalog, record: &AssessmentFile, opts: &QtiOptions) -> R
if !placement.bonus {
total_points += points;
}
let shown = item.administered(&placement.key, &placement.distractors);
items.push(build_item(
&record.assessment.id,
&placement.item,
item,
&shown,
points,
opts,
));
@@ -564,14 +566,21 @@ pub fn build(catalog: &Catalog, record: &AssessmentFile, opts: &QtiOptions) -> R
/// # Returns
///
/// The element.
fn build_item(assessment_id: &str, uid: &str, item: &Item, points: f64, opts: &QtiOptions) -> Node {
fn build_item(
assessment_id: &str,
uid: &str,
item: &Item,
shown: &[&crate::item::Choice],
points: f64,
opts: &QtiOptions,
) -> Node {
// An open-response item is an essay in Canvas: no choices, graded by hand.
if !item.format.has_options() {
return build_essay_item(assessment_id, uid, item, points, opts);
}
let order = select::option_order(&opts.form, uid, item.options.len());
let ordered: Vec<&crate::item::Choice> = order.iter().map(|i| &item.options[*i]).collect();
let order = select::option_order(&opts.form, uid, shown.len());
let ordered: Vec<&crate::item::Choice> = order.iter().map(|i| shown[*i]).collect();
// Option identifiers are numeric, mirroring Canvas's own exports, and are
// derived from the item id so they survive regeneration.
@@ -1259,6 +1268,7 @@ mod tests {
defense: None,
feedback_student: None,
selection_rate_expected: None,
retired: None,
}
}
@@ -1389,6 +1399,9 @@ items:
number: 1,
item: "b::q-mcq".into(),
version: None,
stem_digest: None,
distractors: Vec::new(),
variant: None,
fingerprint: None,
points: Some(1.0),
bonus: false,
@@ -1404,6 +1417,9 @@ items:
number: 2,
item: "b::q-open".into(),
version: None,
stem_digest: None,
distractors: Vec::new(),
variant: None,
fingerprint: None,
points: Some(2.0),
bonus: false,
+498
View File
@@ -0,0 +1,498 @@
// SPDX-License-Identifier: Prosperity-3.0.0
// Copyright Scientific Computing Studio
// Source: https://git.scient.ing/education/coursebank
//! The bibliography, in the formats other tools read.
//!
//! `references.yaml` is the authoritative copy, and it is shaped for a reading
//! list: it carries a `label` that reports print, and a `role` saying whether
//! the course requires the work or offers it as background. No general citation
//! format has either field, which is why the course keeps its own.
//!
//! What the other formats are for is everything downstream of the reading list:
//!
//! | Format | Read by | Gets you |
//! |:--|:--|:--|
//! | [`Format::Hayagriva`] | Typst | real citations in a printed exam or report |
//! | [`Format::CslJson`] | Zotero, Pandoc, CSL processors | a bibliography in any style |
//! | [`Format::Bibtex`] | LaTeX, most reference managers | the lowest common denominator |
//!
//! All three are generated. The argument is the same one the lecture reading
//! list makes: two copies of a citation drift within a term, and one copy plus a
//! build step does not.
//!
//! # What does not survive the trip
//!
//! `label`, `role`, `base_url`, and `note` have nowhere to go in any of the
//! three, so they stay behind. That is the reason this is an export rather than
//! a migration: the course file is not recoverable from its own bibliography
//! export, and nothing reads these files back in.
use std::collections::BTreeMap;
use serde::Serialize;
use serde_json::{Value, json};
use crate::course::{CourseFile, Reference, ReferenceKind};
use crate::error::Result;
use crate::yaml;
/// Which citation format to write.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Format {
/// Hayagriva YAML, which Typst's `#bibliography` reads natively.
Hayagriva,
/// CSL-JSON, which Zotero, Pandoc, and every CSL processor read.
CslJson,
/// BibTeX, for LaTeX and for reference managers that read nothing else.
Bibtex,
}
impl Format {
/// The conventional file extension.
pub fn extension(self) -> &'static str {
match self {
Format::Hayagriva => "yml",
Format::CslJson => "json",
Format::Bibtex => "bib",
}
}
/// The name used on the command line.
pub fn label(self) -> &'static str {
match self {
Format::Hayagriva => "hayagriva",
Format::CslJson => "csl-json",
Format::Bibtex => "bibtex",
}
}
}
/// Renders a course's bibliography.
///
/// # Arguments
///
/// * `course` - the loaded course.
/// * `format` - which format to write.
///
/// # Returns
///
/// The document, keyed or ordered by citation key so the output is stable
/// between runs.
///
/// # Errors
///
/// Returns [`crate::error::Error::Other`] if the intermediate structure cannot
/// be serialized.
pub fn render(course: &CourseFile, format: Format) -> Result<String> {
match format {
Format::Hayagriva => hayagriva(course),
Format::CslJson => csl_json(course),
Format::Bibtex => Ok(bibtex(course)),
}
}
// --- Hayagriva ---
/// One Hayagriva entry.
#[derive(Debug, Serialize)]
struct Entry {
#[serde(rename = "type")]
kind: &'static str,
title: String,
#[serde(skip_serializing_if = "Vec::is_empty")]
author: Vec<String>,
#[serde(skip_serializing_if = "Option::is_none")]
date: Option<u32>,
#[serde(skip_serializing_if = "Option::is_none")]
edition: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
publisher: Option<String>,
#[serde(rename = "page-range", skip_serializing_if = "Option::is_none")]
page_range: Option<String>,
#[serde(rename = "serial-number", skip_serializing_if = "BTreeMap::is_empty")]
serial_number: BTreeMap<&'static str, String>,
#[serde(skip_serializing_if = "Option::is_none")]
url: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
parent: Option<Parent>,
}
/// The container a Hayagriva entry sits inside.
#[derive(Debug, Serialize)]
struct Parent {
#[serde(rename = "type")]
kind: &'static str,
title: String,
#[serde(skip_serializing_if = "Option::is_none")]
volume: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
issue: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
publisher: Option<String>,
}
/// Renders the bibliography as Hayagriva YAML.
fn hayagriva(course: &CourseFile) -> Result<String> {
let entries: BTreeMap<&String, Entry> = course
.references
.iter()
.map(|(key, reference)| (key, entry(reference)))
.collect();
yaml::to_string(&entries)
}
/// Maps one reference onto a Hayagriva entry.
fn entry(reference: &Reference) -> Entry {
let mut serial: BTreeMap<&'static str, String> = BTreeMap::new();
for (field, value) in [
("doi", &reference.doi),
("isbn", &reference.isbn),
("arxiv", &reference.arxiv),
("pmid", &reference.pmid),
("pmcid", &reference.pmcid),
] {
if let Some(value) = value {
serial.insert(field, value.clone());
}
}
// A chapter sits in a book and an article sits in a periodical, and
// Hayagriva wants that said with a parent rather than a flat field.
let parent = reference
.container
.as_ref()
.map(|title| match reference.kind {
ReferenceKind::Chapter => Parent {
kind: "book",
title: title.clone(),
volume: reference.volume.clone(),
issue: None,
publisher: reference.publisher.clone(),
},
_ => Parent {
kind: "periodical",
title: title.clone(),
volume: reference.volume.clone(),
issue: reference.issue.clone(),
publisher: None,
},
});
Entry {
kind: hayagriva_kind(reference.kind),
title: reference.title.clone(),
author: reference.authors.clone(),
date: reference.year,
edition: reference.edition.clone(),
// The publisher belongs to the container when there is one.
publisher: if parent.is_some() {
None
} else {
reference.publisher.clone()
},
page_range: reference.pages.clone(),
serial_number: serial,
url: reference.url.clone(),
parent,
}
}
/// The Hayagriva entry type for a course reference kind.
fn hayagriva_kind(kind: ReferenceKind) -> &'static str {
match kind {
ReferenceKind::Book => "book",
ReferenceKind::Chapter => "chapter",
// Hayagriva has no preprint type. An article with no periodical parent
// is what a preprint is anyway.
ReferenceKind::Article | ReferenceKind::Preprint => "article",
ReferenceKind::Thesis => "thesis",
ReferenceKind::Website => "web",
ReferenceKind::Software | ReferenceKind::Dataset => "repository",
ReferenceKind::Video => "video",
ReferenceKind::Other => "misc",
}
}
// --- CSL-JSON ---
/// Renders the bibliography as CSL-JSON.
fn csl_json(course: &CourseFile) -> Result<String> {
let items: Vec<Value> = course
.references
.iter()
.map(|(key, reference)| csl_item(key, reference))
.collect();
serde_json::to_string_pretty(&items)
.map(|text| format!("{text}\n"))
.map_err(crate::error::Error::other)
}
/// One CSL-JSON item.
fn csl_item(key: &str, reference: &Reference) -> Value {
let mut item = json!({
"id": key,
"type": csl_kind(reference.kind),
"title": reference.title,
});
let map = item.as_object_mut().expect("built from a JSON object");
if !reference.authors.is_empty() {
map.insert(
"author".to_string(),
Value::Array(reference.authors.iter().map(|a| csl_name(a)).collect()),
);
}
if let Some(year) = reference.year {
map.insert("issued".to_string(), json!({ "date-parts": [[year]] }));
}
for (field, value) in [
("container-title", &reference.container),
("publisher", &reference.publisher),
("volume", &reference.volume),
("issue", &reference.issue),
("page", &reference.pages),
("edition", &reference.edition),
("DOI", &reference.doi),
("ISBN", &reference.isbn),
("PMID", &reference.pmid),
("PMCID", &reference.pmcid),
("URL", &reference.url),
] {
if let Some(value) = value {
map.insert(field.to_string(), Value::String(value.clone()));
}
}
item
}
/// Splits `Family, Given` into a CSL name, or keeps it whole.
///
/// A name with no comma is not a name this code can take apart — an
/// organization, or a single mononym — so it goes in `literal`, which is what
/// CSL has the field for.
fn csl_name(author: &str) -> Value {
match author.split_once(',') {
Some((family, given)) => json!({
"family": family.trim(),
"given": given.trim(),
}),
None => json!({ "literal": author.trim() }),
}
}
/// The CSL type for a course reference kind.
fn csl_kind(kind: ReferenceKind) -> &'static str {
match kind {
ReferenceKind::Book => "book",
ReferenceKind::Chapter => "chapter",
ReferenceKind::Article => "article-journal",
ReferenceKind::Preprint => "article",
ReferenceKind::Thesis => "thesis",
ReferenceKind::Website => "webpage",
ReferenceKind::Software => "software",
ReferenceKind::Dataset => "dataset",
ReferenceKind::Video => "motion_picture",
ReferenceKind::Other => "document",
}
}
// --- BibTeX ---
/// Renders the bibliography as BibTeX.
fn bibtex(course: &CourseFile) -> String {
let mut out = String::new();
for (key, reference) in &course.references {
out.push_str(&bibtex_entry(key, reference));
out.push('\n');
}
out
}
/// One BibTeX entry.
fn bibtex_entry(key: &str, reference: &Reference) -> String {
let mut fields: Vec<(&str, String)> = vec![("title", reference.title.clone())];
if !reference.authors.is_empty() {
fields.push(("author", reference.authors.join(" and ")));
}
if let Some(year) = reference.year {
fields.push(("year", year.to_string()));
}
if let Some(container) = &reference.container {
let field = match reference.kind {
ReferenceKind::Chapter => "booktitle",
_ => "journal",
};
fields.push((field, container.clone()));
}
for (field, value) in [
("volume", &reference.volume),
("number", &reference.issue),
("edition", &reference.edition),
("publisher", &reference.publisher),
("doi", &reference.doi),
("isbn", &reference.isbn),
("url", &reference.url),
("note", &reference.note),
] {
if let Some(value) = value {
fields.push((field, value.clone()));
}
}
if let Some(pages) = &reference.pages {
fields.push(("pages", en_dash(pages)));
}
let body: String = fields
.iter()
.map(|(field, value)| format!(" {field} = {{{}}},\n", escape_tex(value)))
.collect();
format!("@{}{{{key},\n{body}}}\n", bibtex_kind(reference.kind))
}
/// The BibTeX entry type for a course reference kind.
fn bibtex_kind(kind: ReferenceKind) -> &'static str {
match kind {
ReferenceKind::Book => "book",
ReferenceKind::Chapter => "incollection",
ReferenceKind::Article => "article",
ReferenceKind::Thesis => "phdthesis",
ReferenceKind::Website => "online",
// BibTeX proper has nothing for these. `misc` with a `note` is what
// every style guide says to do, and biblatex users can convert.
ReferenceKind::Preprint
| ReferenceKind::Software
| ReferenceKind::Dataset
| ReferenceKind::Video
| ReferenceKind::Other => "misc",
}
}
/// Escapes the characters BibTeX treats as syntax.
///
/// Deliberately short: a publisher called `John Wiley & Sons` is the case that
/// actually occurs, and escaping more than this risks mangling the `$...$` in a
/// title that carries real mathematics.
fn escape_tex(value: &str) -> String {
value.replace('&', "\\&").replace('%', "\\%")
}
/// Turns a hyphenated page range into the en dash BibTeX expects.
fn en_dash(pages: &str) -> String {
let parts: Vec<&str> = pages.split('-').collect();
if parts.len() == 2 && parts.iter().all(|p| !p.is_empty()) {
return format!("{}--{}", parts[0].trim(), parts[1].trim());
}
pages.to_string()
}
#[cfg(test)]
mod tests {
use super::*;
use crate::course::ReferenceRole;
fn course() -> CourseFile {
let mut c = CourseFile::skeleton("BIOSC 1540", "Computational Biology", "2026f");
c.references.insert(
"ismail2023bioinformatics".into(),
Reference {
label: Some("IBB".into()),
kind: ReferenceKind::Book,
role: ReferenceRole::Supplemental,
title: "Bioinformatics: A practical guide".into(),
authors: vec!["Ismail, H. D.".into()],
year: Some(2023),
publisher: Some("CRC Press".into()),
base_url: Some("https://library.example.org/ismail2023/".into()),
isbn: Some("9781032366423".into()),
..Reference::default()
},
);
c.references.insert(
"altschul1990basic".into(),
Reference {
label: Some("BLAST".into()),
kind: ReferenceKind::Article,
role: ReferenceRole::Required,
title: "Basic local alignment search tool".into(),
authors: vec![
"Altschul, S. F.".into(),
"Gish, W.".into(),
"Wiley & Sons".into(),
],
year: Some(1990),
container: Some("Journal of Molecular Biology".into()),
volume: Some("215".into()),
issue: Some("3".into()),
pages: Some("403-410".into()),
doi: Some("10.1016/S0022-2836(05)80360-2".into()),
pmid: Some("2231712".into()),
..Reference::default()
},
);
c
}
#[test]
fn hayagriva_nests_an_article_under_its_periodical() {
let out = render(&course(), Format::Hayagriva).unwrap();
assert!(out.contains("altschul1990basic:"), "{out}");
assert!(out.contains("type: article"), "{out}");
assert!(out.contains("type: periodical"), "{out}");
assert!(out.contains("Journal of Molecular Biology"), "{out}");
assert!(out.contains("page-range: 403-410"), "{out}");
assert!(out.contains("doi: 10.1016/S0022-2836(05)80360-2"), "{out}");
// Parses back as YAML, which is what Typst will do to it.
let back: serde_yaml_ng::Value = serde_yaml_ng::from_str(&out).unwrap();
assert!(back.get("ismail2023bioinformatics").is_some());
}
#[test]
fn hayagriva_keeps_a_books_publisher_on_the_book() {
let out = render(&course(), Format::Hayagriva).unwrap();
let parsed: serde_yaml_ng::Value = serde_yaml_ng::from_str(&out).unwrap();
let book = parsed.get("ismail2023bioinformatics").unwrap();
assert_eq!(book.get("publisher").unwrap().as_str(), Some("CRC Press"));
assert!(book.get("parent").is_none());
// `label`, `role`, and `base_url` have nowhere to go and stay behind.
assert!(book.get("label").is_none());
assert!(book.get("base_url").is_none());
}
#[test]
fn csl_json_splits_names_and_keeps_organizations_whole() {
let out = render(&course(), Format::CslJson).unwrap();
let items: Vec<Value> = serde_json::from_str(&out).unwrap();
let article = items
.iter()
.find(|i| i["id"] == "altschul1990basic")
.unwrap();
assert_eq!(article["type"], "article-journal");
assert_eq!(article["author"][0]["family"], "Altschul");
assert_eq!(article["author"][0]["given"], "S. F.");
assert_eq!(article["author"][2]["literal"], "Wiley & Sons");
assert_eq!(article["issued"]["date-parts"][0][0], 1990);
assert_eq!(article["DOI"], "10.1016/S0022-2836(05)80360-2");
}
#[test]
fn bibtex_escapes_ampersands_and_dashes_a_page_range() {
let out = render(&course(), Format::Bibtex).unwrap();
assert!(out.contains("@article{altschul1990basic,"), "{out}");
assert!(
out.contains("journal = {Journal of Molecular Biology},"),
"{out}"
);
assert!(out.contains("pages = {403--410},"), "{out}");
assert!(out.contains("Wiley \\& Sons"), "{out}");
assert!(out.contains("@book{ismail2023bioinformatics,"), "{out}");
}
#[test]
fn a_course_with_no_bibliography_renders_empty_rather_than_failing() {
let mut c = course();
c.references.clear();
assert_eq!(render(&c, Format::Bibtex).unwrap(), "");
assert_eq!(render(&c, Format::CslJson).unwrap().trim(), "[]");
}
}
+15 -22
View File
@@ -34,7 +34,7 @@
use crate::assessment::{AssessmentFile, Form, Placement};
use crate::catalog::Catalog;
use crate::course::{CourseFile, Reference};
use crate::course::CourseFile;
use crate::error::{Error, Result};
use crate::item::{Choice, Citation, Item, Solution};
use crate::markup;
@@ -226,12 +226,13 @@ fn question_block(
if item.has_options() {
b.push_str(":::: {.q-choices}\n");
let order = select::option_order(form, &placement.item, item.options.len());
let shown = item.administered(&placement.key, &placement.distractors);
let order = select::option_order(form, &placement.item, shown.len());
for (position, &source) in order.iter().enumerate() {
b.push_str(&format!(
"{}. {}\n",
position + 1,
markup::to_markdown(&item.options[source].text)
markup::to_markdown(&shown[source].text)
));
}
b.push_str("::::\n\n");
@@ -295,11 +296,12 @@ fn fragment(
/// A single-best-answer or multiple-response fragment: the key, the model answer,
/// the explanation, then per-distractor feedback.
fn choice_fragment(course: &CourseFile, placement: &Placement, item: &Item, form: &Form) -> String {
let order = select::option_order(form, &placement.item, item.options.len());
let shown = item.administered(&placement.key, &placement.distractors);
let order = select::option_order(form, &placement.item, shown.len());
let printed: Vec<(usize, &Choice)> = order
.iter()
.enumerate()
.map(|(position, &source)| (position, &item.options[source]))
.map(|(position, &source)| (position, shown[source]))
.collect();
let mut out = String::new();
@@ -472,7 +474,7 @@ fn cite_html(course: &CourseFile, citation: &Citation) -> String {
let Some(reference) = course.references.get(key) else {
return markup::escape_html(&citation.display());
};
let label = reference.label.as_deref().unwrap_or(key);
let label = reference.label_or(key);
let locator = citation.locator.as_deref().unwrap_or("");
let body = if locator.is_empty() {
markup::escape_html(label)
@@ -483,27 +485,12 @@ fn cite_html(course: &CourseFile, citation: &Citation) -> String {
markup::escape_html(locator)
)
};
match resolve_url(citation, reference) {
match citation.href(reference) {
Some(url) => format!("<a href=\"{}\">{body}</a>", markup::escape_html(&url)),
None => body,
}
}
/// Resolves a citation's link, from an explicit URL or a path joined to the
/// reference's base URL.
fn resolve_url(citation: &Citation, reference: &Reference) -> Option<String> {
if let Some(url) = &citation.url {
return Some(url.clone());
}
let path = citation.path.as_deref()?;
let base = reference.base_url.as_deref()?;
Some(match (base.ends_with('/'), path.starts_with('/')) {
(true, true) => format!("{base}{}", &path[1..]),
(false, false) => format!("{base}/{path}"),
_ => format!("{base}{path}"),
})
}
// --- math-aware markup ---
/// One run of source text, split on math delimiters.
@@ -858,6 +845,9 @@ items:
number: 1,
item: "b::q-mcq".into(),
version: None,
stem_digest: None,
distractors: Vec::new(),
variant: None,
fingerprint: None,
points: Some(1.0),
bonus: false,
@@ -873,6 +863,9 @@ items:
number: 2,
item: "b::q-open".into(),
version: None,
stem_digest: None,
distractors: Vec::new(),
variant: None,
fingerprint: None,
points: Some(2.0),
bonus: false,
+7 -1
View File
@@ -431,6 +431,9 @@ mod tests {
number: 1,
item: "b::q-1".into(),
version: None,
stem_digest: None,
distractors: Vec::new(),
variant: None,
fingerprint: None,
points: None,
bonus: false,
@@ -446,6 +449,9 @@ mod tests {
number: 2,
item: "b::q-2".into(),
version: None,
stem_digest: None,
distractors: Vec::new(),
variant: None,
fingerprint: None,
points: None,
bonus: false,
@@ -464,6 +470,6 @@ mod tests {
.filter(|p| p.was_printed())
.map(|p| p.number)
.collect();
assert_eq!(printable, vec![2]);
assert_eq!(printable, vec![1, 2]);
}
}
+1
View File
@@ -1281,6 +1281,7 @@ mod tests {
blueprint: Vec::new(),
patterns: Vec::new(),
warnings: Vec::new(),
dropped_detail: Vec::new(),
}
}
+3 -2
View File
@@ -411,11 +411,12 @@ pub fn build(
_ => (None, None),
};
let order = select::option_order(form, &placement.item, item.options.len());
let shown = item.administered(&placement.key, &placement.distractors);
let order = select::option_order(form, &placement.item, shown.len());
let options: Vec<Opt> = order
.iter()
.enumerate()
.map(|(position, source_index)| option(&item.options[*source_index], position, config))
.map(|(position, source_index)| option(shown[*source_index], position, config))
.collect();
let key = if config.reveal.shows_key() {
+13 -3
View File
@@ -18,12 +18,19 @@
//!
//! | File | Holds | Written by |
//! |:--|:--|:--|
//! | `course.yaml` | identity, policy, objectives, lectures | you |
//! | `course.yaml` | identity, policy, units | you |
//! | `references.yaml` | the works the course cites | you |
//! | `lectures/*.yaml` | one lecture: readings, and what it teaches | you |
//! | `objectives/*.yaml` | one objective and its learning targets | you |
//! | `banks/*.yaml` | items, with design intent and pooled statistics | you, then `calibrate` |
//! | `assessments/*.yaml` | what was given, to whom, when | `assemble`, then you |
//! | `data/*.parquet` | one row per student per item | `ingest` |
//!
//! Three of the four are hand-editable YAML meant to be reviewed in a pull request.
//! The first four are one course file split by subject; `coursebank course build`
//! prints the merged result, and a course that keeps everything in `course.yaml`
//! still loads unchanged. See [`course::fragment`].
//!
//! Most of these are hand-editable YAML meant to be reviewed in a pull request.
//! Only the response data is machine-only, and it is stored in an open columnar
//! format so pandas, polars, R, and DuckDB can all read it without this tool.
//!
@@ -95,6 +102,7 @@ pub mod data;
pub mod error;
pub mod export;
pub mod guide;
pub mod migrate;
pub mod model;
pub mod util;
@@ -102,6 +110,8 @@ pub use util::{date, hash, markup, rng, yaml, zipfile};
pub use model::{assessment, bank, catalog, course, history, item, layout, seal, taxonomy};
pub use course::fragment;
pub use authoring::{jsonschema, lint, select};
pub use data::store_parquet;
@@ -110,7 +120,7 @@ pub use data::{canvas, decode, gradescope, intake, responses, store};
pub use analysis::{calibrate, classical, diagnostic, irt, students};
pub use export::site;
pub use export::{lecture, practice, qti, report, typst};
pub use export::{lecture, practice, qti, references, report, typst};
pub use catalog::Catalog;
pub use course::{CourseFile, SCHEMA_VERSION};
+1961
View File
File diff suppressed because it is too large Load Diff
+3 -1
View File
@@ -12,7 +12,9 @@
//! ```text
//! taxonomy levels, cognitive processes, error types, status, flags
//! │
//! course course.yaml: identity, policy, objectives, lectures, stimuli
//! course identity, policy, objectives, lectures, stimuli
//! │ └─ course::fragment merges course.yaml, references.yaml,
//! │ lectures/*.yaml, objectives/*.yaml
//! │
//! item one question: stem, options, design intent, calibration
//! │
+60 -2
View File
@@ -248,11 +248,28 @@ pub struct Placement {
/// Printed question number. This is the join key to grading exports, which
/// is the entire reason this record exists.
pub number: u32,
/// The item's global id, `bank::item`.
/// The item's id, which names it course-wide. A pre-2.0 `bank::item`
/// value still resolves; `coursebank migrate ids` rewrites it.
pub item: String,
/// The item version used.
#[serde(default, skip_serializing_if = "Option::is_none")]
/// Retained only so a pre-2.0 record still loads. Ignored.
///
/// What it was for — knowing whether the item has changed since this
/// administration — is [`Placement::stem_digest`] and
/// [`Placement::fingerprint`], which say *what* changed rather than that
/// something did.
#[serde(default, skip_serializing)]
pub version: Option<u32>,
/// The stem's digest as administered. See [`crate::item::Item::stem_digest`].
///
/// Distinct from `fingerprint`, which covers the options too. A changed
/// fingerprint means the pooled statistics describe an older wording; a
/// changed stem digest means this is no longer the same question, which
/// [`crate::catalog::Catalog::validate_record`] treats as an error rather
/// than a note.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub stem_digest: Option<String>,
/// The content fingerprint as used, so later edits are detectable.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub fingerprint: Option<String>,
@@ -265,6 +282,26 @@ pub struct Placement {
/// The keyed letters as administered.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub key: Vec<String>,
/// The option ids offered alongside the key.
///
/// Resolved when the assessment is assembled and written out explicitly,
/// never sampled at export time. A blueprint may ask for a draw; the record
/// holds what was drawn. Otherwise a bank edit between assembling and
/// printing silently changes the paper, and the key printed on Tuesday
/// disagrees with the one printed on Wednesday.
///
/// Empty means the whole pool, which is what every pre-2.0 record meant.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub distractors: Vec<String>,
/// The digest of the item as this administration showed it.
///
/// See [`crate::item::Item::variant_digest`]. The key statistics pool on:
/// two administrations of one stem with different distractors are two
/// items, and averaging them is averaging different questions.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub variant: Option<String>,
/// The level as administered, denormalized so a record reads standalone.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub level: Option<Level>,
@@ -340,6 +377,27 @@ pub enum DropStyle {
}
impl Placement {
/// The variant this placement administered.
///
/// Recorded when the assessment was assembled; derived from the option set
/// otherwise, which is what makes every record written before 2.0 poolable
/// without being rewritten. A pre-2.0 placement names its key and no
/// distractors, which means the whole pool — a well-defined option set, and
/// so a well-defined variant.
///
/// # Arguments
///
/// * `item` - the item this placement names.
///
/// # Returns
///
/// The digest.
pub fn variant_of(&self, item: &crate::item::Item) -> String {
self.variant
.clone()
.unwrap_or_else(|| item.variant_digest(&self.key, &self.distractors))
}
/// Whether this placement was dropped by crediting every option.
///
/// # Returns
+77 -63
View File
@@ -352,8 +352,10 @@ fn validate_item(
if it.stem.trim().is_empty() {
issues.push("empty stem".into());
}
if it.version == 0 {
issues.push("version must be at least 1".into());
if let Some(replaced) = &it.supersedes {
if *replaced == it.id {
issues.push("supersedes names this item".into());
}
}
// --- options ----
@@ -379,21 +381,37 @@ fn validate_item(
if o.text.trim().is_empty() {
issues.push(format!("option {pos}: empty text"));
}
let letter_ok = o.id.len() == 1
&& o.id
.chars()
.next()
.map(|c| c.is_ascii_uppercase() && c <= 'H')
.unwrap_or(false);
if !letter_ok {
// Two forms are accepted: the 2.0 name, and the letter that preceded
// it. A bank migrates when `coursebank migrate options` is run on it,
// not when the tool is upgraded, so a course mid-migration still loads.
let slug_ok = o.id.strip_prefix("o-").is_some_and(|rest| {
!rest.is_empty()
&& !rest.starts_with('-')
&& !rest.ends_with('-')
&& !rest.contains("--")
&& rest
.chars()
.all(|c| c.is_ascii_lowercase() || c.is_ascii_digit() || c == '-')
});
if !slug_ok && !Item::is_legacy_option_id(&o.id) {
issues.push(format!(
"option {pos}: id `{}` must be a single letter A through H",
"option {pos}: id `{}` is neither a name such as `o-fourth-line` nor a pre-2.0 \
letter A through H",
o.id
));
}
if seen.contains(&o.id.as_str()) {
issues.push(format!("option {pos}: duplicate option id `{}`", o.id));
}
if let Some(retirement) = &o.retired {
if retirement.reason.trim().is_empty() {
issues.push(format!(
"option {pos}: retired without a reason. The reason is the finding — what \
the option did or failed to do — and it is the only part of a retirement \
that is worth anything later."
));
}
}
seen.push(&o.id);
let credit = o.credit();
@@ -433,14 +451,27 @@ fn validate_item(
}
// --- key ---
// Counted over the pool that can still be drawn: a retired option is a
// record, not an offer.
let (live_keys, live_distractors) = it.pool();
let keys = it.key_indices();
match it.format {
Format::SingleBestAnswer => {
if keys.len() != 1 {
issues.push(format!(
"single_best_answer needs exactly one keyed option, has {}",
keys.len()
));
// Several defensible keys is a pool, not a bug — it is what lets you
// test whether "fourth" or "last of the four" is doing the work.
// Exactly one of them reaches a student, and that is the
// placement's business: see
// [`crate::catalog::Catalog::validate_record`].
if live_keys.is_empty() {
issues.push("single_best_answer needs at least one keyed option".into());
}
if live_distractors.is_empty() {
let retired = it.options.iter().any(|o| o.retired.is_some());
issues.push(if retired {
"every distractor is retired, so nothing can be drawn against the key".into()
} else {
"has no option that is not keyed correct, so it asks nothing".to_string()
});
}
}
Format::MultipleResponse => {
@@ -514,10 +545,10 @@ fn validate_item(
));
}
}
for letter in c.option_stats.keys() {
if it.option(letter).is_none() {
for option in c.option_stats.keys() {
if it.option(option).is_none() {
issues.push(format!(
"calibration.option_stats has `{letter}`, which is not an option of this item"
"calibration.option_stats has `{option}`, which is not an option of this item"
));
}
}
@@ -533,25 +564,6 @@ fn validate_item(
}
}
// --- history must be coherent ------
let mut last_version = 0u32;
for (i, h) in it.history.iter().enumerate() {
if h.version <= last_version {
issues.push(format!(
"history entry {} has version {} which does not increase",
i + 1,
h.version
));
}
last_version = h.version;
}
if !it.history.is_empty() && last_version > it.version {
issues.push(format!(
"history records version {last_version} but the item says version {}",
it.version
));
}
// --- retirement -----
if it.retired.is_some() && it.status != Status::Retired {
issues.push(format!(
@@ -790,20 +802,43 @@ mod tests {
options:
- { id: A, text: a, correct: true }
- { id: B, text: b, correct: true }
- id: q-a-003
status: draft
level: 1
format: single_best_answer
stem: s
options:
- { id: o-key-one, text: a, correct: true }
- { id: o-key-two, text: b, correct: true }
- { id: o-wrong-one, text: c }
- { id: o-wrong-two, text: d }
"#,
);
let issues = b.validate(None);
// No key at all is still a bank problem: nothing can be drawn from it.
assert!(
issues
.iter()
.any(|i| i.contains("exactly one keyed option"))
.any(|i| i.starts_with("q-a-001") && i.contains("at least one keyed option")),
"{issues:?}"
);
assert_eq!(
// Keying every option is still a bank problem, for the older reason: a
// question with nothing to choose against asks nothing.
assert!(
issues
.iter()
.filter(|i| i.contains("exactly one keyed option"))
.count(),
2
.any(|i| i.starts_with("q-a-002") && i.contains("asks nothing")),
"{issues:?}"
);
// Two defensible keys alongside real distractors is not a problem. It
// is the pool doing its job — it is what lets you test whether the
// wording of the key is what students are answering. Exactly one of
// them reaches a student, and that is checked against the placement
// that administers it, in `Catalog::validate_placement`.
assert!(
!issues.iter().any(|i| i.starts_with("q-a-003")
&& (i.contains("keyed option") || i.contains("asks nothing"))),
"{issues:?}"
);
}
@@ -1027,27 +1062,6 @@ learning_targets:
);
}
#[test]
fn history_versions_must_increase() {
let b = bank(
r#"
- id: q-a-001
version: 2
status: draft
level: 1
stem: s
options:
- { id: A, text: a, correct: true }
- { id: B, text: b }
history:
- { version: 2, date: 2026-01-01, change: second }
- { version: 1, date: 2026-01-02, change: first }
"#,
);
let issues = b.validate(None);
assert!(issues.iter().any(|i| i.contains("does not increase")));
}
#[test]
fn level_counts_exclude_drafts_and_bonuses() {
let b = bank(
+232 -62
View File
@@ -5,7 +5,7 @@
//! Loading a whole course at once, and reporting on what it contains.
//!
//! A [`Catalog`] is every bank in a course, indexed so that an item can be found
//! by its global id (`bank::item`), and so that questions like "how many Apply
//! by its id, and so that questions like "how many Apply
//! level items do I have on lecture 12" have a cheap answer.
//!
//! The global id is the join key for everything downstream: assessment records
@@ -20,19 +20,24 @@
use std::collections::{BTreeMap, BTreeSet};
use std::path::{Path, PathBuf};
use crate::assessment::AssessmentFile;
use crate::assessment::{AssessmentFile, Placement};
use crate::bank::BankFile;
use crate::course::CourseFile;
use crate::error::{Error, Result};
use crate::item::Item;
use crate::layout::Layout;
use crate::taxonomy::{Level, Status, Tier};
use crate::taxonomy::{Format, Level, Status, Tier};
use crate::yaml;
/// One item plus everything needed to locate it again.
#[derive(Debug, Clone)]
pub struct Entry {
/// The globally unique id, `bank::item`.
/// The item's id, which names it course-wide.
///
/// The bank is in [`Entry::bank`] and the file in [`Entry::path`], neither
/// of which is part of the identity: this is the join key that response
/// data carries, and a join key with a file name in it renames itself every
/// time the files are reorganized.
pub uid: String,
/// The bank id.
pub bank: String,
@@ -92,7 +97,7 @@ impl Catalog {
/// id, since either makes the join key ambiguous.
pub fn load(root: &Path) -> Result<Catalog> {
let layout = Layout::new(root);
let course = CourseFile::load(&layout.course_file())?;
let course = CourseFile::load_dir(root)?;
let mut catalog = Catalog {
course,
layout,
@@ -117,9 +122,13 @@ impl Catalog {
}
catalog.banks.insert(bank_id.clone(), bank.bank.clone());
for (i, item) in bank.items.into_iter().enumerate() {
let uid = format!("{bank_id}::{}", item.id);
if catalog.index.contains_key(&uid) {
problems.push(format!("duplicate global item id `{uid}`"));
let uid = item.id.clone();
if let Some(first) = catalog.index.get(&uid) {
problems.push(format!(
"item id `{uid}` is used twice: in bank `{}` and in bank `{bank_id}`. An \
id names one question course-wide, because response data joins on it.",
catalog.entries[*first].bank
));
continue;
}
catalog.index.insert(uid.clone(), catalog.entries.len());
@@ -141,15 +150,22 @@ impl Catalog {
/// Looks up an item by global id.
///
/// A pre-2.0 `bank::item` id resolves to the item it used to name, so an
/// assessment record or a parquet file written before the change still
/// joins. See [`crate::item::canonical_id`].
///
/// # Arguments
///
/// * `uid` - the global id, `bank::item`.
/// * `uid` - the item id, in either form.
///
/// # Returns
///
/// The entry, or `None`.
pub fn get(&self, uid: &str) -> Option<&Entry> {
self.index.get(uid).map(|i| &self.entries[*i])
self.index
.get(uid)
.or_else(|| self.index.get(crate::item::canonical_id(uid)))
.map(|i| &self.entries[*i])
}
/// Looks up an item by global id, erroring when absent.
@@ -173,45 +189,37 @@ impl Catalog {
})
}
/// Resolves a possibly-unqualified id to a global id.
/// Resolves an id written in either form to the canonical one.
///
/// Typing `q-mm-kinetics-001` on the command line should work when that id is
/// unambiguous across the course, because remembering which bank a question
/// lives in is exactly the sort of bookkeeping this tool exists to remove.
/// Since 2.0 an item id is already course-wide, so this is the identity for
/// anything current. What it is still for is the old `bank::item` form,
/// which appears in assessment records, seals, and response files written
/// before the change, and which a person may well still type.
///
/// # Arguments
///
/// * `id` - a global id, or a bare item id.
/// * `id` - an item id, in either form.
///
/// # Returns
///
/// The global id.
/// The canonical id.
///
/// # Errors
///
/// Returns [`Error::Unresolved`] when nothing matches, or [`Error::Usage`]
/// when a bare id matches items in more than one bank.
/// Returns [`Error::Unresolved`] when nothing matches.
pub fn resolve(&self, id: &str) -> Result<String> {
if self.index.contains_key(id) {
return Ok(id.to_string());
}
let matches: Vec<&Entry> = self.entries.iter().filter(|e| e.item.id == id).collect();
match matches.len() {
0 => Err(Error::Unresolved {
kind: "item",
id: id.to_string(),
context: None,
}),
1 => Ok(matches[0].uid.clone()),
_ => Err(Error::usage(format!(
"`{id}` is ambiguous; it exists in {}. Use the full `bank::item` form.",
matches
.iter()
.map(|e| e.bank.as_str())
.collect::<Vec<_>>()
.join(", ")
))),
let canonical = crate::item::canonical_id(id);
if self.index.contains_key(canonical) {
return Ok(canonical.to_string());
}
Err(Error::Unresolved {
kind: "item",
id: id.to_string(),
context: None,
})
}
/// Every item that may be placed on a graded assessment.
@@ -232,12 +240,10 @@ impl Catalog {
///
/// Problems, prefixed with the file they came from.
pub fn validate(&self) -> Result<Vec<String>> {
let mut issues: Vec<String> = self
.course
.validate()
.into_iter()
.map(|m| format!("course.yaml: {m}"))
.collect();
// Not prefixed here: `CourseFile::validate` attributes each message to
// the fragment that defined the id, which for an unsplit course is
// `course.yaml` and for a split one is the file worth opening.
let mut issues: Vec<String> = self.course.validate();
for path in yaml::list_yaml(&self.layout.banks())? {
let bank = BankFile::load_resolved(&path)?;
@@ -255,6 +261,154 @@ impl Catalog {
Ok(issues)
}
/// Checks one placement's option set against the item's pool.
///
/// The checks that moved here from the bank when options became a pool. A
/// bank holding two defensible keys and six distractors is sound; what has
/// to hold for a *form* is that exactly one key reached the student, that
/// none of the distractors was true, and that the count matches policy.
/// None of that can be decided by looking at the item alone.
///
/// # Arguments
///
/// * `p` - the placement.
/// * `item` - the item it names.
///
/// # Returns
///
/// One message per problem.
fn validate_placement(&self, p: &Placement, item: &Item) -> Vec<String> {
let mut issues = Vec::new();
if !item.format.has_options() {
return issues;
}
let at = |number: u32| format!("question {number} ({})", p.item);
for id in p.key.iter().chain(p.distractors.iter()) {
if item.option(id).is_none() {
issues.push(format!(
"{}: `{id}` is not an option of this item",
at(p.number)
));
}
}
for id in &p.distractors {
if p.key.iter().any(|k| k == id) {
issues.push(format!(
"{}: `{id}` is listed as both the key and a distractor",
at(p.number)
));
}
if item.option(id).is_some_and(|o| o.correct) {
issues.push(format!(
"{}: `{id}` is offered as a distractor but the bank keys it correct",
at(p.number)
));
}
}
for id in &p.key {
if item.option(id).is_some_and(|o| !o.correct) {
issues.push(format!(
"{}: `{id}` is keyed correct here but the bank does not key it. An option is \
true or it is not; which true option a form uses is this record's choice, \
but not whether it is true.",
at(p.number)
));
}
}
let single = item.format == Format::SingleBestAnswer;
if single && p.key.len() > 1 {
issues.push(format!(
"{}: single_best_answer administers exactly one key, this names {}",
at(p.number),
p.key.len()
));
}
// No distractor list means the whole pool, which is what a pre-2.0
// record means and what an item whose pool is its form still means.
// The record still has to say which key, when the pool offers a choice.
if p.distractors.is_empty() {
let (keys, _) = item.pool();
if p.key.is_empty() && keys.len() > 1 && single {
issues.push(format!(
"{}: the item offers {} defensible keys, so the record has to say which one \
this assessment used",
at(p.number),
keys.len()
));
}
} else {
let shown = item.administered(&p.key, &p.distractors);
let expected = self.course.policy.options_per_item;
if shown.len() != expected {
issues.push(format!(
"{}: administers {} option(s), but course policy is {expected} per item",
at(p.number),
shown.len()
));
}
if single && p.key.is_empty() {
issues.push(format!(
"{}: names its distractors but not its key, so what was marked correct is \
left to whatever the bank says today",
at(p.number)
));
}
}
if let Some(recorded) = &p.variant {
if *recorded != item.variant_digest(&p.key, &p.distractors) {
issues.push(format!(
"{}: an administered option has been reworded since this assessment. \
Statistics pooled under this variant describe the older wording.",
at(p.number)
));
}
}
issues
}
/// Checks every sealed administration's stems against the bank.
///
/// The seal is the authority on what was administered, so this is the
/// comparison that matters: a record can be edited, but a seal is written
/// before the exam is printed and digested against tampering. A stem that
/// no longer matches the one a cohort answered means the id now names a
/// different question, and every statistic pooled under it is describing
/// two things at once.
///
/// # Arguments
///
/// * `seals` - the sealed administrations to check.
///
/// # Returns
///
/// One message per stem that has moved out from under its seal.
pub fn validate_seals(&self, seals: &[crate::seal::SealFile]) -> Vec<String> {
let mut issues = Vec::new();
for seal in seals {
for item in &seal.items {
let Some(digest) = &item.stem_digest else {
continue;
};
let Some(entry) = self.get(&item.item) else {
continue;
};
if *digest != entry.item.stem_digest() {
issues.push(format!(
"{}: question {} (`{}`) was administered with a different stem than the \
bank now holds. Statistics from that administration describe the older \
wording; give the new wording its own id.",
seal.seal.assessment, item.number, item.item
));
}
}
}
issues
}
/// Validates a record's internal invariants, then its references against this
/// catalog: unknown items, keys that drifted, and fingerprints showing the
/// item was reworded since it was administered.
@@ -264,6 +418,22 @@ impl Catalog {
match self.get(&p.item) {
None => issues.push(format!("question {}: unknown item `{}`", p.number, p.item)),
Some(entry) => {
// The rule the stem digest exists to enforce. A changed
// fingerprint is a note: the statistics describe an older
// wording. A changed stem is an error: whatever was
// administered is not the question the bank now holds, so
// the id is being reused for two different questions.
if let Some(digest) = &p.stem_digest {
if *digest != entry.item.stem_digest() {
issues.push(format!(
"question {} ({}): the stem has been reworded since this \
assessment. A reworded stem is a new question: give the new \
wording a new id with `supersedes: {}`, and leave this one as \
it was administered.",
p.number, p.item, p.item
));
}
}
if let Some(fp) = &p.fingerprint {
if *fp != entry.item.fingerprint() {
issues.push(format!(
@@ -273,11 +443,7 @@ impl Catalog {
));
}
}
if !p.key.is_empty() && p.key != entry.item.key_letters() {
issues.push(format!(
"question {} ({}): the recorded key {:?} differs from the item's current key {:?}",
p.number, p.item, p.key, entry.item.key_letters()));
}
issues.extend(self.validate_placement(p, &entry.item));
}
}
}
@@ -705,13 +871,32 @@ items:
let cat = Catalog::load(&dir).expect("catalog loads");
assert_eq!(cat.entries.len(), 1);
assert_eq!(cat.entries[0].uid, "b1::q-x-001");
// The id names the item course-wide; the bank is where it is kept.
assert_eq!(cat.entries[0].uid, "q-x-001");
assert_eq!(cat.entries[0].bank, "b1");
assert!(cat.get("q-x-001").is_some());
assert_eq!(cat.resolve("q-x-001").unwrap(), "q-x-001");
// A record or a parquet file written before 2.0 still joins.
assert!(cat.get("b1::q-x-001").is_some());
assert_eq!(cat.resolve("q-x-001").unwrap(), "b1::q-x-001");
assert_eq!(cat.resolve("b1::q-x-001").unwrap(), "q-x-001");
assert!(cat.get("b1::q-nonexistent").is_none());
assert!(cat.validate().unwrap().is_empty());
let _ = std::fs::remove_dir_all(&dir);
}
#[test]
fn one_item_id_in_two_banks_is_fatal() {
let dir = tmp("dupitem");
write_course(&dir, "");
write_bank(&dir, "b1.yaml", APPROVED);
write_bank(&dir, "b2.yaml", &APPROVED.replace("id: b1", "id: b2"));
let message = Catalog::load(&dir).unwrap_err().to_string();
assert!(message.contains("`q-x-001` is used twice"), "{message}");
assert!(message.contains("course-wide"), "{message}");
let _ = std::fs::remove_dir_all(&dir);
}
#[test]
fn duplicate_bank_ids_are_fatal() {
let dir = tmp("dupbank");
@@ -723,21 +908,6 @@ items:
let _ = std::fs::remove_dir_all(&dir);
}
#[test]
fn ambiguous_bare_ids_are_rejected() {
let dir = tmp("ambig");
write_course(&dir, "");
write_bank(&dir, "a.yaml", APPROVED);
write_bank(&dir, "b.yaml", &APPROVED.replace("id: b1", "id: b2"));
let cat = Catalog::load(&dir).expect("distinct banks load");
assert_eq!(cat.entries.len(), 2);
let err = cat.resolve("q-x-001").expect_err("bare id is ambiguous");
assert!(format!("{err}").contains("ambiguous"));
// The fully qualified form still works.
assert_eq!(cat.resolve("b2::q-x-001").unwrap(), "b2::q-x-001");
let _ = std::fs::remove_dir_all(&dir);
}
#[test]
fn coverage_finds_real_gaps() {
let dir = tmp("coverage");
+455 -23
View File
@@ -4,20 +4,29 @@
//! The course file: identity plus the registries every bank references.
//!
//! Learning objectives, their learning targets, and lectures are declared once,
//! in `course.yaml`, and referenced by id from items. That is the single most load-bearing decision in
//! Learning objectives, their learning targets, and lectures are declared once
//! and referenced by id from items. That is the single most load-bearing decision in
//! the schema. It means an objective's wording lives in exactly one place, so
//! rewording it updates every report; it means a report can name what a student
//! missed by objective rather than by question number; and it means a dangling
//! reference is a hard error instead of a silently misspelled string that splits
//! your coverage table into two near-identical rows.
//!
//! "Once" is a claim about ids, not about files. [`CourseFile`] is the resolved
//! model, and it may be assembled from a directory of fragments — one file per
//! lecture, one per objective — as well as from a single `course.yaml`. Either
//! way an id has exactly one definition site, and [`CourseFile::origins`]
//! records which file that was. See [`fragment`] for the merge and the rules
//! that keep it honest.
//!
//! The course file also declares the term. Items live across terms, so the term
//! belongs to the course and the administration, never to the item.
pub mod fragment;
use std::collections::BTreeMap;
use std::fmt;
use std::path::Path;
use std::path::{Path, PathBuf};
use serde::de::{self, MapAccess, Visitor};
use serde::ser::SerializeMap;
@@ -28,8 +37,19 @@ use crate::error::{Error, Result};
use crate::taxonomy::Level;
use crate::yaml;
use fragment::Section;
/// The schema version this build of the tool writes.
pub const SCHEMA_VERSION: &str = "1.0";
pub const SCHEMA_VERSION: &str = "2.0";
/// The schema major versions this build can read.
///
/// A 1.0 repository loads unchanged. What 2.0 changes is the shape of two
/// things, and both are tolerated on the way in: a course file may be split
/// into fragments, and an item is named course-wide rather than as
/// `bank::item`. `coursebank migrate` rewrites files into the 2.0 form when you
/// are ready; nothing forces it.
pub const SUPPORTED_MAJORS: [&str; 2] = ["1", "2"];
/// The canonical file name inside a course directory.
pub const COURSE_FILE: &str = "course.yaml";
@@ -91,6 +111,16 @@ pub struct CourseFile {
/// Shared stimuli for case-based testlets, keyed by id.
#[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
pub stimuli: BTreeMap<String, Stimulus>,
/// Which file defined each id, relative to the course root.
///
/// Populated by [`fragment::assemble`] and empty for a course parsed
/// straight out of one file by [`CourseFile::load`]. It is what makes a
/// validation message able to name the file to open, which matters rather a
/// lot once one course is forty files. Not serialized: it describes where
/// the model came from, not what it says.
#[serde(skip)]
pub origins: BTreeMap<(Section, String), PathBuf>,
}
/// Course identity.
@@ -285,6 +315,15 @@ pub struct Lecture {
/// Where the slides live, for study guidance in student reports.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub slides_url: Option<String>,
/// The objectives this session develops.
///
/// The registration direction: you write what a lecture covers while
/// planning the lecture, and each named objective gains this lecture in its
/// [`Objective::lectures`] list during [`fragment::assemble`]. Declaring the
/// pair from the objective's side instead is equivalent, and declaring it
/// from both is redundant rather than contradictory — the two are unioned.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub teaches: Vec<String>,
/// Assigned readings for the session, in the order you assign them.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub readings: Vec<Reading>,
@@ -335,8 +374,21 @@ pub struct Reference {
#[serde(default, skip_serializing_if = "Option::is_none")]
pub pages: Option<String>,
/// DOI, bare: `10.1038/nature12373`.
///
/// For a manuscript this is usually the only link worth storing: it is the
/// identifier of the work rather than of one copy of it, and [`Reference::href`]
/// turns it into a URL.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub doi: Option<String>,
/// arXiv id, bare: `2301.00001` or `q-bio/0501001`.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub arxiv: Option<String>,
/// PubMed Central id, which hosts the full text: `PMC3084216`.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub pmcid: Option<String>,
/// PubMed id, which hosts a record about the work: `21471563`.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub pmid: Option<String>,
/// ISBN, for a book.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub isbn: Option<String>,
@@ -353,6 +405,123 @@ pub struct Reference {
pub note: Option<String>,
}
impl Reference {
/// Where to send a reader, most specific first.
///
/// The one link resolution in the crate. Every exporter used to carry its
/// own copy of the `base_url` join, which meant a reading list, a printed
/// key, a practice sheet, and a student report could disagree about where a
/// citation points — and that none of them linked a journal article, since
/// an article has no `base_url` to join a path to.
///
/// The order is from the exact location outward: a link to §1.4 beats a link
/// to the work, and a link to the work beats nothing.
///
/// # Arguments
///
/// * `url` - a full URL for the exact location, from a reading or citation.
/// * `path` - a location under this work's `base_url`.
///
/// # Returns
///
/// The most specific link available, or `None` for a work with no online
/// location at all.
pub fn href(&self, url: Option<&str>, path: Option<&str>) -> Option<String> {
if let Some(url) = url {
return Some(url.to_string());
}
if let (Some(base), Some(path)) = (self.base_url.as_deref(), path) {
return Some(join_url(base, path));
}
if let Some(url) = &self.url {
return Some(url.clone());
}
self.identifier_url()
}
/// The link this work's identifiers resolve to, ignoring any location inside
/// it.
///
/// DOI first, because it names the work rather than one copy of it. Then
/// arXiv and PubMed Central, which host the article itself, before PubMed,
/// which hosts a record about it.
///
/// # Returns
///
/// A URL, or `None` when the work carries no identifier.
pub fn identifier_url(&self) -> Option<String> {
if let Some(doi) = self.doi.as_deref().map(bare_doi) {
return Some(format!("https://doi.org/{doi}"));
}
if let Some(id) = self.arxiv.as_deref().map(bare_arxiv) {
return Some(format!("https://arxiv.org/abs/{id}"));
}
if let Some(id) = self.pmcid.as_deref().map(str::trim) {
let id = if id.starts_with("PMC") {
id.to_string()
} else {
format!("PMC{id}")
};
return Some(format!("https://www.ncbi.nlm.nih.gov/pmc/articles/{id}/"));
}
if let Some(id) = self.pmid.as_deref().map(str::trim) {
return Some(format!("https://pubmed.ncbi.nlm.nih.gov/{id}/"));
}
None
}
/// The short form a reading list shows: the label, or the citation key.
///
/// # Arguments
///
/// * `key` - the citation key, used when the work declares no label.
///
/// # Returns
///
/// The label to print.
pub fn label_or<'a>(&'a self, key: &'a str) -> &'a str {
self.label.as_deref().unwrap_or(key)
}
}
/// A DOI with any resolver prefix stripped, so `href` cannot produce
/// `https://doi.org/https://doi.org/10...`.
fn bare_doi(doi: &str) -> &str {
let doi = doi.trim();
for prefix in [
"https://doi.org/",
"http://doi.org/",
"https://dx.doi.org/",
"http://dx.doi.org/",
"doi:",
] {
if let Some(rest) = doi.strip_prefix(prefix) {
return rest;
}
}
doi
}
/// An arXiv id with the `arXiv:` prefix stripped.
fn bare_arxiv(id: &str) -> &str {
let id = id.trim();
for prefix in ["arXiv:", "arxiv:", "https://arxiv.org/abs/"] {
if let Some(rest) = id.strip_prefix(prefix) {
return rest;
}
}
id
}
/// Joins a base URL and a path without doubling or dropping the separator.
fn join_url(base: &str, path: &str) -> String {
match (base.ends_with('/'), path.starts_with('/')) {
(true, true) => format!("{base}{}", &path[1..]),
(false, false) => format!("{base}/{path}"),
_ => format!("{base}{path}"),
}
}
/// The kind of work, chosen to map onto BibTeX entry types.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
#[serde(rename_all = "kebab-case")]
@@ -447,19 +616,10 @@ impl Reading {
///
/// # Returns
///
/// `url` when given, otherwise the reference's `base_url` joined with `path`,
/// otherwise `None`.
/// The most specific link available, which for a manuscript with a DOI and
/// no `path` is the DOI. See [`Reference::href`].
pub fn resolve_url(&self, reference: &Reference) -> Option<String> {
if let Some(url) = &self.url {
return Some(url.clone());
}
let path = self.path.as_deref()?;
let base = reference.base_url.as_deref()?;
Some(match (base.ends_with('/'), path.starts_with('/')) {
(true, true) => format!("{base}{}", &path[1..]),
(false, false) => format!("{base}/{path}"),
_ => format!("{base}{path}"),
})
reference.href(self.url.as_deref(), self.path.as_deref())
}
/// A short citation for a report: `KKW §6.1`.
@@ -476,7 +636,7 @@ impl Reading {
if let Some(text) = &self.text {
return text.clone();
}
let label = reference.label.as_deref().unwrap_or(key);
let label = reference.label_or(key);
match &self.locator {
Some(locator) => format!("{label} {locator}"),
None => label.to_string(),
@@ -761,7 +921,12 @@ impl CourseFile {
yaml::read(path)
}
/// Finds and loads the course file for a course directory.
/// Loads the course for a course directory, merging every fragment it holds.
///
/// This is the entry point every command uses. A directory holding only
/// `course.yaml` gives the same result it always did; one that also holds
/// `lectures/`, `objectives/`, or `references.yaml` gets them merged in. See
/// [`fragment::assemble`].
///
/// # Arguments
///
@@ -769,34 +934,167 @@ impl CourseFile {
///
/// # Returns
///
/// The parsed course file.
/// The merged course file.
///
/// # Errors
///
/// Propagates load errors, including absence of `course.yaml`.
/// Propagates load errors, including absence of `course.yaml`, and returns
/// [`Error::Invalid`] when two files define the same id.
pub fn load_dir(dir: &Path) -> Result<CourseFile> {
CourseFile::load(&dir.join(COURSE_FILE))
fragment::assemble(dir)
}
/// Which file defined an id, and which registry it was in.
///
/// # Arguments
///
/// * `id` - a unit, lecture, objective, target, reference, or stimulus id.
///
/// # Returns
///
/// The section and the path relative to the course root, or `None` for an
/// unknown id or a course that was not assembled from fragments.
pub fn origin(&self, id: &str) -> Option<(Section, &Path)> {
Section::ALL.iter().find_map(|section| {
self.origins
.get(&(*section, id.to_string()))
.map(|path| (*section, path.as_path()))
})
}
/// Every file this course was assembled from, in sorted order.
///
/// # Returns
///
/// The paths relative to the course root, empty for a course parsed from a
/// single file by [`CourseFile::load`].
pub fn fragment_paths(&self) -> Vec<&Path> {
let mut paths: Vec<&Path> = self.origins.values().map(PathBuf::as_path).collect();
paths.sort_unstable();
paths.dedup();
paths
}
/// Writes the course file back out as YAML.
///
/// Refuses to write a course that was assembled from more than one file,
/// because the merged model has no home on disk: writing it to
/// `course.yaml` would leave every fragment defining ids the root file also
/// defines, which is the one thing [`fragment::assemble`] treats as an
/// error. Use [`CourseFile::write_resolved`] for an inspection copy.
///
/// # Arguments
///
/// * `path` - destination path.
///
/// # Errors
///
/// Returns [`Error::Io`] on a write failure.
/// Returns [`Error::Usage`] for a fragmented course and [`Error::Io`] on a
/// write failure.
pub fn save(&self, path: &Path) -> Result<()> {
let sources = self.fragment_paths();
if sources.len() > 1 {
return Err(Error::usage(format!(
"this course is assembled from {} files, so it cannot be written back to one. \
Edit the fragment that owns what you are changing, or use `coursebank course \
build` for a merged copy.",
sources.len()
)));
}
yaml::write(path, self)
}
/// Writes the merged course as YAML, for reading rather than for loading.
///
/// The output carries a banner saying so. It is what `coursebank course
/// build` writes, and nothing in the tool reads it back: a generated file
/// that commands depend on is a file that goes stale.
///
/// # Arguments
///
/// * `path` - destination path.
///
/// # Errors
///
/// Returns [`Error::Io`] on a write failure, or [`Error::Other`] if the
/// model cannot be represented as YAML.
pub fn write_resolved(&self, path: &Path) -> Result<()> {
let body = yaml::to_string(self)?;
let banner = format!(
"# Generated by `coursebank course build` from {} file(s). Do not edit: nothing\n\
# reads this, and the next build overwrites it. Edit the fragments instead.\n",
self.fragment_paths().len().max(1)
);
yaml::write_text(path, &format!("{banner}{body}"))
}
/// Checks internal consistency of the registries.
///
/// Each message is prefixed with the file that defined the id it is about,
/// when that is known. For an unsplit course that is always `course.yaml`,
/// which is what the messages used to say.
///
/// # Returns
///
/// Every problem found, empty when the file is sound.
pub fn validate(&self) -> Vec<String> {
self.problems()
.into_iter()
.map(|issue| self.attribute(issue))
.collect()
}
/// Prefixes one validation message with the fragment it concerns.
///
/// The id is taken from the first backticked token in the message, since
/// every message that is about a registry entry names it first. Messages
/// about `course`, `policy`, or `units` are attributed by section instead,
/// because the first thing they quote is a field or a grade letter.
///
/// # Arguments
///
/// * `issue` - the message.
///
/// # Returns
///
/// The message, prefixed with a path when one is known.
fn attribute(&self, issue: String) -> String {
let section = if issue.starts_with("course.") {
Some(Section::Course)
} else if issue.starts_with("policy.") {
Some(Section::Policy)
} else if issue.starts_with("units") {
Some(Section::Units)
} else {
None
};
let path = match section {
Some(section) => self.section_origin(section),
None => issue
.split('`')
.nth(1)
.and_then(|id| self.origin(id))
.map(|(_, path)| path),
};
match path {
Some(path) => format!("{}: {issue}", path.display()),
None => issue,
}
}
/// The file that declared a whole section, for the sections that are not
/// keyed by id.
fn section_origin(&self, section: Section) -> Option<&Path> {
self.origins
.iter()
.find(|((s, _), _)| *s == section)
.map(|(_, path)| path.as_path())
}
/// The validation messages, before they are attributed to files.
fn problems(&self) -> Vec<String> {
let mut issues = Vec::new();
if self.course.code.trim().is_empty() {
@@ -874,6 +1172,25 @@ impl CourseFile {
if reference.title.trim().is_empty() {
issues.push(format!("reference `{key}`: empty title"));
}
// Checked rather than silently coerced: `href` strips a resolver
// prefix, but something that is not a DOI at all would become a
// link that 404s on a student's reading list.
if let Some(doi) = &reference.doi {
if !bare_doi(doi).starts_with("10.") {
issues.push(format!(
"reference `{key}`: `{doi}` is not a DOI. Write it bare, as \
10.1038/nature12373."
));
}
}
if let Some(pmid) = &reference.pmid {
if !pmid.trim().chars().all(|c| c.is_ascii_digit()) {
issues.push(format!(
"reference `{key}`: pmid `{pmid}` is not a number. A `PMC...` id goes in \
`pmcid`."
));
}
}
if let Some(label) = &reference.label {
labels.entry(label.as_str()).or_default().push(key);
}
@@ -897,6 +1214,20 @@ impl CourseFile {
issues.push(format!("lecture `{id}`: unknown unit `{u}`"));
}
}
for objective in &lec.teaches {
if !self.learning_objectives.contains_key(objective) {
if self.learning_targets.contains_key(objective) {
issues.push(format!(
"lecture `{id}`: `teaches` names the target `{objective}`, but it \
registers objectives. A target is reached through its objective."
));
} else {
issues.push(format!(
"lecture `{id}`: `teaches` names an unknown objective `{objective}`"
));
}
}
}
let mut seen: Vec<(&str, &str)> = Vec::new();
for (index, reading) in lec.readings.iter().enumerate() {
issues.extend(self.reading_issues(id, index, reading, &mut seen));
@@ -1770,6 +2101,7 @@ impl CourseFile {
date: None,
unit: Some("u-intro".to_string()),
slides_url: None,
teaches: Vec::new(),
readings: Vec::new(),
},
);
@@ -1825,6 +2157,7 @@ impl CourseFile {
learning_targets: targets,
references: BTreeMap::new(),
stimuli: BTreeMap::new(),
origins: BTreeMap::new(),
}
}
}
@@ -1893,7 +2226,7 @@ course:
term: Spring 2026
"#,
);
assert_eq!(c.schema_version, "1.0");
assert_eq!(c.schema_version, SCHEMA_VERSION);
assert_eq!(c.policy.options_per_item, 4);
assert_eq!(c.course.slug(), "biosc-1540");
assert!(c.validate().is_empty());
@@ -2040,6 +2373,105 @@ learning_objectives:
assert_eq!(reading.cite("kuriyan2013molecules", reference), "KKW §6.1");
}
#[test]
fn a_link_resolves_from_the_exact_location_outward() {
let mut reference = Reference {
title: "Basic local alignment search tool".into(),
kind: ReferenceKind::Article,
..Reference::default()
};
// Nothing at all to link to.
assert_eq!(reference.href(None, None), None);
// A DOI is a link to the work, which beats nothing.
reference.doi = Some("10.1016/S0022-2836(05)80360-2".into());
assert_eq!(
reference.href(None, None).as_deref(),
Some("https://doi.org/10.1016/S0022-2836(05)80360-2")
);
// The work's own URL is more use than its identifier.
reference.url = Some("https://example.org/blast".into());
assert_eq!(
reference.href(None, None).as_deref(),
Some("https://example.org/blast")
);
// A location inside the work beats the work.
reference.base_url = Some("https://example.org/blast/".into());
assert_eq!(
reference.href(None, Some("/§2")).as_deref(),
Some("https://example.org/blast/§2")
);
assert_eq!(
reference
.href(Some("https://example.org/exact"), Some("§2"))
.as_deref(),
Some("https://example.org/exact")
);
}
#[test]
fn identifiers_are_normalized_before_they_become_links() {
let doi_as_url = Reference {
title: "T".into(),
doi: Some("https://doi.org/10.1/x".into()),
..Reference::default()
};
assert_eq!(
doi_as_url.href(None, None).as_deref(),
Some("https://doi.org/10.1/x")
);
let preprint = Reference {
title: "T".into(),
kind: ReferenceKind::Preprint,
arxiv: Some("arXiv:2301.00001".into()),
..Reference::default()
};
assert_eq!(
preprint.href(None, None).as_deref(),
Some("https://arxiv.org/abs/2301.00001")
);
// A bare number is still a PMC id.
let open_access = Reference {
title: "T".into(),
pmcid: Some("3084216".into()),
..Reference::default()
};
assert_eq!(
open_access.href(None, None).as_deref(),
Some("https://www.ncbi.nlm.nih.gov/pmc/articles/PMC3084216/")
);
}
#[test]
fn something_that_is_not_a_doi_is_reported() {
let c = parse(
r#"
course: { code: X, title: Y, term: Z }
references:
bad:
title: A work
doi: nature12373
worse:
title: Another work
pmid: PMC3084216
"#,
);
let issues = c.validate();
assert!(
issues.iter().any(|i| i.contains("is not a DOI")),
"{issues:?}"
);
assert!(
issues.iter().any(|i| i.contains("is not a number")),
"{issues:?}"
);
}
#[test]
fn a_bare_string_reading_still_parses_and_round_trips() {
let c = parse(
+866
View File
@@ -0,0 +1,866 @@
// SPDX-License-Identifier: Prosperity-3.0.0
// Copyright Scientific Computing Studio
// Source: https://git.scient.ing/education/coursebank
//! One course, several files.
//!
//! A course of forty lectures does not fit in a file anyone wants to scroll. So
//! the registries [`CourseFile`] holds may be spread across a directory and
//! merged on load:
//!
//! ```text
//! course.yaml course, policy, units
//! references.yaml references
//! lectures/l-1-2.yaml one lecture, its readings, and what it teaches
//! objectives/lo-x.yaml one objective and its targets
//! ```
//!
//! The merge happens in memory on every command. Nothing is generated on disk
//! and no command depends on a build step, because a generated file that other
//! commands read is a file that goes stale. `coursebank course build` exists to
//! show you the merged result, and nothing reads what it writes.
//!
//! # What this does not relax
//!
//! Splitting a file is only worth doing if it cannot introduce a second
//! definition of the same thing. Two rules keep that true, and both are enforced
//! here rather than left to convention:
//!
//! * **One definition site per id.** Two files defining `lo-read-file-formats`
//! is an error naming both paths. The winner is not the last file loaded,
//! because there is no winner.
//! * **A section belongs to a kind of file.** A file under `lectures/` may not
//! define `learning_objectives`. Otherwise the layout decays into forty files
//! that each might hold anything, which is the same navigation problem in a
//! worse shape.
//!
//! `course.yaml` is exempt from the second rule: a course that has not been
//! split is a single fragment that happens to define everything, and it keeps
//! loading unchanged.
//!
//! # Two derivations
//!
//! Splitting by lecture makes two fields tedious to maintain by hand, so they
//! are derived instead:
//!
//! * A lecture's `teaches:` list adds that lecture to each named objective's
//! `lectures`. You write what a lecture covers while planning the lecture,
//! which is when you know.
//! * A target with no `lectures` of its own inherits its objective's, the same
//! way it already inherits `level_ceiling`.
//!
//! Both are unions and both are idempotent, so declaring a pair on both sides
//! is redundant rather than contradictory.
use std::collections::BTreeMap;
use std::fmt;
use std::path::{Path, PathBuf};
use serde::{Deserialize, Serialize};
use super::{
COURSE_FILE, Course, CourseFile, Lecture, Objective, Policy, Reference, SCHEMA_VERSION,
SUPPORTED_MAJORS, Stimulus, Target, Unit,
};
use crate::error::{Error, Result};
use crate::layout::Layout;
use crate::yaml;
/// The file holding the bibliography when it is kept out of `course.yaml`.
pub const REFERENCES_FILE: &str = "references.yaml";
/// One registry section of a course.
///
/// Used to say which file a fact came from, and to keep a fragment from
/// defining something that belongs somewhere else.
#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)]
pub enum Section {
/// Course identity.
Course,
/// Course-wide policy.
Policy,
/// Units.
Units,
/// Lectures.
Lectures,
/// Learning objectives.
Objectives,
/// Learning targets.
Targets,
/// Works the course cites.
References,
/// Shared stimuli.
Stimuli,
}
impl Section {
/// Every section, in the order a merged course lists them.
pub const ALL: [Section; 8] = [
Section::Course,
Section::Policy,
Section::Units,
Section::Lectures,
Section::Objectives,
Section::Targets,
Section::References,
Section::Stimuli,
];
/// The YAML key this section is written under.
pub fn key(self) -> &'static str {
match self {
Section::Course => "course",
Section::Policy => "policy",
Section::Units => "units",
Section::Lectures => "lectures",
Section::Objectives => "learning_objectives",
Section::Targets => "learning_targets",
Section::References => "references",
Section::Stimuli => "stimuli",
}
}
/// What one of its entries is called in a message.
pub fn noun(self) -> &'static str {
match self {
Section::Course => "course identity",
Section::Policy => "policy",
Section::Units => "unit",
Section::Lectures => "lecture",
Section::Objectives => "objective",
Section::Targets => "target",
Section::References => "reference",
Section::Stimuli => "stimulus",
}
}
/// Where a file defining this section is expected to live.
pub fn home(self) -> &'static str {
match self {
Section::Course | Section::Policy | Section::Units => COURSE_FILE,
Section::Lectures => "lectures/*.yaml",
Section::Objectives | Section::Targets | Section::Stimuli => "objectives/*.yaml",
Section::References => REFERENCES_FILE,
}
}
}
impl fmt::Display for Section {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
f.write_str(self.key())
}
}
/// What kind of file a fragment is, which fixes what it may define.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Role {
/// `course.yaml`. May define anything, so an unsplit course still loads.
Root,
/// `references.yaml`.
References,
/// A file under `lectures/`.
Lecture,
/// A file under `objectives/`.
Objective,
}
impl Role {
/// Whether a file in this role may define a section.
pub fn allows(self, section: Section) -> bool {
match self {
Role::Root => true,
Role::References => section == Section::References,
Role::Lecture => section == Section::Lectures,
Role::Objective => matches!(
section,
Section::Objectives | Section::Targets | Section::Stimuli
),
}
}
/// A short name for a message.
pub fn label(self) -> &'static str {
match self {
Role::Root => "course",
Role::References => "references",
Role::Lecture => "lecture",
Role::Objective => "objective",
}
}
}
/// One file's worth of course registries.
///
/// Every section is optional, which is what makes this both the fragment schema
/// and — with every section filled in — the schema of an unsplit `course.yaml`.
/// [`CourseFile`] is the resolved model the rest of the crate reads; this is
/// only what one file on disk is allowed to say.
#[derive(Debug, Clone, Default, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct Fragment {
/// Schema version this file targets.
#[serde(
default,
deserialize_with = "yaml::flexible_string_opt",
skip_serializing_if = "Option::is_none"
)]
pub schema_version: Option<String>,
/// Course identity. Exactly one fragment must carry it.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub course: Option<Course>,
/// Course-wide policy.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub policy: Option<Policy>,
/// Units, in teaching order.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub units: Vec<Unit>,
/// Lectures by id.
#[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
pub lectures: BTreeMap<String, Lecture>,
/// Learning objectives by id.
#[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
pub learning_objectives: BTreeMap<String, Objective>,
/// Learning targets by id.
#[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
pub learning_targets: BTreeMap<String, Target>,
/// Works the course cites, by citation key.
#[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
pub references: BTreeMap<String, Reference>,
/// Shared stimuli by id.
#[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
pub stimuli: BTreeMap<String, Stimulus>,
}
impl Fragment {
/// Loads one fragment from disk.
///
/// # Arguments
///
/// * `path` - the file to read.
///
/// # Returns
///
/// The parsed fragment.
///
/// # Errors
///
/// Returns [`Error::Io`] if unreadable and [`Error::Yaml`] if it does not
/// match the schema. Unknown keys are errors, so a misspelled section name
/// is caught here rather than silently contributing nothing.
pub fn load(path: &Path) -> Result<Fragment> {
yaml::read(path)
}
/// Which sections this fragment actually defines.
pub fn sections(&self) -> Vec<Section> {
let mut out = Vec::new();
if self.course.is_some() {
out.push(Section::Course);
}
if self.policy.is_some() {
out.push(Section::Policy);
}
if !self.units.is_empty() {
out.push(Section::Units);
}
if !self.lectures.is_empty() {
out.push(Section::Lectures);
}
if !self.learning_objectives.is_empty() {
out.push(Section::Objectives);
}
if !self.learning_targets.is_empty() {
out.push(Section::Targets);
}
if !self.references.is_empty() {
out.push(Section::References);
}
if !self.stimuli.is_empty() {
out.push(Section::Stimuli);
}
out
}
}
/// The fragment files of a course directory, in load order, with their roles.
///
/// `course.yaml` is listed whether or not it exists, so a directory that is not
/// a course fails with a message naming the file it wanted rather than an empty
/// merge. Directory contents are sorted, which is what makes the merged course
/// independent of filesystem order.
///
/// # Arguments
///
/// * `layout` - the resolved course layout.
///
/// # Returns
///
/// Paths paired with what each file is allowed to define.
///
/// # Errors
///
/// Returns [`Error::Io`] when a fragment directory exists but cannot be read.
pub fn files(layout: &Layout) -> Result<Vec<(PathBuf, Role)>> {
let mut out = vec![(layout.course_file(), Role::Root)];
let references = layout.references_file();
if references.is_file() {
out.push((references, Role::References));
}
for path in yaml::list_yaml(&layout.lectures())? {
out.push((path, Role::Lecture));
}
for path in yaml::list_yaml(&layout.objectives())? {
out.push((path, Role::Objective));
}
Ok(out)
}
/// Loads every fragment in a course directory and merges them into one course.
///
/// # Arguments
///
/// * `root` - the course directory.
///
/// # Returns
///
/// The merged course, with [`CourseFile::origins`] recording which file defined
/// each id.
///
/// # Errors
///
/// Propagates load errors, and returns [`Error::Invalid`] with every merge
/// problem at once: an id defined twice, a section in the wrong kind of file, a
/// fragment written against another major schema version, or no `course:`
/// section anywhere.
///
/// Cross-references are *not* checked here. A dangling objective id is a
/// content problem, and content problems are [`CourseFile::validate`]'s, so that
/// they are reported the same way whether or not the course is split.
pub fn assemble(root: &Path) -> Result<CourseFile> {
let layout = Layout::new(root);
let mut merge = Merge::default();
for (path, role) in files(&layout)? {
let fragment = Fragment::load(&path)?;
let shown = path.strip_prefix(root).unwrap_or(&path).to_path_buf();
merge.take(&shown, role, fragment);
}
merge.resolve();
merge.finish(root)
}
/// Accumulates fragments, remembering where each id came from.
#[derive(Debug, Default)]
struct Merge {
schema_version: Option<String>,
course: Option<Course>,
policy: Option<Policy>,
units: Vec<Unit>,
lectures: BTreeMap<String, Lecture>,
objectives: BTreeMap<String, Objective>,
targets: BTreeMap<String, Target>,
references: BTreeMap<String, Reference>,
stimuli: BTreeMap<String, Stimulus>,
origins: BTreeMap<(Section, String), PathBuf>,
issues: Vec<String>,
}
impl Merge {
/// Folds one fragment in.
///
/// # Arguments
///
/// * `path` - the fragment's path relative to the course root, for messages.
/// * `role` - what this file is allowed to define.
/// * `fragment` - the parsed fragment.
fn take(&mut self, path: &Path, role: Role, fragment: Fragment) {
for section in fragment.sections() {
if !role.allows(section) {
self.issues.push(format!(
"{}: a {} file may not define `{}`; that section belongs in {}",
path.display(),
role.label(),
section.key(),
section.home()
));
}
}
if let Some(declared) = &fragment.schema_version {
if !SUPPORTED_MAJORS.contains(&major(declared)) {
self.issues.push(format!(
"{}: declares schema_version {declared}, which this build cannot read. It \
writes {SCHEMA_VERSION} and reads {}.",
path.display(),
SUPPORTED_MAJORS
.iter()
.map(|m| format!("{m}.x"))
.collect::<Vec<_>>()
.join(" and ")
));
}
if self.schema_version.is_none() {
self.schema_version = Some(declared.clone());
}
}
if role.allows(Section::Course) {
if let Some(course) = fragment.course {
if self.claim(Section::Course, path) {
self.course = Some(course);
}
}
}
if role.allows(Section::Policy) {
if let Some(policy) = fragment.policy {
if self.claim(Section::Policy, path) {
self.policy = Some(policy);
}
}
}
if role.allows(Section::Units) {
for unit in fragment.units {
let key = (Section::Units, unit.id.clone());
if let Some(first) = self.origins.get(&key) {
let message = duplicate(Section::Units, &unit.id, first.as_path(), path);
self.issues.push(message);
continue;
}
self.origins.insert(key, path.to_path_buf());
self.units.push(unit);
}
}
if role.allows(Section::Lectures) {
absorb(
&mut self.lectures,
fragment.lectures,
Section::Lectures,
path,
&mut self.origins,
&mut self.issues,
);
}
if role.allows(Section::Objectives) {
absorb(
&mut self.objectives,
fragment.learning_objectives,
Section::Objectives,
path,
&mut self.origins,
&mut self.issues,
);
}
if role.allows(Section::Targets) {
absorb(
&mut self.targets,
fragment.learning_targets,
Section::Targets,
path,
&mut self.origins,
&mut self.issues,
);
}
if role.allows(Section::References) {
absorb(
&mut self.references,
fragment.references,
Section::References,
path,
&mut self.origins,
&mut self.issues,
);
}
if role.allows(Section::Stimuli) {
absorb(
&mut self.stimuli,
fragment.stimuli,
Section::Stimuli,
path,
&mut self.origins,
&mut self.issues,
);
}
}
/// Records a section that may only be declared once.
///
/// # Arguments
///
/// * `section` - the section being claimed.
/// * `path` - the file claiming it.
///
/// # Returns
///
/// Whether the claim was the first, and so whether the caller should store
/// what it parsed.
fn claim(&mut self, section: Section, path: &Path) -> bool {
let key = (section, String::new());
if let Some(first) = self.origins.get(&key) {
let message = format!(
"`{}` is declared twice: {} and {}. It applies to the whole course, so it has \
one definition site.",
section.key(),
first.display(),
path.display()
);
self.issues.push(message);
return false;
}
self.origins.insert(key, path.to_path_buf());
true
}
/// Fills in the two fields a split layout would otherwise duplicate.
fn resolve(&mut self) {
// A lecture says what it teaches; the objective's lecture list follows.
for (lecture_id, lecture) in &self.lectures {
for objective_id in &lecture.teaches {
if let Some(objective) = self.objectives.get_mut(objective_id) {
if !objective.lectures.iter().any(|l| l == lecture_id) {
objective.lectures.push(lecture_id.clone());
}
}
}
}
// A target with no lecture of its own is taught wherever its objective
// is. Collected first: the read of `objectives` and the write to
// `targets` cannot overlap in one pass.
let inherited: Vec<(String, Vec<String>)> = self
.targets
.iter()
.filter(|(_, target)| target.lectures.is_empty())
.filter_map(|(id, target)| {
self.objectives
.get(&target.objective)
.map(|objective| (id.clone(), objective.lectures.clone()))
})
.collect();
for (id, lectures) in inherited {
if let Some(target) = self.targets.get_mut(&id) {
target.lectures = lectures;
}
}
}
/// Builds the course, or reports every merge problem at once.
fn finish(self, root: &Path) -> Result<CourseFile> {
let mut issues = self.issues;
let course = match self.course {
Some(course) => course,
None => {
issues.push(format!(
"no file in {} declares a `course:` section, so the course has no code, \
title, or term",
root.display()
));
return Err(Error::Invalid(issues));
}
};
if !issues.is_empty() {
return Err(Error::Invalid(issues));
}
Ok(CourseFile {
schema_version: self
.schema_version
.unwrap_or_else(|| SCHEMA_VERSION.to_string()),
course,
policy: self.policy.unwrap_or_default(),
units: self.units,
lectures: self.lectures,
learning_objectives: self.objectives,
learning_targets: self.targets,
references: self.references,
stimuli: self.stimuli,
origins: self.origins,
})
}
}
/// Moves one section's entries across, refusing a second definition.
fn absorb<T>(
into: &mut BTreeMap<String, T>,
from: BTreeMap<String, T>,
section: Section,
path: &Path,
origins: &mut BTreeMap<(Section, String), PathBuf>,
issues: &mut Vec<String>,
) {
for (id, value) in from {
let key = (section, id.clone());
if let Some(first) = origins.get(&key) {
issues.push(duplicate(section, &id, first.as_path(), path));
continue;
}
origins.insert(key, path.to_path_buf());
into.insert(id, value);
}
}
/// The message for an id defined in two files.
fn duplicate(section: Section, id: &str, first: &Path, second: &Path) -> String {
format!(
"{} `{id}` is defined in two places: {} and {}. An id has one definition site; delete \
one or rename it.",
section.noun(),
first.display(),
second.display()
)
}
/// The part of a schema version before the first dot.
fn major(version: &str) -> &str {
version.split('.').next().unwrap_or(version)
}
#[cfg(test)]
mod tests {
use super::*;
fn tmp(tag: &str) -> PathBuf {
let p = std::env::temp_dir().join(format!("coursebank-frag-{tag}-{}", std::process::id()));
let _ = std::fs::remove_dir_all(&p);
std::fs::create_dir_all(&p).unwrap();
p
}
fn write(root: &Path, relative: &str, body: &str) {
let path = root.join(relative);
std::fs::create_dir_all(path.parent().unwrap()).unwrap();
std::fs::write(path, body).unwrap();
}
const ROOT: &str = r#"
course:
code: BIOSC 1540
title: Computational Biology
term: 2026f
policy:
points_per_item: 1.0
units:
- id: u1
title: Search and Similarity
"#;
#[test]
fn a_split_course_merges_into_one_model() {
let root = tmp("merge");
write(&root, "course.yaml", ROOT);
write(
&root,
"references.yaml",
"references:\n ismail2023:\n title: Bioinformatics\n",
);
write(
&root,
"lectures/l-1-2.yaml",
"lectures:\n L1.2:\n title: The Digital Genome\n unit: u1\n \
teaches: [lo-read-file-formats]\n",
);
write(
&root,
"objectives/lo-read-file-formats.yaml",
"learning_objectives:\n lo-read-file-formats:\n text: Read the text formats.\n \
unit: u1\nlearning_targets:\n t-fastq-structure:\n text: Identify the four \
lines.\n objective: lo-read-file-formats\n",
);
let course = assemble(&root).unwrap();
assert_eq!(course.course.code, "BIOSC 1540");
assert_eq!(course.units.len(), 1);
assert_eq!(course.references.len(), 1);
assert!(course.validate().is_empty(), "{:?}", course.validate());
// Derived: the lecture registered the objective, and the target
// inherited the objective's lecture.
assert_eq!(
course.learning_objectives["lo-read-file-formats"].lectures,
vec!["L1.2".to_string()]
);
assert_eq!(
course.learning_targets["t-fastq-structure"].lectures,
vec!["L1.2".to_string()]
);
assert_eq!(course.lecture_targets("L1.2"), vec!["t-fastq-structure"]);
}
#[test]
fn an_unsplit_course_file_still_loads() {
let root = tmp("monolith");
write(
&root,
"course.yaml",
&format!(
"{ROOT}lectures:\n L1.2:\n title: The Digital Genome\nlearning_objectives:\n \
lo-x:\n text: Do the thing.\n lectures: [L1.2]\nlearning_targets:\n \
t-x:\n text: Do the smaller thing.\n objective: lo-x\nreferences:\n \
ismail2023:\n title: Bioinformatics\n"
),
);
let course = assemble(&root).unwrap();
assert!(course.validate().is_empty(), "{:?}", course.validate());
assert_eq!(course.lecture_objectives("L1.2"), vec!["lo-x"]);
assert_eq!(
course.origin("lo-x").map(|(_, p)| p.to_path_buf()),
Some(PathBuf::from(COURSE_FILE))
);
}
#[test]
fn an_id_defined_twice_names_both_files() {
let root = tmp("dup");
write(&root, "course.yaml", ROOT);
let body = "learning_objectives:\n lo-x:\n text: Do the thing.\n";
write(&root, "objectives/lo-x.yaml", body);
write(&root, "objectives/lo-x-old.yaml", body);
let err = assemble(&root).unwrap_err();
let message = err.to_string();
assert!(message.contains("objectives/lo-x.yaml"), "{message}");
assert!(message.contains("objectives/lo-x-old.yaml"), "{message}");
assert!(message.contains("one definition site"), "{message}");
}
#[test]
fn a_section_in_the_wrong_kind_of_file_is_rejected() {
let root = tmp("misplaced");
write(&root, "course.yaml", ROOT);
write(
&root,
"lectures/l-1-2.yaml",
"lectures:\n L1.2:\n title: The Digital Genome\nlearning_objectives:\n lo-x:\n \
text: Do the thing.\n",
);
let message = assemble(&root).unwrap_err().to_string();
assert!(
message.contains("may not define `learning_objectives`"),
"{message}"
);
assert!(message.contains("objectives/*.yaml"), "{message}");
}
#[test]
fn a_course_with_no_identity_says_so() {
let root = tmp("no-course");
write(&root, "course.yaml", "units:\n - id: u1\n title: One\n");
let message = assemble(&root).unwrap_err().to_string();
assert!(message.contains("`course:` section"), "{message}");
}
#[test]
fn a_fragment_from_another_major_version_is_refused() {
let root = tmp("version");
write(&root, "course.yaml", ROOT);
write(
&root,
"objectives/lo-x.yaml",
"schema_version: '9.0'\nlearning_objectives:\n lo-x:\n text: Do the thing.\n",
);
let message = assemble(&root).unwrap_err().to_string();
assert!(message.contains("schema_version 9.0"), "{message}");
}
#[test]
fn origins_point_at_the_fragment_that_defined_each_id() {
let root = tmp("origins");
write(&root, "course.yaml", ROOT);
write(
&root,
"lectures/l-1-2.yaml",
"lectures:\n L1.2:\n title: The Digital Genome\n",
);
write(
&root,
"objectives/lo-x.yaml",
"learning_objectives:\n lo-x:\n text: Do the thing.\n",
);
let course = assemble(&root).unwrap();
let (section, path) = course.origin("lo-x").unwrap();
assert_eq!(section, Section::Objectives);
assert_eq!(path, Path::new("objectives/lo-x.yaml"));
let (section, path) = course.origin("L1.2").unwrap();
assert_eq!(section, Section::Lectures);
assert_eq!(path, Path::new("lectures/l-1-2.yaml"));
assert!(course.origin("nothing-like-this").is_none());
}
#[test]
fn teaching_the_same_objective_from_two_lectures_unions() {
let root = tmp("union");
write(&root, "course.yaml", ROOT);
write(
&root,
"lectures/l-1-2.yaml",
"lectures:\n L1.2:\n title: One\n teaches: [lo-x]\n",
);
write(
&root,
"lectures/l-1-3.yaml",
"lectures:\n L1.3:\n title: Two\n teaches: [lo-x]\n",
);
write(
&root,
"objectives/lo-x.yaml",
"learning_objectives:\n lo-x:\n text: Do the thing.\n",
);
let course = assemble(&root).unwrap();
assert_eq!(
course.learning_objectives["lo-x"].lectures,
vec!["L1.2".to_string(), "L1.3".to_string()]
);
}
#[test]
fn a_declaration_on_both_sides_is_not_duplicated() {
let root = tmp("both-sides");
write(&root, "course.yaml", ROOT);
write(
&root,
"lectures/l-1-2.yaml",
"lectures:\n L1.2:\n title: One\n teaches: [lo-x]\n",
);
write(
&root,
"objectives/lo-x.yaml",
"learning_objectives:\n lo-x:\n text: Do the thing.\n lectures: [L1.2]\n",
);
let course = assemble(&root).unwrap();
assert_eq!(
course.learning_objectives["lo-x"].lectures,
vec!["L1.2".to_string()]
);
}
#[test]
fn teaching_an_unknown_objective_is_a_validation_problem_not_a_merge_one() {
let root = tmp("unknown-teaches");
write(&root, "course.yaml", ROOT);
write(
&root,
"lectures/l-1-2.yaml",
"lectures:\n L1.2:\n title: One\n teaches: [lo-nope]\n",
);
let course = assemble(&root).unwrap();
let issues = course.validate();
assert!(issues.iter().any(|i| i.contains("lo-nope")), "{issues:?}");
}
}
+470 -18
View File
@@ -25,6 +25,7 @@ use serde::de::{self, MapAccess, Visitor};
use serde::ser::SerializeMap;
use serde::{Deserialize, Deserializer, Serialize, Serializer};
use crate::course::Reference;
use crate::date::Date;
use crate::hash::fingerprint;
use crate::taxonomy::{
@@ -44,9 +45,28 @@ pub struct Item {
/// Revision counter, bumped whenever the content changes in a way that
/// invalidates pooled statistics.
#[serde(default = "one_u32")]
/// Retained only so a pre-2.0 bank still loads. Ignored.
///
/// A version number on a question answered the wrong question. It recorded
/// that *something* changed without constraining what, which meant an item
/// at version 3 might have a reworded distractor — fair, the statistics
/// still describe the same question — or a reworded stem, which makes it a
/// different question wearing the same id. Since 2.0 the stem *is* the
/// identity: reword it and you have a new item, with a new id and
/// [`Item::supersedes`] pointing back. [`Item::stem_digest`] is what
/// enforces that, against the seals of every administration.
///
/// `coursebank migrate stems` removes it.
#[serde(default, skip_serializing)]
pub version: u32,
/// The item this one replaces, when it is a rewording of an earlier stem.
///
/// Lineage rather than versioning: both items stay in the bank, each with
/// its own statistics, and a report can say which one a cohort answered.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub supersedes: Option<String>,
/// Workflow state; only [`Status::Approved`] items may be assembled.
pub status: Status,
@@ -145,7 +165,16 @@ pub struct Item {
pub review: Option<Review>,
/// Append-only change log.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
/// Retained only so a pre-2.0 bank still loads. Ignored.
///
/// A hand-maintained change log inside a version-controlled file, every
/// entry of which duplicated what `git log -p` already knew, with no
/// guarantee of agreeing with it. What git cannot express is a claim about
/// the item rather than a record of an edit, and that has its own fields:
/// [`Item::retired`] and [`Item::supersedes`].
///
/// `coursebank migrate stems` removes it.
#[serde(default, skip_serializing)]
pub history: Vec<HistoryEntry>,
/// The author of record.
@@ -170,7 +199,17 @@ pub struct Item {
#[derive(Debug, Clone, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct Choice {
/// Option letter, `A` through `H`.
/// The option's id, unique within its item: `o-fourth-line`.
///
/// A name rather than a position. Until 2.0 this was a letter, which put a
/// position in a field that pooled statistics, student feedback, and
/// `credit_overrides` all join on — so reordering a YAML block silently
/// moved the misconception recorded against one option onto another. The
/// letter a student sees is derived per form from the form's seed and lives
/// in the seal; see [`crate::seal::printed_letter`].
///
/// A single letter `A` through `H` still loads, so a bank migrates when you
/// run `coursebank migrate options` rather than when you upgrade.
pub id: String,
/// The option text.
@@ -218,6 +257,17 @@ pub struct Choice {
/// Your a priori guess at how often this option is chosen.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub selection_rate_expected: Option<f64>,
/// Why this option is no longer drawn, when it is not.
///
/// A retired option stays in the file forever. It has to: a seal and four
/// terms of response rows refer to it by id, and deleting it would turn
/// every one of those references into a dangling one. What retirement does
/// is take it out of the pool an assessment draws from, with the reason
/// attached — "selected by 1 of 96 across two administrations" is a finding
/// about the option, and the place for it is next to the option.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub retired: Option<Retirement>,
}
impl Choice {
@@ -319,6 +369,19 @@ impl Citation {
(None, None) => String::new(),
}
}
/// The link for this location, resolved against the work it points into.
///
/// # Arguments
///
/// * `reference` - the work, looked up from the citation key.
///
/// # Returns
///
/// The most specific link available. See [`Reference::href`].
pub fn href(&self, reference: &Reference) -> Option<String> {
reference.href(self.url.as_deref(), self.path.as_deref())
}
}
/// Writes a citation as a mapping, or as a bare string when that is all it holds.
@@ -553,6 +616,89 @@ pub struct Calibration {
/// Machine-detected problems.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub flags: Vec<Flag>,
/// One record per option set ever administered.
///
/// What the flat fields above cannot express once options are a pool. A
/// stem shown with distractors `{third, second, first}` is a measurably
/// easier item than the same stem with `{third, plus-line, line-two}`, so a
/// p-value pooled across both is the average of two different questions.
/// Statistics are computed and compared per variant; the flat fields remain
/// as the pre-2.0 summary, and for an item whose pool is its form the two
/// agree.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub variants: Vec<VariantCalibration>,
/// One record per option, pooled across every set it appeared in.
///
/// The capability the pool is worth the trouble for. Selection rates are
/// shares of a fixed set, so they are only comparable *within* a variant —
/// which means this view supports exactly one kind of claim, and it is the
/// useful one: this option draws nobody, anywhere. That is the evidence
/// that retires a distractor, and one administration cannot supply it.
#[serde(default, skip_serializing_if = "std::collections::BTreeMap::is_empty")]
pub options: std::collections::BTreeMap<String, OptionHistory>,
}
/// Statistics for one option set, as administered.
#[derive(Debug, Clone, Default, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct VariantCalibration {
/// The digest this record describes. See [`Item::variant_digest`].
pub variant: String,
/// The option ids keyed correct.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub key: Vec<String>,
/// The option ids offered alongside them.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub distractors: Vec<String>,
/// The administrations pooled into these numbers.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub administrations: Vec<String>,
/// Examinees pooled.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub n_examinees: Option<usize>,
/// Proportion correct.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub p_value: Option<f64>,
/// Corrected item-total point-biserial correlation.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub point_biserial: Option<f64>,
/// Upper-minus-lower-group discrimination index.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub discrimination_index: Option<f64>,
/// Per-option behaviour within this set, keyed by option id.
#[serde(default, skip_serializing_if = "std::collections::BTreeMap::is_empty")]
pub option_stats: std::collections::BTreeMap<String, OptionStat>,
/// Fitted item response theory parameters for this set.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub irt: Option<IrtParams>,
/// Machine-detected problems with this set.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub flags: Vec<Flag>,
}
/// What one option has done across every set it has appeared in.
///
/// Deliberately coarse. Averaging selection rates across variants is not
/// meaningful — each is a share of a different set — so `mean_selection_rate`
/// is a summary for reading, not a statistic to act on. `never_chosen` is the
/// one field that carries weight, and it needs several administrations to earn.
#[derive(Debug, Clone, Default, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct OptionHistory {
/// How many distinct variants this option has appeared in.
#[serde(default)]
pub appearances: usize,
/// Examinees who saw it, summed across those variants.
#[serde(default)]
pub n_examinees: usize,
/// Mean of its within-variant selection rates. For reading only.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub mean_selection_rate: Option<f64>,
/// Whether it has never been chosen, anywhere.
#[serde(default, skip_serializing_if = "is_false")]
pub never_chosen: bool,
}
/// How one option behaved.
@@ -661,7 +807,7 @@ pub struct Retirement {
#[derive(Debug, Clone, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct HistoryEntry {
/// The version this change produced.
/// The version this change produced. Ignored since 2.0.
pub version: u32,
/// When it was made.
pub date: Date,
@@ -672,6 +818,32 @@ pub struct HistoryEntry {
pub change: String,
}
/// The separator a pre-2.0 bank-qualified item id used: `b-1-2::q-fastq-line`.
pub const LEGACY_QUALIFIER: &str = "::";
/// An item id with any pre-2.0 bank qualifier removed.
///
/// Until 2.0 an item was named `bank::item`, which made the file it happened to
/// live in part of its identity — and therefore part of the join key on every
/// row of response data ever collected. Moving a question between banks renamed
/// it. Since 2.0 the id names the item course-wide and the bank is only where it
/// is kept, so anything reading an old id strips the qualifier rather than
/// failing to match.
///
/// # Arguments
///
/// * `id` - an item id in either form.
///
/// # Returns
///
/// The part after the qualifier, or the whole id when there is none.
pub fn canonical_id(id: &str) -> &str {
match id.split_once(LEGACY_QUALIFIER) {
Some((_, rest)) => rest,
None => id,
}
}
impl Item {
/// Builds a draft item with everything optional left empty.
///
@@ -725,6 +897,7 @@ impl Item {
author: None,
notes_private: None,
retired: None,
supersedes: None,
}
}
@@ -758,19 +931,38 @@ impl Item {
out
}
/// Looks up an option by letter.
/// Looks up an option by id.
///
/// # Arguments
///
/// * `letter` - the option id, case insensitive.
/// * `id` - the option id. A pre-2.0 letter matches case-insensitively,
/// which a slug never needs but a hand-typed `d` does.
///
/// # Returns
///
/// The option, or `None`.
pub fn option(&self, letter: &str) -> Option<&Choice> {
pub fn option(&self, id: &str) -> Option<&Choice> {
self.options
.iter()
.find(|o| o.id.eq_ignore_ascii_case(letter))
.find(|o| o.id == id)
.or_else(|| self.options.iter().find(|o| o.id.eq_ignore_ascii_case(id)))
}
/// Whether an option id is a pre-2.0 letter rather than a name.
///
/// # Arguments
///
/// * `id` - the option id.
///
/// # Returns
///
/// `true` for `A` through `H`.
pub fn is_legacy_option_id(id: &str) -> bool {
id.len() == 1
&& id
.chars()
.next()
.is_some_and(|c| c.is_ascii_uppercase() && c <= 'H')
}
/// Whether the item keys more than one option.
@@ -834,6 +1026,159 @@ impl Item {
fingerprint(parts.iter().map(|s| s.as_str()))
}
/// The options an assessment administers, in the order the bank declares
/// them.
///
/// Since 2.0 `options` is a *pool*: it may hold several defensible keys and
/// more distractors than any one form shows, and which of them a student
/// saw is a property of the placement rather than of the item. Everything
/// that renders, seals, decodes, or scores an administration has to work
/// from this rather than from `options`, or the paper and the key disagree.
///
/// Bank order, not administered order: the per-form permutation is
/// [`crate::select::option_order`]'s business, and keeping the two separate
/// is what lets one item appear on three forms with one set of statistics.
///
/// # Arguments
///
/// * `key` - the option ids keyed correct for this administration.
/// * `distractors` - the option ids offered alongside them.
///
/// # Returns
///
/// The named options, or the whole live pool when `distractors` is empty.
///
/// `distractors` is what says the set was chosen, not `key`. A pre-2.0
/// record names its key and nothing else — `key: [D]` with no distractor
/// list — and it means "all of them, and D is the right one". Reading that
/// as "administer D alone" would print a one-option paper for every
/// assessment ever recorded.
pub fn administered(&self, key: &[String], distractors: &[String]) -> Vec<&Choice> {
if distractors.is_empty() {
return self
.options
.iter()
.filter(|o| o.retired.is_none())
.collect();
}
self.options
.iter()
.filter(|o| key.contains(&o.id) || distractors.contains(&o.id))
.collect()
}
/// The options that may still be drawn.
///
/// # Returns
///
/// Every option not retired, split into candidate keys and distractors.
pub fn pool(&self) -> (Vec<&Choice>, Vec<&Choice>) {
let live = || self.options.iter().filter(|o| o.retired.is_none());
(
live().filter(|o| o.correct).collect(),
live().filter(|o| !o.correct).collect(),
)
}
/// A digest of the item as one administration showed it.
///
/// The pooling key for statistics, and the reason
/// [`Item::fingerprint`] cannot be. A stem with distractors
/// `{third, second, first}` is a measurably easier item than the same stem
/// with `{third, plus-line, line-two}`, so pooling a p-value across both is
/// averaging two different questions. Covers the stem, the administered
/// options, and which of them was keyed — the last because the same option
/// set with a different key is again a different item.
///
/// # Arguments
///
/// * `key` - the option ids keyed correct for this administration.
/// * `distractors` - the option ids offered alongside them.
///
/// # Returns
///
/// The digest as hex.
pub fn variant_digest(&self, key: &[String], distractors: &[String]) -> String {
let mut parts = vec![self.stem_digest()];
let mut shown: Vec<&Choice> = self.administered(key, distractors);
shown.sort_by(|a, b| a.id.cmp(&b.id));
for option in shown {
let keyed = if key.is_empty() {
option.correct
} else {
key.contains(&option.id)
};
parts.push(format!(
"{}|{}|{}",
option.id,
if keyed { "1" } else { "0" },
option.text.trim()
));
}
fingerprint(parts.iter().map(|s| s.as_str()))
}
/// A digest of what the item asks, without its options.
///
/// The identity check. [`Item::fingerprint`] covers the options too, which
/// is right for calibration — reword a distractor and the pooled selection
/// rates no longer describe what students saw — but wrong for identity,
/// because a question whose distractors changed is still the same question.
/// This covers the stem and the stimulus, and nothing else.
///
/// Compared against the digest each seal recorded, which is what makes
/// "a reworded stem is a new stem" a rule the tool enforces rather than a
/// convention that decays.
///
/// # Returns
///
/// The digest as hex.
pub fn stem_digest(&self) -> String {
let mut parts: Vec<String> = vec![self.stem.trim().to_string()];
if let Some(s) = &self.stimulus {
parts.push(format!("stimulus:{s}"));
}
fingerprint(parts.iter().map(|s| s.as_str()))
}
/// The calibration recorded for one option set.
///
/// # Arguments
///
/// * `variant` - the digest from [`Item::variant_digest`].
///
/// # Returns
///
/// The record, or `None` when this set has not been calibrated.
pub fn calibration_for(&self, variant: &str) -> Option<&VariantCalibration> {
self.calibration
.as_ref()?
.variants
.iter()
.find(|v| v.variant == variant)
}
/// Whether a variant's recorded statistics still describe it.
///
/// Staleness gets *narrower* with a pool rather than wider: rewording one
/// distractor used to invalidate the item's whole calibration, and now it
/// invalidates only the sets that distractor appeared in.
///
/// # Arguments
///
/// * `variant` - the digest to check.
///
/// # Returns
///
/// `false` only when a record exists for that digest and the digest no
/// longer matches what the option ids now say.
pub fn variant_is_current(&self, variant: &str) -> bool {
match self.calibration_for(variant) {
Some(record) => variant == self.variant_digest(&record.key, &record.distractors),
None => true,
}
}
/// Whether the recorded calibration matches the current content.
///
/// # Returns
@@ -889,16 +1234,20 @@ impl Item {
}
}
/// Appends a change-log entry and bumps the version.
/// Appends a change-log entry.
///
/// Kept for the pre-2.0 banks that still carry a `history:` block, so
/// reading one and writing it back does not silently drop entries. New
/// entries belong in a commit message.
///
/// # Arguments
///
/// * `change` - a description of what changed.
/// * `author` - who made the change.
pub fn record_change(&mut self, change: &str, author: Option<&str>) {
self.version += 1;
let version = self.history.iter().map(|h| h.version).max().unwrap_or(0) + 1;
self.history.push(HistoryEntry {
version: self.version,
version,
date: Date::today(),
author: author.map(|a| a.to_string()),
change: change.to_string(),
@@ -906,9 +1255,6 @@ impl Item {
}
}
fn one_u32() -> u32 {
1
}
fn default_format() -> Format {
Format::SingleBestAnswer
}
@@ -938,7 +1284,6 @@ options:
#[test]
fn minimal_item_parses_with_defaults() {
let it = item(MINIMAL);
assert_eq!(it.version, 1);
assert_eq!(it.format, Format::SingleBestAnswer);
assert!(!it.bonus);
assert_eq!(it.key_letters(), vec!["A"]);
@@ -1010,6 +1355,101 @@ options:
assert_eq!(a.fingerprint(), b.fingerprint());
}
#[test]
fn a_pool_administers_a_subset_and_defaults_to_everything() {
let mut it = item(MINIMAL);
let all: Vec<String> = it.options.iter().map(|o| o.id.clone()).collect();
// Unstated means the whole pool, which is what a pre-2.0 record meant.
assert_eq!(it.administered(&[], &[]).len(), all.len());
// A retired option leaves the pool but not the file.
it.options[1].retired = Some(Retirement {
on: Date::new(2026, 9, 20).unwrap(),
reason: "chosen by 1 of 96 across two administrations".into(),
replaced_by: None,
});
let shown = it.administered(&[], &[]);
assert_eq!(shown.len(), all.len() - 1);
assert!(!shown.iter().any(|o| o.id == all[1]));
// Still resolvable: a seal and four terms of rows refer to it.
assert!(it.option(&all[1]).is_some());
// A key with no distractor list is a pre-2.0 record, and it means all
// of them. Reading it as "administer the key alone" would print a
// one-option paper for every assessment already recorded.
assert_eq!(it.administered(&[all[2].clone()], &[]).len(), all.len() - 1);
// Named explicitly, bank order is kept whatever order the lists are in.
let shown = it.administered(&[all[2].clone()], &[all[0].clone()]);
assert_eq!(
shown.iter().map(|o| o.id.clone()).collect::<Vec<_>>(),
vec![all[0].clone(), all[2].clone()]
);
}
#[test]
fn the_variant_digest_tracks_the_option_set_and_the_stem_does_not() {
let it = item(MINIMAL);
let ids: Vec<String> = it.options.iter().map(|o| o.id.clone()).collect();
let one = it.variant_digest(&[ids[0].clone()], &[ids[1].clone()]);
let two = it.variant_digest(&[ids[0].clone()], &[ids[2].clone()]);
// A different distractor is a different item: same stem, different
// difficulty, so pooling a p-value across both would average two
// questions.
assert_ne!(one, two, "a swapped distractor is a new variant");
// The stem is unmoved by any of it.
assert_eq!(it.stem_digest(), item(MINIMAL).stem_digest());
// Order of the lists is not part of the identity.
assert_eq!(
it.variant_digest(&[ids[0].clone()], &[ids[2].clone(), ids[1].clone()]),
it.variant_digest(&[ids[0].clone()], &[ids[1].clone(), ids[2].clone()])
);
}
#[test]
fn a_variant_goes_stale_alone_rather_than_taking_the_item_with_it() {
let mut it = item(MINIMAL);
let ids: Vec<String> = it.options.iter().map(|o| o.id.clone()).collect();
let one = it.variant_digest(&[ids[0].clone()], &[ids[1].clone()]);
let two = it.variant_digest(&[ids[0].clone()], &[ids[2].clone()]);
it.calibration = Some(Calibration {
variants: vec![
VariantCalibration {
variant: one.clone(),
key: vec![ids[0].clone()],
distractors: vec![ids[1].clone()],
n_examinees: Some(96),
p_value: Some(0.84),
..VariantCalibration::default()
},
VariantCalibration {
variant: two.clone(),
key: vec![ids[0].clone()],
distractors: vec![ids[2].clone()],
n_examinees: Some(32),
..VariantCalibration::default()
},
],
..Calibration::default()
});
assert_eq!(it.calibration_for(&one).unwrap().n_examinees, Some(96));
assert!(it.calibration_for("nothing-like-this").is_none());
assert!(it.variant_is_current(&one));
assert!(it.variant_is_current(&two));
// Rewording the option that only the second set used leaves the first
// set's numbers standing. Before the pool, one distractor edit
// invalidated every statistic the item had.
it.options[2].text = "a different distractor".into();
assert!(it.variant_is_current(&one), "the first set never showed it");
assert!(!it.variant_is_current(&two));
}
#[test]
fn stale_calibration_is_detectable() {
let mut it = item(MINIMAL);
@@ -1053,12 +1493,24 @@ options:
}
#[test]
fn record_change_bumps_version_and_logs() {
fn the_stem_is_the_identity_rather_than_a_version_number() {
let mut it = item(MINIMAL);
let before = it.stem_digest();
// A change log entry numbers itself and leaves the item alone: since
// 2.0 nothing reads `version`, and rewording a stem is not a version
// bump but a new item.
it.record_change("clarified the stem", Some("Alex"));
assert_eq!(it.version, 2);
assert_eq!(it.version, 0);
assert_eq!(it.history.len(), 1);
assert_eq!(it.history[0].version, 2);
assert_eq!(it.history[0].version, 1);
assert_eq!(it.stem_digest(), before);
// The options are the fingerprint's business, not the stem's.
it.options[1].text = "a different distractor".into();
assert_eq!(it.stem_digest(), before);
it.stem = "What is y?".into();
assert_ne!(it.stem_digest(), before);
}
#[test]
+26
View File
@@ -3,6 +3,11 @@
// Source: https://git.scient.ing/education/coursebank
//! The on-disk layout of a course directory.
//!
//! Two of these directories hold fragments of the course file rather than files
//! of their own kind: `lectures/` and `objectives/` are merged into one
//! [`crate::course::CourseFile`] on load, along with `references.yaml`. See
//! [`crate::course::fragment`].
use std::path::PathBuf;
@@ -35,6 +40,25 @@ impl Layout {
self.root.join(COURSE_FILE)
}
/// Path to `references.yaml`, the bibliography when it is kept out of
/// `course.yaml`.
///
/// Optional: absent means the course keeps its `references:` section in the
/// course file, which is how an unsplit course is arranged.
pub fn references_file(&self) -> PathBuf {
self.root.join(crate::course::fragment::REFERENCES_FILE)
}
/// Directory holding one file per lecture.
pub fn lectures(&self) -> PathBuf {
self.root.join("lectures")
}
/// Directory holding one file per learning objective.
pub fn objectives(&self) -> PathBuf {
self.root.join("objectives")
}
/// Directory holding item bank YAML files.
pub fn banks(&self) -> PathBuf {
self.root.join("banks")
@@ -93,6 +117,8 @@ impl Layout {
pub fn create_all(&self) -> Result<()> {
for dir in [
self.root.clone(),
self.lectures(),
self.objectives(),
self.banks(),
self.assessments(),
self.data(),
+48 -4
View File
@@ -145,8 +145,16 @@ pub struct SealedItem {
/// The item's global id.
pub item: String,
/// The item version as administered.
/// Retained only so a pre-2.0 seal still loads. Ignored when reasoning
/// about the item, but still written: [`SealFile::digest_input`] covers it,
/// so dropping it on a round trip would invalidate the digest of every
/// administration sealed before 2.0.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub version: Option<u32>,
/// The stem's digest as administered.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub stem_digest: Option<String>,
/// The item's content fingerprint, the same one
/// [`crate::item::Item::fingerprint`] computes, so a seal and a bank can be
/// compared without re-hashing either by hand.
@@ -484,7 +492,8 @@ fn sealed_item(
SealedItem {
number: placement.number,
item: placement.item.clone(),
version: placement.version.or(Some(item.version)),
version: None,
stem_digest: Some(item.stem_digest()),
fingerprint: item.fingerprint(),
points: placement
.points
@@ -530,7 +539,9 @@ fn sealed_form(catalog: &Catalog, record: &AssessmentFile, form: &Form) -> Resul
for (index, placement) in printed.iter().enumerate() {
let entry = catalog.require(&placement.item)?;
let item = &entry.item;
let n = item.options.len();
// The pool is not the paper: seal what this placement administered.
let shown = item.administered(&placement.key, &placement.distractors);
let n = shown.len();
let order = select::option_order(form, &placement.item, n);
let canonical_key: BTreeSet<String> = if placement.key.is_empty() {
@@ -542,8 +553,7 @@ fn sealed_form(catalog: &Catalog, record: &AssessmentFile, form: &Form) -> Resul
let mut options = Vec::with_capacity(n);
let mut printed_key = Vec::new();
for (position, source_index) in order.iter().enumerate() {
let canonical = item
.options
let canonical = shown
.get(*source_index)
.map(|c| c.id.clone())
.unwrap_or_else(|| printed_letter(*source_index));
@@ -638,6 +648,33 @@ impl SealFile {
yaml::read(path)
}
/// Loads every seal in a directory, oldest administration first.
///
/// # Arguments
///
/// * `dir` - the seals directory.
///
/// # Returns
///
/// The seals, empty when the directory does not exist.
///
/// # Errors
///
/// Propagates load failures, including a seal that does not parse.
pub fn load_all(dir: &Path) -> Result<Vec<SealFile>> {
let mut out = Vec::new();
for path in crate::yaml::list_yaml(dir)? {
out.push(SealFile::load(&path)?);
}
out.sort_by(|a, b| {
a.seal
.date
.cmp(&b.seal.date)
.then(a.seal.assessment.cmp(&b.seal.assessment))
});
Ok(out)
}
/// Loads the seal for an assessment, if one has been written.
///
/// Absence is not an error. A course that has never sealed anything should
@@ -719,6 +756,12 @@ impl SealFile {
self.schema_version, self.seal.assessment, self.seal.course, self.seal.term
));
for item in &self.items {
// Appended only when present: a seal written before 2.0 has to keep
// producing the input it was digested from, or every older
// administration fails verification.
if let Some(stem) = &item.stem_digest {
buf.push_str(&format!("stem\u{1f}{}\u{1f}{stem}\n", item.number));
}
buf.push_str(&format!(
"item\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}",
item.number,
@@ -1157,6 +1200,7 @@ mod tests {
number: 1,
item: "b::q-1".into(),
version: Some(1),
stem_digest: None,
fingerprint: "abc".into(),
points: 1.0,
bonus: false,
+48 -1
View File
@@ -16,7 +16,7 @@ use std::fs;
use std::path::Path;
use serde::de::{self, DeserializeOwned, Visitor};
use serde::{Deserializer, Serialize};
use serde::{Deserialize, Deserializer, Serialize};
use crate::error::{Error, Result};
@@ -61,6 +61,26 @@ pub fn write<T: Serialize>(path: &Path, value: &T) -> Result<()> {
fs::write(path, text).map_err(|e| Error::io(path, e))
}
/// Serializes a value to a YAML string.
///
/// Used where the caller needs to put something in front of the document, such
/// as the banner on a generated file.
///
/// # Arguments
///
/// * `value` - the value to serialize.
///
/// # Returns
///
/// The YAML text.
///
/// # Errors
///
/// Returns [`Error::Other`] if the value cannot be represented as YAML.
pub fn to_string<T: Serialize>(value: &T) -> Result<String> {
serde_yaml_ng::to_string(value).map_err(Error::other)
}
/// Deserializes a JSON file into any type.
///
/// Used only for importing legacy banks and for reading emitted schemas back in
@@ -202,6 +222,33 @@ where
d.deserialize_any(V)
}
/// Deserializes an optional scalar as a string, quoted or not.
///
/// The [`flexible_string`] of a field that may be absent, which is what a
/// fragment's `schema_version` is: one file in a course declares it and the
/// rest inherit.
///
/// # Arguments
///
/// * `d` - the deserializer.
///
/// # Returns
///
/// The value as a string, or `None`.
///
/// # Errors
///
/// Returns a deserialization error for non-scalar input.
pub fn flexible_string_opt<'de, D>(d: D) -> std::result::Result<Option<String>, D::Error>
where
D: Deserializer<'de>,
{
#[derive(serde::Deserialize)]
struct Wrapper(#[serde(deserialize_with = "flexible_string")] String);
Ok(Option::<Wrapper>::deserialize(d)?.map(|w| w.0))
}
#[cfg(test)]
mod tests {
use super::*;