@@ -0,0 +1,863 @@
|
||||
// SPDX-License-Identifier: Prosperity-3.0.0
|
||||
// Copyright Scientific Computing Studio
|
||||
// Source: https://git.scient.ing/education/coursebank
|
||||
|
||||
//! Writing statistics back onto the items.
|
||||
//!
|
||||
//! This is the step that makes the whole system compound. Analysis that lives only
|
||||
//! in a report gets read once; analysis written back onto the item is there the
|
||||
//! next time you consider using it, and the linter can refuse to reuse a question
|
||||
//! that behaved badly.
|
||||
//!
|
||||
//! Calibration accumulates. An item's `calibration` block is not overwritten
|
||||
//! with the latest administration's numbers; the raw responses from every
|
||||
//! administration are pooled and the statistics recomputed.
|
||||
//!
|
||||
//! Rewording resets it. Every calibration records the fingerprint of the item
|
||||
//! text it was computed from. Change the stem or an option and the fingerprint
|
||||
//! changes, the calibration is stale, and the linter says so. Retag the metadata
|
||||
//! and nothing changes, because the fingerprint covers only what a student saw.
|
||||
//! Without that rule, pooled statistics quietly become a mixture of two different
|
||||
//! questions.
|
||||
//!
|
||||
//! Nothing is written without being shown. Calibration produces a diff you
|
||||
//! read before it touches the file. The numbers here decide whether a question gets
|
||||
//! used again, and a silent automated rewrite of a reviewed bank is not something
|
||||
//! you want in a course repository.
|
||||
|
||||
use std::collections::{BTreeMap, BTreeSet};
|
||||
use std::path::PathBuf;
|
||||
|
||||
use crate::assessment::AssessmentFile;
|
||||
use crate::bank::BankFile;
|
||||
use crate::catalog::Catalog;
|
||||
use crate::classical::{self, Analysis, ItemAnalysis, Thresholds};
|
||||
use crate::date::Date;
|
||||
use crate::error::{Error, Result};
|
||||
use crate::irt::{self, Fit};
|
||||
use crate::item::{Calibration, IrtParams, OptionStat};
|
||||
use crate::responses::ResponseSet;
|
||||
use crate::store::Store;
|
||||
use crate::taxonomy::Flag;
|
||||
use crate::yaml;
|
||||
|
||||
/// What calibration would change about one item.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct Change {
|
||||
/// The item's global id.
|
||||
pub uid: String,
|
||||
/// The bank file it lives in.
|
||||
pub path: PathBuf,
|
||||
/// The calibration that would be written.
|
||||
pub calibration: Calibration,
|
||||
/// Human-readable before-and-after lines.
|
||||
pub diff: Vec<String>,
|
||||
/// Whether the item previously had no calibration at all.
|
||||
pub is_new: bool,
|
||||
/// Whether the previous calibration was computed from different item text.
|
||||
pub was_stale: bool,
|
||||
}
|
||||
|
||||
/// The full plan, across items.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct Plan {
|
||||
/// One entry per item whose calibration would change.
|
||||
pub changes: Vec<Change>,
|
||||
/// Items that were analyzed but could not be matched to a bank item.
|
||||
pub unmatched: Vec<String>,
|
||||
/// Cautions worth printing before the diff.
|
||||
pub warnings: Vec<String>,
|
||||
/// How many administrations were pooled.
|
||||
pub administrations: Vec<String>,
|
||||
}
|
||||
|
||||
impl Plan {
|
||||
/// Whether anything would change.
|
||||
pub fn is_empty(&self) -> bool {
|
||||
self.changes.is_empty()
|
||||
}
|
||||
|
||||
/// Renders the plan for a terminal.
|
||||
///
|
||||
/// # Returns
|
||||
///
|
||||
/// A human-readable diff.
|
||||
pub fn render(&self) -> String {
|
||||
let mut out = String::new();
|
||||
if !self.administrations.is_empty() {
|
||||
out.push_str(&format!(
|
||||
"Pooling {} administration(s): {}\n\n",
|
||||
self.administrations.len(),
|
||||
self.administrations.join(", ")
|
||||
));
|
||||
}
|
||||
for w in &self.warnings {
|
||||
out.push_str(&format!("! {w}\n"));
|
||||
}
|
||||
if !self.warnings.is_empty() {
|
||||
out.push('\n');
|
||||
}
|
||||
|
||||
if self.changes.is_empty() {
|
||||
out.push_str("No calibration changes.\n");
|
||||
}
|
||||
for c in &self.changes {
|
||||
let tag = if c.is_new {
|
||||
" (new)"
|
||||
} else if c.was_stale {
|
||||
" (previous calibration was computed from different text)"
|
||||
} else {
|
||||
""
|
||||
};
|
||||
out.push_str(&format!("{}{tag}\n", c.uid));
|
||||
for line in &c.diff {
|
||||
out.push_str(&format!(" {line}\n"));
|
||||
}
|
||||
out.push('\n');
|
||||
}
|
||||
|
||||
if !self.unmatched.is_empty() {
|
||||
out.push_str(&format!(
|
||||
"{} analyzed item(s) could not be traced to a bank item and were skipped: {}\n",
|
||||
self.unmatched.len(),
|
||||
self.unmatched.join(", ")
|
||||
));
|
||||
}
|
||||
out
|
||||
}
|
||||
}
|
||||
|
||||
/// Options for building a plan.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct Options {
|
||||
/// Classical thresholds.
|
||||
pub thresholds: Thresholds,
|
||||
/// Whether to fit IRT and record the parameters.
|
||||
pub irt: bool,
|
||||
/// IRT settings.
|
||||
pub irt_options: irt::Options,
|
||||
/// Whether to include practice assessments. Off by default: practice
|
||||
/// conditions differ enough that pooling them contaminates the statistics.
|
||||
pub include_practice: bool,
|
||||
/// Minimum pooled examinees before writing anything at all.
|
||||
pub minimum_n: usize,
|
||||
}
|
||||
|
||||
impl Default for Options {
|
||||
fn default() -> Options {
|
||||
Options {
|
||||
thresholds: Thresholds::default(),
|
||||
irt: true,
|
||||
irt_options: irt::Options::default(),
|
||||
include_practice: false,
|
||||
minimum_n: 10,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Builds a calibration plan by pooling every stored administration.
|
||||
///
|
||||
/// # Arguments
|
||||
///
|
||||
/// * `catalog` - the loaded course.
|
||||
/// * `store` - the response store.
|
||||
/// * `opts` - calibration options.
|
||||
///
|
||||
/// # Returns
|
||||
///
|
||||
/// The plan, which changes nothing until [`apply`] is called.
|
||||
///
|
||||
/// # Errors
|
||||
///
|
||||
/// Returns [`Error::Io`] when the store cannot be read.
|
||||
pub fn plan(catalog: &Catalog, store: &Store, opts: &Options) -> Result<Plan> {
|
||||
let records = AssessmentFile::load_all(&catalog.layout.assessments())?;
|
||||
let by_id: BTreeMap<&str, &AssessmentFile> = records
|
||||
.iter()
|
||||
.map(|r| (r.assessment.id.as_str(), r))
|
||||
.collect();
|
||||
|
||||
let stored = store.read_all()?;
|
||||
let mut warnings = stored.warnings.clone();
|
||||
|
||||
// Analysis has to happen per administration, not per item. A corrected
|
||||
// point-biserial is a correlation against the rest of that test, so it can
|
||||
// only be computed while the whole administration is in hand. Pooling happens
|
||||
// afterward, on the resulting numbers.
|
||||
let mut per_admin: BTreeMap<String, Analysis> = BTreeMap::new();
|
||||
let mut administrations: BTreeSet<String> = BTreeSet::new();
|
||||
|
||||
for admin in stored.administrations() {
|
||||
let rows: Vec<crate::responses::Response> = stored
|
||||
.rows
|
||||
.iter()
|
||||
.filter(|r| r.administration_id == admin)
|
||||
.cloned()
|
||||
.collect();
|
||||
let Some(first) = rows.first() else { continue };
|
||||
let record = by_id.get(first.assessment_id.as_str()).copied();
|
||||
|
||||
if let Some(rec) = record {
|
||||
if !opts.include_practice && !rec.assessment.kind.counts_for_calibration() {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
let mut set = ResponseSet::new();
|
||||
set.rows = rows;
|
||||
let analysis = classical::analyze(&set, &opts.thresholds, record, Some(catalog));
|
||||
administrations.insert(admin.clone());
|
||||
per_admin.insert(admin, analysis);
|
||||
}
|
||||
|
||||
// Now group the per-item results by the bank item they refer to. The same item
|
||||
// may have been question 7 one term and question 12 the next.
|
||||
let mut by_item: BTreeMap<String, Vec<(String, ItemAnalysis)>> = BTreeMap::new();
|
||||
let mut unmatched: BTreeSet<String> = BTreeSet::new();
|
||||
|
||||
for (admin, analysis) in &per_admin {
|
||||
for item in &analysis.items {
|
||||
match &item.item_ref {
|
||||
Some(uid) if catalog.get(uid).is_some() => {
|
||||
by_item
|
||||
.entry(uid.clone())
|
||||
.or_default()
|
||||
.push((admin.clone(), item.clone()));
|
||||
}
|
||||
Some(uid) => {
|
||||
unmatched.insert(uid.clone());
|
||||
}
|
||||
None => {
|
||||
unmatched.insert(format!("{admin}#{}", item.number));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if by_item.is_empty() {
|
||||
warnings.push(
|
||||
"no stored responses could be traced to bank items; check that `ingest` was run with \
|
||||
an assessment record so item references were recorded"
|
||||
.to_string(),
|
||||
);
|
||||
}
|
||||
|
||||
// The IRT fit uses whichever administration has the most complete matrix,
|
||||
// rather than a pooled matrix. Pooling responses across forms into one matrix
|
||||
// would treat students who never saw an item as having missed it in a way the
|
||||
// likelihood cannot distinguish from a linked design, and honest linking is a
|
||||
// bigger problem than this tool should pretend to solve.
|
||||
let irt_fit = if opts.irt {
|
||||
best_fit(&stored, opts)
|
||||
} else {
|
||||
None
|
||||
};
|
||||
if opts.irt && irt_fit.is_none() {
|
||||
warnings.push(
|
||||
"IRT was requested but no single administration had enough data to fit; classical \
|
||||
statistics will still be written"
|
||||
.to_string(),
|
||||
);
|
||||
}
|
||||
if let Some((admin, fit)) = &irt_fit {
|
||||
warnings.extend(fit.warnings.clone());
|
||||
warnings.push(format!(
|
||||
"IRT parameters come from a single administration ({admin}) rather than from the \
|
||||
pooled data, because linking across forms is not attempted"
|
||||
));
|
||||
}
|
||||
|
||||
let mut changes = Vec::new();
|
||||
|
||||
for (uid, appearances) in &by_item {
|
||||
let entry = match catalog.get(uid) {
|
||||
Some(e) => e,
|
||||
None => continue,
|
||||
};
|
||||
|
||||
let pooled = pool(appearances);
|
||||
if pooled.n < opts.minimum_n {
|
||||
continue;
|
||||
}
|
||||
|
||||
// The IRT parameters come from whichever single administration was fitted,
|
||||
// matched by the question number this item held there.
|
||||
let irt_params = match &irt_fit {
|
||||
Some((fitted_admin, fit)) => appearances
|
||||
.iter()
|
||||
.find(|(admin, _)| admin == fitted_admin)
|
||||
.and_then(|(_, item)| fit.items.iter().find(|i| i.number == item.number))
|
||||
.map(|i| i.to_params()),
|
||||
None => None,
|
||||
};
|
||||
|
||||
let calibration = Calibration {
|
||||
administrations: appearances.iter().map(|(a, _)| a.clone()).collect(),
|
||||
updated: Some(Date::today()),
|
||||
fingerprint: Some(entry.item.fingerprint()),
|
||||
n_examinees: Some(pooled.n),
|
||||
p_value: Some(round4(pooled.p_value)),
|
||||
point_biserial: pooled.point_biserial.map(round4),
|
||||
discrimination_index: pooled.discrimination_index.map(round4),
|
||||
mean_response_time_seconds: None,
|
||||
rapid_guess_rate: None,
|
||||
option_stats: pooled.option_stats.clone(),
|
||||
irt: irt_params,
|
||||
flags: pooled.flags.clone(),
|
||||
};
|
||||
|
||||
let previous = entry.item.calibration.as_ref();
|
||||
let diff = diff_calibration(previous, &calibration);
|
||||
if diff.is_empty() {
|
||||
continue;
|
||||
}
|
||||
|
||||
changes.push(Change {
|
||||
uid: uid.clone(),
|
||||
path: entry.path.clone(),
|
||||
calibration,
|
||||
diff,
|
||||
is_new: previous.is_none(),
|
||||
was_stale: previous
|
||||
.map(|p| p.fingerprint.as_deref() != Some(entry.item.fingerprint().as_str()))
|
||||
.unwrap_or(false),
|
||||
});
|
||||
}
|
||||
|
||||
Ok(Plan {
|
||||
changes,
|
||||
unmatched: unmatched.into_iter().collect(),
|
||||
warnings,
|
||||
administrations: administrations.into_iter().collect(),
|
||||
})
|
||||
}
|
||||
|
||||
/// Pooled statistics for one item.
|
||||
struct Pooled {
|
||||
/// Total examinees across administrations.
|
||||
n: usize,
|
||||
/// Examinee-weighted difficulty.
|
||||
p_value: f64,
|
||||
/// Examinee-weighted point-biserial.
|
||||
point_biserial: Option<f64>,
|
||||
/// Examinee-weighted discrimination index.
|
||||
discrimination_index: Option<f64>,
|
||||
/// Pooled per-option statistics.
|
||||
option_stats: BTreeMap<String, OptionStat>,
|
||||
/// The union of flags raised in any administration.
|
||||
flags: Vec<Flag>,
|
||||
}
|
||||
|
||||
/// Pools one item's statistics across the administrations it appeared in.
|
||||
///
|
||||
/// Difficulty pools by weighted average, since a proportion correct is comparable
|
||||
/// across administrations of the same text. Discrimination is also averaged rather
|
||||
/// than recomputed, and that is the important subtlety: a corrected point-biserial
|
||||
/// is a correlation against the rest of that test, so the only meaningful pooled
|
||||
/// value is a weighted average of the within-administration correlations. Merging
|
||||
/// response matrices from different exams and correlating across the whole thing
|
||||
/// would produce a number that looks more precise and means less.
|
||||
///
|
||||
/// # Arguments
|
||||
///
|
||||
/// * `appearances` - the administration id and item analysis for each appearance.
|
||||
///
|
||||
/// # Returns
|
||||
///
|
||||
/// The pooled statistics.
|
||||
fn pool(appearances: &[(String, ItemAnalysis)]) -> Pooled {
|
||||
let mut total_n = 0usize;
|
||||
let mut p_weighted = 0.0;
|
||||
let mut rpb_weighted = 0.0;
|
||||
let mut rpb_weight = 0.0;
|
||||
let mut d_weighted = 0.0;
|
||||
let mut d_weight = 0.0;
|
||||
let mut flags: BTreeSet<Flag> = BTreeSet::new();
|
||||
|
||||
// Per-option accumulators, since option letters are stable across forms even
|
||||
// when the printed order is not.
|
||||
let mut rate_weighted: BTreeMap<String, f64> = BTreeMap::new();
|
||||
let mut option_rpb: BTreeMap<String, (f64, f64)> = BTreeMap::new();
|
||||
let mut upper: BTreeMap<String, (f64, f64)> = BTreeMap::new();
|
||||
let mut lower: BTreeMap<String, (f64, f64)> = BTreeMap::new();
|
||||
|
||||
for (_, item) in appearances {
|
||||
let n = item.n as f64;
|
||||
total_n += item.n;
|
||||
p_weighted += item.p_value * n;
|
||||
|
||||
if let Some(r) = item.point_biserial {
|
||||
rpb_weighted += r * n;
|
||||
rpb_weight += n;
|
||||
}
|
||||
if let Some(d) = item.discrimination_index {
|
||||
d_weighted += d * n;
|
||||
d_weight += n;
|
||||
}
|
||||
for f in &item.flags {
|
||||
flags.insert(*f);
|
||||
}
|
||||
for (letter, o) in &item.options {
|
||||
*rate_weighted.entry(letter.clone()).or_insert(0.0) += o.rate * n;
|
||||
if let Some(r) = o.point_biserial {
|
||||
let e = option_rpb.entry(letter.clone()).or_insert((0.0, 0.0));
|
||||
e.0 += r * n;
|
||||
e.1 += n;
|
||||
}
|
||||
if let Some(r) = o.upper_rate {
|
||||
let e = upper.entry(letter.clone()).or_insert((0.0, 0.0));
|
||||
e.0 += r * n;
|
||||
e.1 += n;
|
||||
}
|
||||
if let Some(r) = o.lower_rate {
|
||||
let e = lower.entry(letter.clone()).or_insert((0.0, 0.0));
|
||||
e.0 += r * n;
|
||||
e.1 += n;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let denominator = total_n.max(1) as f64;
|
||||
let average = |m: &BTreeMap<String, (f64, f64)>, letter: &str| -> Option<f64> {
|
||||
m.get(letter)
|
||||
.filter(|(_, w)| *w > 0.0)
|
||||
.map(|(sum, w)| round4(sum / w))
|
||||
};
|
||||
|
||||
let option_stats: BTreeMap<String, OptionStat> = rate_weighted
|
||||
.keys()
|
||||
.map(|letter| {
|
||||
(
|
||||
letter.clone(),
|
||||
OptionStat {
|
||||
selection_rate: Some(round4(rate_weighted[letter] / denominator)),
|
||||
point_biserial: average(&option_rpb, letter),
|
||||
upper_group_rate: average(&upper, letter),
|
||||
lower_group_rate: average(&lower, letter),
|
||||
},
|
||||
)
|
||||
})
|
||||
.collect();
|
||||
|
||||
Pooled {
|
||||
n: total_n,
|
||||
p_value: p_weighted / denominator,
|
||||
point_biserial: if rpb_weight > 0.0 {
|
||||
Some(rpb_weighted / rpb_weight)
|
||||
} else {
|
||||
None
|
||||
},
|
||||
discrimination_index: if d_weight > 0.0 {
|
||||
Some(d_weighted / d_weight)
|
||||
} else {
|
||||
None
|
||||
},
|
||||
option_stats,
|
||||
flags: flags.into_iter().collect(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Picks the administration with the most complete matrix and fits IRT to it.
|
||||
///
|
||||
/// # Arguments
|
||||
///
|
||||
/// * `stored` - every stored response.
|
||||
/// * `opts` - calibration options.
|
||||
///
|
||||
/// # Returns
|
||||
///
|
||||
/// The administration id and its fit, or `None` when none is large enough.
|
||||
fn best_fit(stored: &ResponseSet, opts: &Options) -> Option<(String, Fit)> {
|
||||
let mut best: Option<(String, usize, usize)> = None;
|
||||
for admin in stored.administrations() {
|
||||
let mut set = ResponseSet::new();
|
||||
set.rows = stored
|
||||
.rows
|
||||
.iter()
|
||||
.filter(|r| r.administration_id == admin)
|
||||
.cloned()
|
||||
.collect();
|
||||
let m = set.matrix(false);
|
||||
let cells = m.n_students() * m.n_items();
|
||||
if m.n_students() < opts.minimum_n || m.n_items() < 5 {
|
||||
continue;
|
||||
}
|
||||
if best.as_ref().map(|(_, c, _)| cells > *c).unwrap_or(true) {
|
||||
best = Some((admin, cells, m.n_items()));
|
||||
}
|
||||
}
|
||||
|
||||
let (admin, _, _) = best?;
|
||||
let mut set = ResponseSet::new();
|
||||
set.rows = stored
|
||||
.rows
|
||||
.iter()
|
||||
.filter(|r| r.administration_id == admin)
|
||||
.cloned()
|
||||
.collect();
|
||||
let fit = irt::fit(&set.matrix(false), &opts.irt_options);
|
||||
Some((admin, fit))
|
||||
}
|
||||
|
||||
/// Describes the difference between two calibrations.
|
||||
///
|
||||
/// # Arguments
|
||||
///
|
||||
/// * `previous` - the existing calibration, if any.
|
||||
/// * `next` - the computed calibration.
|
||||
///
|
||||
/// # Returns
|
||||
///
|
||||
/// One line per changed field; empty when nothing meaningful changed.
|
||||
fn diff_calibration(previous: Option<&Calibration>, next: &Calibration) -> Vec<String> {
|
||||
let mut out = Vec::new();
|
||||
let show = |label: &str, before: Option<f64>, after: Option<f64>, out: &mut Vec<String>| match (
|
||||
before, after,
|
||||
) {
|
||||
(Some(b), Some(a)) if (b - a).abs() > 5e-4 => {
|
||||
out.push(format!("{label}: {b:.3} -> {a:.3}"));
|
||||
}
|
||||
(None, Some(a)) => out.push(format!("{label}: (none) -> {a:.3}")),
|
||||
_ => {}
|
||||
};
|
||||
|
||||
let p = previous;
|
||||
show("p-value", p.and_then(|c| c.p_value), next.p_value, &mut out);
|
||||
show(
|
||||
"point-biserial",
|
||||
p.and_then(|c| c.point_biserial),
|
||||
next.point_biserial,
|
||||
&mut out,
|
||||
);
|
||||
show(
|
||||
"discrimination index",
|
||||
p.and_then(|c| c.discrimination_index),
|
||||
next.discrimination_index,
|
||||
&mut out,
|
||||
);
|
||||
|
||||
let before_n = p.and_then(|c| c.n_examinees).unwrap_or(0);
|
||||
if let Some(n) = next.n_examinees {
|
||||
if n != before_n {
|
||||
out.push(format!("examinees: {before_n} -> {n}"));
|
||||
}
|
||||
}
|
||||
|
||||
let before_flags: BTreeSet<Flag> = p
|
||||
.map(|c| c.flags.iter().copied().collect())
|
||||
.unwrap_or_default();
|
||||
let after_flags: BTreeSet<Flag> = next.flags.iter().copied().collect();
|
||||
let added: Vec<&str> = after_flags
|
||||
.difference(&before_flags)
|
||||
.map(|f| f.as_str())
|
||||
.collect();
|
||||
let removed: Vec<&str> = before_flags
|
||||
.difference(&after_flags)
|
||||
.map(|f| f.as_str())
|
||||
.collect();
|
||||
if !added.is_empty() {
|
||||
out.push(format!("flags added: {}", added.join(", ")));
|
||||
}
|
||||
if !removed.is_empty() {
|
||||
out.push(format!("flags cleared: {}", removed.join(", ")));
|
||||
}
|
||||
|
||||
match (p.and_then(|c| c.irt.as_ref()), next.irt.as_ref()) {
|
||||
(Some(b), Some(a)) if (b.a - a.a).abs() > 5e-3 || (b.b - a.b).abs() > 5e-3 => {
|
||||
out.push(format!(
|
||||
"IRT: a {:.2} -> {:.2}, b {:+.2} -> {:+.2}",
|
||||
b.a, a.a, b.b, a.b
|
||||
));
|
||||
}
|
||||
(None, Some(a)) => out.push(format!("IRT: (none) -> a {:.2}, b {:+.2}", a.a, a.b)),
|
||||
_ => {}
|
||||
}
|
||||
|
||||
if p.map(|c| c.fingerprint.as_deref()) != Some(next.fingerprint.as_deref()) {
|
||||
out.push("fingerprint updated to the current item text".to_string());
|
||||
}
|
||||
|
||||
out
|
||||
}
|
||||
|
||||
/// Applies a plan, rewriting the affected bank files.
|
||||
///
|
||||
/// Files are rewritten one at a time and each is re-read before editing, so a plan
|
||||
/// built against a bank that has since changed on disk fails loudly rather than
|
||||
/// clobbering the newer version.
|
||||
///
|
||||
/// # Arguments
|
||||
///
|
||||
/// * `plan` - the plan to apply.
|
||||
///
|
||||
/// # Returns
|
||||
///
|
||||
/// The bank files rewritten.
|
||||
///
|
||||
/// # Errors
|
||||
///
|
||||
/// Returns [`Error::Unresolved`] when an item in the plan is no longer in its bank,
|
||||
/// and [`Error::Io`] on a write failure.
|
||||
pub fn apply(plan: &Plan) -> Result<Vec<PathBuf>> {
|
||||
// Group by file so each is read and written once.
|
||||
let mut by_file: BTreeMap<&PathBuf, Vec<&Change>> = BTreeMap::new();
|
||||
for change in &plan.changes {
|
||||
by_file.entry(&change.path).or_default().push(change);
|
||||
}
|
||||
|
||||
let mut written = Vec::new();
|
||||
for (path, changes) in by_file {
|
||||
let mut bank: BankFile = yaml::read(path)?;
|
||||
for change in changes {
|
||||
// The uid is `bank::item`; match on the item part.
|
||||
let item_id = change
|
||||
.uid
|
||||
.split_once("::")
|
||||
.map(|(_, id)| id)
|
||||
.unwrap_or(&change.uid);
|
||||
let target = bank.items.iter_mut().find(|i| i.id == item_id);
|
||||
match target {
|
||||
Some(item) => item.calibration = Some(change.calibration.clone()),
|
||||
None => {
|
||||
return Err(Error::Unresolved {
|
||||
kind: "item",
|
||||
id: change.uid.clone(),
|
||||
context: Some(format!(
|
||||
"{} — the bank changed since the plan was built; re-run calibration",
|
||||
path.display()
|
||||
)),
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
yaml::write(path, &bank)?;
|
||||
written.push(path.clone());
|
||||
}
|
||||
Ok(written)
|
||||
}
|
||||
|
||||
/// Builds a plan for a single administration, from an in-memory analysis.
|
||||
///
|
||||
/// Useful right after an exam, before deciding whether to drop a question.
|
||||
///
|
||||
/// # Arguments
|
||||
///
|
||||
/// * `catalog` - the loaded course.
|
||||
/// * `record` - the assessment record.
|
||||
/// * `analysis` - the analysis of that administration.
|
||||
/// * `fit` - an optional IRT fit.
|
||||
///
|
||||
/// # Returns
|
||||
///
|
||||
/// The plan.
|
||||
pub fn plan_from_analysis(
|
||||
catalog: &Catalog,
|
||||
record: &AssessmentFile,
|
||||
analysis: &Analysis,
|
||||
fit: Option<&Fit>,
|
||||
) -> Plan {
|
||||
let admin = crate::responses::administration_id(
|
||||
&catalog.course.course.code,
|
||||
record
|
||||
.assessment
|
||||
.term
|
||||
.as_deref()
|
||||
.unwrap_or(&catalog.course.course.term),
|
||||
&record.assessment.id,
|
||||
);
|
||||
|
||||
let mut changes = Vec::new();
|
||||
let mut unmatched = Vec::new();
|
||||
|
||||
for item in &analysis.items {
|
||||
let Some(uid) = item.item_ref.clone() else {
|
||||
unmatched.push(format!("question {}", item.number));
|
||||
continue;
|
||||
};
|
||||
let Some(entry) = catalog.get(&uid) else {
|
||||
unmatched.push(uid);
|
||||
continue;
|
||||
};
|
||||
|
||||
let irt_params: Option<IrtParams> = fit.and_then(|f| {
|
||||
f.items
|
||||
.iter()
|
||||
.find(|i| i.number == item.number)
|
||||
.map(|i| i.to_params())
|
||||
});
|
||||
|
||||
let calibration = Calibration {
|
||||
administrations: vec![admin.clone()],
|
||||
updated: Some(Date::today()),
|
||||
fingerprint: Some(entry.item.fingerprint()),
|
||||
n_examinees: Some(item.n),
|
||||
p_value: Some(round4(item.p_value)),
|
||||
point_biserial: item.point_biserial.map(round4),
|
||||
discrimination_index: item.discrimination_index.map(round4),
|
||||
mean_response_time_seconds: None,
|
||||
rapid_guess_rate: None,
|
||||
option_stats: item
|
||||
.options
|
||||
.iter()
|
||||
.map(|(letter, o)| (letter.clone(), o.to_option_stat()))
|
||||
.collect(),
|
||||
irt: irt_params,
|
||||
flags: item.flags.clone(),
|
||||
};
|
||||
|
||||
let previous = entry.item.calibration.as_ref();
|
||||
let diff = diff_calibration(previous, &calibration);
|
||||
if diff.is_empty() {
|
||||
continue;
|
||||
}
|
||||
changes.push(Change {
|
||||
uid,
|
||||
path: entry.path.clone(),
|
||||
calibration,
|
||||
diff,
|
||||
is_new: previous.is_none(),
|
||||
was_stale: previous
|
||||
.map(|p| p.fingerprint.as_deref() != Some(entry.item.fingerprint().as_str()))
|
||||
.unwrap_or(false),
|
||||
});
|
||||
}
|
||||
|
||||
Plan {
|
||||
changes,
|
||||
unmatched,
|
||||
warnings: analysis.warnings.clone(),
|
||||
administrations: vec![admin],
|
||||
}
|
||||
}
|
||||
|
||||
/// Rounds to four decimals.
|
||||
fn round4(x: f64) -> f64 {
|
||||
(x * 1e4).round() / 1e4
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::item::IrtModel;
|
||||
|
||||
fn calibration(p: f64, rpb: Option<f64>, n: usize, fingerprint: &str) -> Calibration {
|
||||
Calibration {
|
||||
administrations: vec!["C/2026s/e1".into()],
|
||||
updated: None,
|
||||
fingerprint: Some(fingerprint.to_string()),
|
||||
n_examinees: Some(n),
|
||||
p_value: Some(p),
|
||||
point_biserial: rpb,
|
||||
discrimination_index: None,
|
||||
mean_response_time_seconds: None,
|
||||
rapid_guess_rate: None,
|
||||
option_stats: BTreeMap::new(),
|
||||
irt: None,
|
||||
flags: Vec::new(),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_first_calibration_is_all_new() {
|
||||
let next = calibration(0.7, Some(0.3), 24, "abc");
|
||||
let diff = diff_calibration(None, &next);
|
||||
assert!(diff.iter().any(|d| d.contains("p-value: (none)")));
|
||||
assert!(diff.iter().any(|d| d.contains("examinees: 0 -> 24")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn identical_calibrations_produce_no_diff() {
|
||||
let previous = calibration(0.7, Some(0.3), 24, "abc");
|
||||
let next = calibration(0.7, Some(0.3), 24, "abc");
|
||||
assert!(diff_calibration(Some(&previous), &next).is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tiny_changes_are_not_reported() {
|
||||
let previous = calibration(0.7000, Some(0.30), 24, "abc");
|
||||
let next = calibration(0.7001, Some(0.30), 24, "abc");
|
||||
assert!(
|
||||
diff_calibration(Some(&previous), &next).is_empty(),
|
||||
"a change below the display precision is noise"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_changed_fingerprint_is_called_out() {
|
||||
let previous = calibration(0.7, Some(0.3), 24, "old-text");
|
||||
let next = calibration(0.7, Some(0.3), 24, "new-text");
|
||||
let diff = diff_calibration(Some(&previous), &next);
|
||||
assert!(diff.iter().any(|d| d.contains("fingerprint")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn flag_changes_are_reported_in_both_directions() {
|
||||
let mut previous = calibration(0.7, Some(0.3), 24, "abc");
|
||||
previous.flags = vec![Flag::TooEasy];
|
||||
let mut next = calibration(0.7, Some(0.3), 24, "abc");
|
||||
next.flags = vec![Flag::NegativeDiscrimination];
|
||||
|
||||
let diff = diff_calibration(Some(&previous), &next);
|
||||
assert!(
|
||||
diff.iter()
|
||||
.any(|d| d.contains("flags added") && d.contains("negative_discrimination"))
|
||||
);
|
||||
assert!(
|
||||
diff.iter()
|
||||
.any(|d| d.contains("flags cleared") && d.contains("too_easy"))
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn irt_changes_are_reported() {
|
||||
let previous = calibration(0.7, None, 24, "abc");
|
||||
let mut next = calibration(0.7, None, 24, "abc");
|
||||
next.irt = Some(IrtParams {
|
||||
model: IrtModel::TwoPl,
|
||||
a: 1.2,
|
||||
b: -0.4,
|
||||
c: None,
|
||||
se_a: None,
|
||||
se_b: None,
|
||||
n: Some(24),
|
||||
bayesian: true,
|
||||
});
|
||||
let diff = diff_calibration(Some(&previous), &next);
|
||||
assert!(diff.iter().any(|d| d.contains("IRT: (none)")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_empty_plan_renders_readably() {
|
||||
let plan = Plan {
|
||||
changes: Vec::new(),
|
||||
unmatched: Vec::new(),
|
||||
warnings: Vec::new(),
|
||||
administrations: Vec::new(),
|
||||
};
|
||||
assert!(plan.is_empty());
|
||||
assert!(plan.render().contains("No calibration changes"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_plan_renders_its_diff_and_warnings() {
|
||||
let plan = Plan {
|
||||
changes: vec![Change {
|
||||
uid: "bank::q-a-001".into(),
|
||||
path: PathBuf::from("banks/bank.yaml"),
|
||||
calibration: calibration(0.7, Some(0.3), 24, "abc"),
|
||||
diff: vec!["p-value: (none) -> 0.700".into()],
|
||||
is_new: true,
|
||||
was_stale: false,
|
||||
}],
|
||||
unmatched: vec!["question 9".into()],
|
||||
warnings: vec!["only 24 examinees".into()],
|
||||
administrations: vec!["C/2026s/e1".into()],
|
||||
};
|
||||
let text = plan.render();
|
||||
assert!(text.contains("bank::q-a-001 (new)"));
|
||||
assert!(text.contains("p-value"));
|
||||
assert!(text.contains("only 24 examinees"));
|
||||
assert!(text.contains("question 9"));
|
||||
assert!(text.contains("Pooling 1 administration"));
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
+1176
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
Reference in New Issue
Block a user