refactor: improve cohort report

This commit is contained in:
2026-09-19 22:19:25 -04:00
parent c06d5caa8c
commit 92c8a8fc75
6 changed files with 2194 additions and 311 deletions
+119 -17
View File
@@ -61,6 +61,13 @@ pub struct Thresholds {
pub nonfunctioning: f64,
/// How far observed difficulty may drift from the authored expectation.
pub design_tolerance: f64,
/// How many examinees an item's calibration needs before its recorded
/// expectations are treated as evidence rather than as the author's guess.
///
/// Fifty is the point at which the standard error of a proportion near 0.5
/// drops to about 0.07, which is small enough that a quarter-point miss is
/// about the item rather than about the sample.
pub calibrated_n: usize,
/// Fraction of the class in the upper and lower comparison groups. Kelley's
/// 0.27 maximizes the difference between the groups for a normal
/// distribution, and it remains the convention.
@@ -78,6 +85,7 @@ impl Default for Thresholds {
negative_discrimination: -0.05,
nonfunctioning: 0.05,
design_tolerance: 0.25,
calibrated_n: 50,
group_fraction: 0.27,
small_sample: 100,
}
@@ -154,6 +162,38 @@ pub struct ItemAnalysis {
pub flags: Vec<Flag>,
/// Human-readable explanations tied to the flags.
pub notes: Vec<String>,
/// How the item behaved against what its author predicted, when the item
/// records a prediction.
pub prediction: Option<Prediction>,
}
/// An authored expectation, checked against what happened.
///
/// Kept apart from [`ItemAnalysis::flags`] on purpose. Before an item has been
/// administered, `design.expected_difficulty` is the author's guess, and a guess
/// that turns out wrong says something about the guess rather than about the
/// item. Flagging it anyway is how a report ends up with thirty
/// `design_mismatch` findings and no way to see the four that matter. So the
/// discrepancy is always recorded here, and it only becomes a
/// [`Flag::DesignMismatch`] once the expectation has data behind it.
#[derive(Debug, Clone)]
pub struct Prediction {
/// The difficulty the author expected.
pub expected_p: Option<f64>,
/// The discrimination band the author expected, as `(low, high)`.
pub expected_band: Option<(f64, f64)>,
/// Whether the expectation rests on a calibration with enough examinees
/// behind it, rather than on the author's judgement alone.
pub calibrated: bool,
/// Signed difficulty error, observed minus expected. Positive means the item
/// was easier than predicted.
pub p_error: Option<f64>,
/// Whether observed difficulty landed inside the tolerance.
pub p_within: Option<bool>,
/// Whether observed discrimination landed inside the expected band.
pub band_hit: Option<bool>,
/// What to say about it, phrased for whichever case applies.
pub notes: Vec<String>,
}
impl ItemAnalysis {
@@ -493,9 +533,29 @@ pub fn analyze(
options,
flags: Vec::new(),
notes: Vec::new(),
prediction: None,
};
flag_item(&mut analysis, t, design.as_ref(), &rows);
// Whether the authored expectation is evidence or a guess. An item that
// has never been administered has no calibration block, and one edited
// since its last calibration has a fingerprint that no longer matches.
let calibrated = record
.and_then(|r| r.placement(*number))
.and_then(|p| catalog.and_then(|c| c.get(&p.item)))
.and_then(|entry| {
let cal = entry.item.calibration.as_ref()?;
let enough = cal.n_examinees.unwrap_or(0) >= t.calibrated_n;
let current = match &cal.fingerprint {
Some(recorded) => *recorded == entry.item.fingerprint(),
// An older calibration block with no fingerprint cannot be
// shown stale, so it is taken at its word.
None => true,
};
Some(enough && current)
})
.unwrap_or(false);
flag_item(&mut analysis, t, design.as_ref(), &rows, calibrated);
p_values.push(p_value);
if let Some(r) = rpb {
@@ -521,11 +581,13 @@ pub fn analyze(
/// * `t` - the thresholds.
/// * `design` - the authored expectation, when available.
/// * `rows` - the raw responses, for partial-credit detection.
/// * `calibrated` - whether that expectation rests on prior data.
fn flag_item(
a: &mut ItemAnalysis,
t: &Thresholds,
design: Option<&Design>,
rows: &[&crate::responses::Response],
calibrated: bool,
) {
// Discrimination first: it is the finding that changes what you do.
match a.point_biserial {
@@ -678,31 +740,71 @@ fn flag_item(
}
}
// Did the item behave as authored?
// Did the item behave as authored? This is the one check whose meaning
// depends on where the expectation came from, so it is recorded either way
// and flagged only when the expectation had data behind it.
if let Some(d) = design {
let mut prediction = Prediction {
expected_p: d.expected_difficulty,
expected_band: d.expected_discrimination.map(|b| b.expected_band()),
calibrated,
p_error: None,
p_within: None,
band_hit: None,
notes: Vec::new(),
};
if let Some(expected) = d.expected_difficulty {
if (expected - a.p_value).abs() > t.design_tolerance {
a.flags.push(Flag::DesignMismatch);
a.notes.push(format!(
"you expected about {:.0}% correct and observed {:.0}%. Worth knowing whether \
your model of the students or the item is off.",
expected * 100.0,
a.p_value * 100.0
));
let error = a.p_value - expected;
let within = error.abs() <= t.design_tolerance;
prediction.p_error = Some(error);
prediction.p_within = Some(within);
if !within {
if calibrated {
a.flags.push(Flag::DesignMismatch);
prediction.notes.push(format!(
"this item is calibrated at about {:.0}% correct and came out at {:.0}%. \
Something changed: the cohort, the teaching, or the item.",
expected * 100.0,
a.p_value * 100.0
));
} else {
prediction.notes.push(format!(
"you predicted about {:.0}% correct and observed {:.0}%. This is the \
first data on the item, so it corrects the prediction rather than \
condemning the item.",
expected * 100.0,
a.p_value * 100.0
));
}
}
}
if let (Some(band), Some(r)) = (d.expected_discrimination, a.point_biserial) {
let (low, high) = band.expected_band();
if r < low || r > high {
if !a.flags.contains(&Flag::DesignMismatch) {
a.flags.push(Flag::DesignMismatch);
let hit = r >= low && r <= high;
prediction.band_hit = Some(hit);
if !hit {
if calibrated {
if !a.flags.contains(&Flag::DesignMismatch) {
a.flags.push(Flag::DesignMismatch);
}
prediction.notes.push(format!(
"calibrated for {} discrimination ({low:.2} to {high:.2}), observed \
{r:.2}.",
format!("{band:?}").to_lowercase()
));
} else {
prediction.notes.push(format!(
"you predicted {} discrimination ({low:.2} to {high:.2}) and observed \
{r:.2}.",
format!("{band:?}").to_lowercase()
));
}
a.notes.push(format!(
"you expected {} discrimination ({low:.2} to {high:.2}) and observed {r:.2}.",
format!("{band:?}").to_lowercase()
));
}
}
a.prediction = Some(prediction);
}
a.flags.sort();