refactor: improve cohort report
This commit is contained in:
+119
-17
@@ -61,6 +61,13 @@ pub struct Thresholds {
|
||||
pub nonfunctioning: f64,
|
||||
/// How far observed difficulty may drift from the authored expectation.
|
||||
pub design_tolerance: f64,
|
||||
/// How many examinees an item's calibration needs before its recorded
|
||||
/// expectations are treated as evidence rather than as the author's guess.
|
||||
///
|
||||
/// Fifty is the point at which the standard error of a proportion near 0.5
|
||||
/// drops to about 0.07, which is small enough that a quarter-point miss is
|
||||
/// about the item rather than about the sample.
|
||||
pub calibrated_n: usize,
|
||||
/// Fraction of the class in the upper and lower comparison groups. Kelley's
|
||||
/// 0.27 maximizes the difference between the groups for a normal
|
||||
/// distribution, and it remains the convention.
|
||||
@@ -78,6 +85,7 @@ impl Default for Thresholds {
|
||||
negative_discrimination: -0.05,
|
||||
nonfunctioning: 0.05,
|
||||
design_tolerance: 0.25,
|
||||
calibrated_n: 50,
|
||||
group_fraction: 0.27,
|
||||
small_sample: 100,
|
||||
}
|
||||
@@ -154,6 +162,38 @@ pub struct ItemAnalysis {
|
||||
pub flags: Vec<Flag>,
|
||||
/// Human-readable explanations tied to the flags.
|
||||
pub notes: Vec<String>,
|
||||
/// How the item behaved against what its author predicted, when the item
|
||||
/// records a prediction.
|
||||
pub prediction: Option<Prediction>,
|
||||
}
|
||||
|
||||
/// An authored expectation, checked against what happened.
|
||||
///
|
||||
/// Kept apart from [`ItemAnalysis::flags`] on purpose. Before an item has been
|
||||
/// administered, `design.expected_difficulty` is the author's guess, and a guess
|
||||
/// that turns out wrong says something about the guess rather than about the
|
||||
/// item. Flagging it anyway is how a report ends up with thirty
|
||||
/// `design_mismatch` findings and no way to see the four that matter. So the
|
||||
/// discrepancy is always recorded here, and it only becomes a
|
||||
/// [`Flag::DesignMismatch`] once the expectation has data behind it.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct Prediction {
|
||||
/// The difficulty the author expected.
|
||||
pub expected_p: Option<f64>,
|
||||
/// The discrimination band the author expected, as `(low, high)`.
|
||||
pub expected_band: Option<(f64, f64)>,
|
||||
/// Whether the expectation rests on a calibration with enough examinees
|
||||
/// behind it, rather than on the author's judgement alone.
|
||||
pub calibrated: bool,
|
||||
/// Signed difficulty error, observed minus expected. Positive means the item
|
||||
/// was easier than predicted.
|
||||
pub p_error: Option<f64>,
|
||||
/// Whether observed difficulty landed inside the tolerance.
|
||||
pub p_within: Option<bool>,
|
||||
/// Whether observed discrimination landed inside the expected band.
|
||||
pub band_hit: Option<bool>,
|
||||
/// What to say about it, phrased for whichever case applies.
|
||||
pub notes: Vec<String>,
|
||||
}
|
||||
|
||||
impl ItemAnalysis {
|
||||
@@ -493,9 +533,29 @@ pub fn analyze(
|
||||
options,
|
||||
flags: Vec::new(),
|
||||
notes: Vec::new(),
|
||||
prediction: None,
|
||||
};
|
||||
|
||||
flag_item(&mut analysis, t, design.as_ref(), &rows);
|
||||
// Whether the authored expectation is evidence or a guess. An item that
|
||||
// has never been administered has no calibration block, and one edited
|
||||
// since its last calibration has a fingerprint that no longer matches.
|
||||
let calibrated = record
|
||||
.and_then(|r| r.placement(*number))
|
||||
.and_then(|p| catalog.and_then(|c| c.get(&p.item)))
|
||||
.and_then(|entry| {
|
||||
let cal = entry.item.calibration.as_ref()?;
|
||||
let enough = cal.n_examinees.unwrap_or(0) >= t.calibrated_n;
|
||||
let current = match &cal.fingerprint {
|
||||
Some(recorded) => *recorded == entry.item.fingerprint(),
|
||||
// An older calibration block with no fingerprint cannot be
|
||||
// shown stale, so it is taken at its word.
|
||||
None => true,
|
||||
};
|
||||
Some(enough && current)
|
||||
})
|
||||
.unwrap_or(false);
|
||||
|
||||
flag_item(&mut analysis, t, design.as_ref(), &rows, calibrated);
|
||||
|
||||
p_values.push(p_value);
|
||||
if let Some(r) = rpb {
|
||||
@@ -521,11 +581,13 @@ pub fn analyze(
|
||||
/// * `t` - the thresholds.
|
||||
/// * `design` - the authored expectation, when available.
|
||||
/// * `rows` - the raw responses, for partial-credit detection.
|
||||
/// * `calibrated` - whether that expectation rests on prior data.
|
||||
fn flag_item(
|
||||
a: &mut ItemAnalysis,
|
||||
t: &Thresholds,
|
||||
design: Option<&Design>,
|
||||
rows: &[&crate::responses::Response],
|
||||
calibrated: bool,
|
||||
) {
|
||||
// Discrimination first: it is the finding that changes what you do.
|
||||
match a.point_biserial {
|
||||
@@ -678,31 +740,71 @@ fn flag_item(
|
||||
}
|
||||
}
|
||||
|
||||
// Did the item behave as authored?
|
||||
// Did the item behave as authored? This is the one check whose meaning
|
||||
// depends on where the expectation came from, so it is recorded either way
|
||||
// and flagged only when the expectation had data behind it.
|
||||
if let Some(d) = design {
|
||||
let mut prediction = Prediction {
|
||||
expected_p: d.expected_difficulty,
|
||||
expected_band: d.expected_discrimination.map(|b| b.expected_band()),
|
||||
calibrated,
|
||||
p_error: None,
|
||||
p_within: None,
|
||||
band_hit: None,
|
||||
notes: Vec::new(),
|
||||
};
|
||||
|
||||
if let Some(expected) = d.expected_difficulty {
|
||||
if (expected - a.p_value).abs() > t.design_tolerance {
|
||||
a.flags.push(Flag::DesignMismatch);
|
||||
a.notes.push(format!(
|
||||
"you expected about {:.0}% correct and observed {:.0}%. Worth knowing whether \
|
||||
your model of the students or the item is off.",
|
||||
expected * 100.0,
|
||||
a.p_value * 100.0
|
||||
));
|
||||
let error = a.p_value - expected;
|
||||
let within = error.abs() <= t.design_tolerance;
|
||||
prediction.p_error = Some(error);
|
||||
prediction.p_within = Some(within);
|
||||
if !within {
|
||||
if calibrated {
|
||||
a.flags.push(Flag::DesignMismatch);
|
||||
prediction.notes.push(format!(
|
||||
"this item is calibrated at about {:.0}% correct and came out at {:.0}%. \
|
||||
Something changed: the cohort, the teaching, or the item.",
|
||||
expected * 100.0,
|
||||
a.p_value * 100.0
|
||||
));
|
||||
} else {
|
||||
prediction.notes.push(format!(
|
||||
"you predicted about {:.0}% correct and observed {:.0}%. This is the \
|
||||
first data on the item, so it corrects the prediction rather than \
|
||||
condemning the item.",
|
||||
expected * 100.0,
|
||||
a.p_value * 100.0
|
||||
));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if let (Some(band), Some(r)) = (d.expected_discrimination, a.point_biserial) {
|
||||
let (low, high) = band.expected_band();
|
||||
if r < low || r > high {
|
||||
if !a.flags.contains(&Flag::DesignMismatch) {
|
||||
a.flags.push(Flag::DesignMismatch);
|
||||
let hit = r >= low && r <= high;
|
||||
prediction.band_hit = Some(hit);
|
||||
if !hit {
|
||||
if calibrated {
|
||||
if !a.flags.contains(&Flag::DesignMismatch) {
|
||||
a.flags.push(Flag::DesignMismatch);
|
||||
}
|
||||
prediction.notes.push(format!(
|
||||
"calibrated for {} discrimination ({low:.2} to {high:.2}), observed \
|
||||
{r:.2}.",
|
||||
format!("{band:?}").to_lowercase()
|
||||
));
|
||||
} else {
|
||||
prediction.notes.push(format!(
|
||||
"you predicted {} discrimination ({low:.2} to {high:.2}) and observed \
|
||||
{r:.2}.",
|
||||
format!("{band:?}").to_lowercase()
|
||||
));
|
||||
}
|
||||
a.notes.push(format!(
|
||||
"you expected {} discrimination ({low:.2} to {high:.2}) and observed {r:.2}.",
|
||||
format!("{band:?}").to_lowercase()
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
a.prediction = Some(prediction);
|
||||
}
|
||||
|
||||
a.flags.sort();
|
||||
|
||||
Reference in New Issue
Block a user