Files
coursebank/src/analysis/classical.rs
T
alexm 5ac1e317c0
Pipeline / check (pull_request) Successful in 3m9s
Pipeline / docs (pull_request) Skipped
Pipeline / nightly (pull_request) Skipped
Pipeline / release (pull_request) Skipped
feat: cooked up something fierce
2026-09-26 18:22:25 -04:00

1224 lines
45 KiB
Rust

// SPDX-License-Identifier: Prosperity-3.0.0
// Copyright Scientific Computing Studio
// Source: https://git.scient.ing/education/coursebank
//! Classical item analysis.
//!
//! The corrected point-biserial is the one to look at first. It correlates
//! each student's response with their total score on the other items, which is
//! the only version that answers the question you care about: did students who
//! knew the material get this right? The uncorrected version includes the item in
//! its own criterion and is therefore biased upward, which flatters bad items.
//! A negative corrected point-biserial almost always means the key is wrong or
//! the stem has a second defensible reading, and it is the single most reliable
//! signal in the whole tool.
//!
//! The p-value is difficulty, and on its own it means very little. An item
//! everyone answered correctly is not a bad item; it may be a deliberate anchor.
//! An item everyone missed is only a problem if it also failed to discriminate.
//!
//! Distractor analysis is where poorly worded questions reveal themselves. If
//! the strongest quarter of the class picked distractor B at a higher rate than
//! the key, B is either the better answer or the stem is ambiguous. That is a
//! different diagnosis from "hard," and it calls for rewriting rather than
//! reteaching.
//!
//! One convention worth stating: a blank response counts as incorrect for
//! difficulty and correlations, because on a scored exam it is worth zero and
//! excluding it would make an item that students skipped look easier than it was.
//! The blank rate is reported separately so a widely skipped item is still visible
//! as such.
use std::collections::{BTreeMap, BTreeSet};
use crate::assessment::AssessmentFile;
use crate::catalog::Catalog;
use crate::item::{Design, OptionStat};
use crate::responses::ResponseSet;
use crate::taxonomy::Flag;
/// Cut points for flagging items.
#[derive(Debug, Clone)]
pub struct Thresholds {
/// A p-value above this is "too easy".
pub too_easy: f64,
/// A p-value below this is "too hard", but only when discrimination is also
/// poor: a hard item that separates students is doing its job.
pub too_hard: f64,
/// A point-biserial below this is weak discrimination.
pub low_discrimination: f64,
/// A point-biserial below this is treated as genuinely *negative*, which is
/// the blocking "check your key" finding.
///
/// This is not zero, and the gap matters. In a class of twenty-four the
/// standard error of a point-biserial is around 0.2, so an observed -0.004 is
/// indistinguishable from no relationship at all. Alarming on it would send
/// you hunting for a keying error that is not there, and after a few false
/// alarms the flag stops being read. Values in the dead zone are reported as
/// low discrimination instead, which is what they actually are.
pub negative_discrimination: f64,
/// A selection rate at or below this makes a distractor nonfunctioning.
pub nonfunctioning: f64,
/// How far observed difficulty may drift from the authored expectation.
pub design_tolerance: f64,
/// How many examinees an item's calibration needs before its recorded
/// expectations are treated as evidence rather than as the author's guess.
///
/// Fifty is the point at which the standard error of a proportion near 0.5
/// drops to about 0.07, which is small enough that a quarter-point miss is
/// about the item rather than about the sample.
pub calibrated_n: usize,
/// Fraction of the class in the upper and lower comparison groups. Kelley's
/// 0.27 maximizes the difference between the groups for a normal
/// distribution, and it remains the convention.
pub group_fraction: f64,
/// Below this many examinees, statistics are reported with a caution.
pub small_sample: usize,
}
impl Default for Thresholds {
fn default() -> Thresholds {
Thresholds {
too_easy: 0.95,
too_hard: 0.25,
low_discrimination: 0.15,
negative_discrimination: -0.05,
nonfunctioning: 0.05,
design_tolerance: 0.25,
calibrated_n: 50,
group_fraction: 0.27,
small_sample: 100,
}
}
}
/// Statistics for one option.
#[derive(Debug, Clone)]
pub struct OptionAnalysis {
/// The option letter.
pub letter: String,
/// How many students chose it.
pub count: usize,
/// Fraction of responses that chose it.
pub rate: f64,
/// Correlation between choosing this option and total score on other items.
/// Should be strongly positive for the key and negative for distractors.
pub point_biserial: Option<f64>,
/// Selection rate in the upper group.
pub upper_rate: Option<f64>,
/// Selection rate in the lower group.
pub lower_rate: Option<f64>,
/// Whether this option is keyed.
pub is_key: bool,
/// Mean credit awarded to students who chose it, which exposes partial credit
/// granted during grading.
pub mean_credit: f64,
}
impl OptionAnalysis {
/// Converts to the form stored on an item's calibration block.
pub fn to_option_stat(&self) -> OptionStat {
OptionStat {
selection_rate: Some(self.rate),
point_biserial: self.point_biserial,
upper_group_rate: self.upper_rate,
lower_group_rate: self.lower_rate,
}
}
}
/// Statistics for one item.
#[derive(Debug, Clone)]
pub struct ItemAnalysis {
/// The question number on the form.
pub number: u32,
/// The item's global id, when known.
pub item_ref: Option<String>,
/// The option set administered, when the rows agree on one.
///
/// Within one administration an item has one variant, because a placement's
/// distractors are drawn once and shared by every form — only the printed
/// order differs. `None` means the rows disagreed, which happens when a
/// course opts into drawing distractors per form; the statistics below then
/// describe a mixture and cannot be pooled by option set.
pub variant: Option<String>,
/// How many students the item was administered to.
pub n: usize,
/// How many gave a non-blank response.
pub n_answered: usize,
/// Fraction blank.
pub blank_rate: f64,
/// Proportion correct, counting blanks as incorrect.
pub p_value: f64,
/// Mean credit earned, which differs from `p_value` when partial credit was
/// awarded.
pub mean_credit: f64,
/// Corrected item-total point-biserial. `None` when the item has no variance,
/// which happens whenever everyone answered the same way.
pub point_biserial: Option<f64>,
/// Upper-group minus lower-group proportion correct.
pub discrimination_index: Option<f64>,
/// Proportion correct in the upper group.
pub upper_rate: Option<f64>,
/// Proportion correct in the lower group.
pub lower_rate: Option<f64>,
/// The keyed letters.
pub key: Vec<String>,
/// Per-option statistics, by letter.
pub options: BTreeMap<String, OptionAnalysis>,
/// Machine-detected problems.
pub flags: Vec<Flag>,
/// Human-readable explanations tied to the flags.
pub notes: Vec<String>,
/// How the item behaved against what its author predicted, when the item
/// records a prediction.
pub prediction: Option<Prediction>,
}
/// An authored expectation, checked against what happened.
///
/// Kept apart from [`ItemAnalysis::flags`] on purpose. Before an item has been
/// administered, `design.expected_difficulty` is the author's guess, and a guess
/// that turns out wrong says something about the guess rather than about the
/// item. Flagging it anyway is how a report ends up with thirty
/// `design_mismatch` findings and no way to see the four that matter. So the
/// discrepancy is always recorded here, and it only becomes a
/// [`Flag::DesignMismatch`] once the expectation has data behind it.
#[derive(Debug, Clone)]
pub struct Prediction {
/// The difficulty the author expected.
pub expected_p: Option<f64>,
/// The discrimination band the author expected, as `(low, high)`.
pub expected_band: Option<(f64, f64)>,
/// Whether the expectation rests on a calibration with enough examinees
/// behind it, rather than on the author's judgement alone.
pub calibrated: bool,
/// Signed difficulty error, observed minus expected. Positive means the item
/// was easier than predicted.
pub p_error: Option<f64>,
/// Whether observed difficulty landed inside the tolerance.
pub p_within: Option<bool>,
/// Whether observed discrimination landed inside the expected band.
pub band_hit: Option<bool>,
/// What to say about it, phrased for whichever case applies.
pub notes: Vec<String>,
}
impl ItemAnalysis {
/// Whether the item needs attention before reuse.
pub fn needs_revision(&self) -> bool {
self.flags.iter().any(|f| f.is_blocking())
}
/// The item's most serious flag, for sorting a work queue.
pub fn worst_flag(&self) -> Option<Flag> {
self.flags
.iter()
.copied()
.find(|f| f.is_blocking())
.or_else(|| self.flags.first().copied())
}
}
/// Whole-test reliability.
#[derive(Debug, Clone)]
pub struct Reliability {
/// Number of scored items.
pub n_items: usize,
/// Number of examinees.
pub n_students: usize,
/// Mean total score, in items correct.
pub mean: f64,
/// Standard deviation of total score.
pub sd: f64,
/// KR-20, which is Cronbach's alpha for dichotomous items. `None` when there
/// is too little variance to compute it.
pub alpha: Option<f64>,
/// Standard error of measurement, in the same units as the total score.
pub sem: Option<f64>,
/// Mean p-value across items.
pub mean_p: f64,
/// Mean point-biserial across items that had one.
pub mean_point_biserial: Option<f64>,
}
impl Reliability {
/// A plain-language reading of the alpha value.
///
/// Interpretation bands are worth printing because alpha is routinely
/// over-read: it depends on test length as much as item quality, so a
/// thirty-item classroom exam at 0.6 is unremarkable, not broken.
///
/// # Returns
///
/// A sentence about what the value means for this test.
pub fn interpretation(&self) -> String {
let Some(alpha) = self.alpha else {
return "Reliability could not be computed: there is not enough variation in total \
scores."
.to_string();
};
let band = if alpha >= 0.9 {
"very high, typical of a long standardized test"
} else if alpha >= 0.8 {
"high"
} else if alpha >= 0.7 {
"acceptable for a classroom exam"
} else if alpha >= 0.6 {
"modest, which is common for a short exam in a small class"
} else if alpha >= 0.5 {
"low; treat individual student scores as rough"
} else {
"very low; the total score is not measuring one thing consistently"
};
let mut s = format!("KR-20 is {alpha:.2} ({band}).");
if let Some(sem) = self.sem {
s.push_str(&format!(
" The standard error of measurement is {sem:.1} items, so a student's true score \
is roughly within ±{:.0} items of what they scored.",
sem * 1.96
));
}
if self.n_items < 20 {
s.push_str(
" Alpha rises with test length, so a short test understates how well its items \
work.",
);
}
s
}
}
/// The result of analyzing one administration.
#[derive(Debug, Clone)]
pub struct Analysis {
/// Per-item statistics, in question order.
pub items: Vec<ItemAnalysis>,
/// Whole-test reliability.
pub reliability: Reliability,
/// Cautions about the analysis itself.
pub warnings: Vec<String>,
}
impl Analysis {
/// Items that need revision, worst first.
///
/// # Returns
///
/// References to the flagged items, blocking flags before advisory ones.
pub fn revise_queue(&self) -> Vec<&ItemAnalysis> {
let mut out: Vec<&ItemAnalysis> =
self.items.iter().filter(|i| !i.flags.is_empty()).collect();
out.sort_by(|a, b| {
b.needs_revision()
.cmp(&a.needs_revision())
.then_with(|| {
a.point_biserial
.unwrap_or(1.0)
.partial_cmp(&b.point_biserial.unwrap_or(1.0))
.unwrap_or(std::cmp::Ordering::Equal)
})
.then_with(|| a.number.cmp(&b.number))
});
out
}
}
/// Runs item analysis over one administration.
///
/// # Arguments
///
/// * `set` - the responses. Only scored, undropped items are analyzed.
/// * `t` - flag thresholds.
/// * `record` - the assessment record, for keys and item references.
/// * `catalog` - the loaded course, for authored expectations.
///
/// # Returns
///
/// The analysis, including cautions when the sample is too small to trust.
pub fn analyze(
set: &ResponseSet,
t: &Thresholds,
record: Option<&AssessmentFile>,
catalog: Option<&Catalog>,
) -> Analysis {
let matrix = set.matrix(false);
let mut warnings = Vec::new();
if !matrix.is_analyzable() {
return Analysis {
items: Vec::new(),
reliability: Reliability {
n_items: 0,
n_students: 0,
mean: 0.0,
sd: 0.0,
alpha: None,
sem: None,
mean_p: 0.0,
mean_point_biserial: None,
},
warnings: vec![
"there are no scored responses to analyze; check that ingest matched the \
assessment record"
.to_string(),
],
};
}
let n_students = matrix.n_students();
if n_students < t.small_sample {
warnings.push(format!(
"these statistics come from {n_students} examinees. Point-biserials from a class this \
size have a standard error near {:.2}, so treat anything between -0.2 and 0.2 as \
indistinguishable from zero and pool several administrations before retiring an item",
1.0 / ((n_students as f64 - 1.0).max(1.0)).sqrt()
));
}
// Blanks count as incorrect for scoring purposes.
let coded: Vec<Vec<f64>> = matrix
.coded
.iter()
.map(|row| row.iter().map(|c| c.unwrap_or(0) as f64).collect())
.collect();
let totals: Vec<f64> = coded.iter().map(|row| row.iter().sum()).collect();
// Upper and lower groups by total score, Kelley's fraction.
let group_size = ((n_students as f64 * t.group_fraction).round() as usize).max(1);
let mut order: Vec<usize> = (0..n_students).collect();
order.sort_by(|a, b| {
totals[*b]
.partial_cmp(&totals[*a])
.unwrap_or(std::cmp::Ordering::Equal)
.then_with(|| matrix.students[*a].cmp(&matrix.students[*b]))
});
let upper: BTreeSet<usize> = order.iter().take(group_size).copied().collect();
let lower: BTreeSet<usize> = order.iter().rev().take(group_size).copied().collect();
let groups_usable = n_students >= 6 && upper.is_disjoint(&lower);
if !groups_usable && n_students > 0 {
warnings.push(format!(
"with {n_students} examinees the upper and lower comparison groups would overlap, so \
the discrimination index is omitted; the point-biserial uses the whole class and is \
reported instead"
));
}
// A drop changes every student's percentage, so the two ways to get it wrong
// are worth saying out loud. Both are silent otherwise: the numbers simply
// come out different from the platform's.
if let Some(record) = record {
for placement in record.items.iter().filter(|p| p.dropped) {
let rows = set.for_item(placement.number);
if rows.is_empty() {
continue;
}
let all_credited = rows.iter().all(|r| r.credit >= 0.999);
if placement.dropped_with_credit() && !all_credited {
let short = rows.iter().filter(|r| r.credit < 0.999).count();
warnings.push(format!(
"question {} is marked `dropped_as: full_credit`, but {short} of {} responses \
carry less than full credit. Either the platform was not regraded or the \
export predates the regrade; until one of those is fixed this report's \
percentages will sit below the grade of record",
placement.number,
rows.len()
));
}
if !placement.dropped_with_credit() && all_credited {
warnings.push(format!(
"question {} is dropped and every response carries full credit, which is what \
crediting every option on the platform looks like. It is being removed from \
the denominator here, so this report will read slightly lower than the \
platform. Set `dropped_as: full_credit` if the platform kept the point",
placement.number
));
}
}
}
let mut items = Vec::new();
let mut p_values = Vec::new();
let mut rpbs = Vec::new();
for (j, number) in matrix.items.iter().enumerate() {
let rows = set.for_item(*number);
let x: Vec<f64> = coded.iter().map(|r| r[j]).collect();
let rest: Vec<f64> = totals
.iter()
.zip(x.iter())
.map(|(total, xi)| total - xi)
.collect();
let n = matrix.n_students();
let n_answered = matrix.coded.iter().filter(|r| r[j].is_some()).count();
let p_value = mean(&x);
let rpb = correlation(&x, &rest);
let (upper_rate, lower_rate, discrimination_index) = if groups_usable {
let u = mean_of(&x, &upper);
let l = mean_of(&x, &lower);
(Some(u), Some(l), Some(u - l))
} else {
(None, None, None)
};
// Keys: prefer the record, fall back to what the data says earned credit.
let key: Vec<String> = record
.and_then(|r| r.placement(*number))
.map(|p| p.key.clone())
.filter(|k| !k.is_empty())
.unwrap_or_else(|| infer_key(&rows));
let mean_credit = if rows.is_empty() {
0.0
} else {
rows.iter().map(|r| r.credit).sum::<f64>() / rows.len() as f64
};
// Per-option statistics.
let student_index: BTreeMap<&str, usize> = matrix
.students
.iter()
.enumerate()
.map(|(i, s)| (s.as_str(), i))
.collect();
let mut chose: BTreeMap<String, Vec<usize>> = BTreeMap::new();
let mut credits: BTreeMap<String, Vec<f64>> = BTreeMap::new();
let mut blank = 0usize;
for r in &rows {
if r.chosen().is_empty() {
blank += 1;
continue;
}
// A multiple-response item is credited to the joined set, so that
// "chose A and C" is one response pattern rather than two options.
let label = r.chosen().join("+");
if let Some(&si) = student_index.get(r.student_key.as_str()) {
chose.entry(label.clone()).or_default().push(si);
}
credits.entry(label).or_default().push(r.credit);
}
// Every declared option appears, even one nobody chose: a rate of zero is
// the finding.
let mut letters: BTreeSet<String> = chose.keys().cloned().collect();
if let (Some(rec), Some(cat)) = (record, catalog) {
if let Some(p) = rec.placement(*number) {
if let Some(entry) = cat.get(&p.item) {
for o in &entry.item.options {
letters.insert(o.id.clone());
}
}
}
}
let responded = rows.len().max(1);
let mut options = BTreeMap::new();
for letter in letters {
let indices = chose.get(&letter).cloned().unwrap_or_default();
let count = indices.len();
let indicator: Vec<f64> = (0..n)
.map(|i| if indices.contains(&i) { 1.0 } else { 0.0 })
.collect();
let set_indices: BTreeSet<usize> = indices.iter().copied().collect();
let cr = credits.get(&letter).cloned().unwrap_or_default();
options.insert(
letter.clone(),
OptionAnalysis {
letter: letter.clone(),
count,
rate: count as f64 / responded as f64,
point_biserial: correlation(&indicator, &rest),
upper_rate: if groups_usable {
Some(mean_of(&indicator, &upper))
} else {
None
},
lower_rate: if groups_usable {
Some(mean_of(&indicator, &lower))
} else {
None
},
is_key: key.contains(&letter) || (key.len() > 1 && letter == key.join("+")),
mean_credit: if cr.is_empty() {
0.0
} else {
cr.iter().sum::<f64>() / cr.len() as f64
},
},
);
let _ = set_indices;
}
let design = record
.and_then(|r| r.placement(*number))
.and_then(|p| catalog.and_then(|c| c.get(&p.item)))
.and_then(|e| e.item.design.clone());
let mut analysis = ItemAnalysis {
number: *number,
item_ref: record
.and_then(|r| r.placement(*number))
.map(|p| p.item.clone()),
variant: one_variant(set, *number),
n,
n_answered,
blank_rate: blank as f64 / responded as f64,
p_value,
mean_credit,
point_biserial: rpb,
discrimination_index,
upper_rate,
lower_rate,
key,
options,
flags: Vec::new(),
notes: Vec::new(),
prediction: None,
};
// Whether the authored expectation is evidence or a guess. An item that
// has never been administered has no calibration block, and one edited
// since its last calibration has a fingerprint that no longer matches.
let calibrated = record
.and_then(|r| r.placement(*number))
.and_then(|p| catalog.and_then(|c| c.get(&p.item)))
.and_then(|entry| {
let cal = entry.item.calibration.as_ref()?;
let enough = cal.n_examinees.unwrap_or(0) >= t.calibrated_n;
let current = match &cal.fingerprint {
Some(recorded) => *recorded == entry.item.fingerprint(),
// An older calibration block with no fingerprint cannot be
// shown stale, so it is taken at its word.
None => true,
};
Some(enough && current)
})
.unwrap_or(false);
flag_item(&mut analysis, t, design.as_ref(), &rows, calibrated);
p_values.push(p_value);
if let Some(r) = rpb {
rpbs.push(r);
}
items.push(analysis);
}
let reliability = reliability(&coded, &totals, &p_values, &rpbs);
Analysis {
items,
reliability,
warnings,
}
}
/// Applies the flag rules to one item.
///
/// # Arguments
///
/// * `a` - the item analysis, updated in place.
/// * `t` - the thresholds.
/// * `design` - the authored expectation, when available.
/// * `rows` - the raw responses, for partial-credit detection.
/// * `calibrated` - whether that expectation rests on prior data.
fn flag_item(
a: &mut ItemAnalysis,
t: &Thresholds,
design: Option<&Design>,
rows: &[&crate::responses::Response],
calibrated: bool,
) {
// Discrimination first: it is the finding that changes what you do.
match a.point_biserial {
Some(r) if r < t.negative_discrimination => {
a.flags.push(Flag::NegativeDiscrimination);
a.notes.push(format!(
"students who did better overall did worse on this item (r = {r:.2}). Check the \
key before anything else."
));
}
Some(r) if r < t.low_discrimination => {
a.flags.push(Flag::LowDiscrimination);
a.notes.push(if r < 0.0 {
format!(
"this item did not separate stronger from weaker students (r = {r:.2}, which \
is indistinguishable from zero at this sample size)."
)
} else {
format!("this item barely separates stronger from weaker students (r = {r:.2}).")
});
}
None => {
a.notes.push(
"every student responded the same way, so this item has no variance and no \
correlation can be computed."
.to_string(),
);
}
_ => {}
}
if a.p_value > t.too_easy {
a.flags.push(Flag::TooEasy);
a.notes.push(format!(
"{:.0}% answered correctly. Fine as an opening anchor, but it carries little \
information about who knows what.",
a.p_value * 100.0
));
}
if a.p_value < t.too_hard {
// Hard and discriminating is a good item, not a broken one.
let discriminates = a
.point_biserial
.map(|r| r >= t.low_discrimination)
.unwrap_or(false);
if !discriminates {
a.flags.push(Flag::TooHard);
a.notes.push(format!(
"only {:.0}% answered correctly, and the item did not separate students. That \
pattern usually means the stem is unclear or a prerequisite is missing, not that \
the content is hard.",
a.p_value * 100.0
));
}
}
// Distractor analysis: the real source of "poorly worded question" findings.
let key_rpb = a
.options
.values()
.filter(|o| o.is_key)
.filter_map(|o| o.point_biserial)
.fold(f64::NEG_INFINITY, f64::max);
for o in a.options.values() {
if o.is_key {
continue;
}
if let Some(r) = o.point_biserial {
if key_rpb.is_finite() && r > key_rpb && o.rate >= 0.1 {
a.flags.push(Flag::DistractorOutperformsKey);
a.notes.push(format!(
"option {} correlates with overall performance better than the key does \
(r = {r:.2} against {key_rpb:.2}), and {:.0}% chose it. Either it is the \
better answer or the stem admits it.",
o.letter,
o.rate * 100.0
));
}
}
if let (Some(upper), Some(key_upper)) = (
o.upper_rate,
a.options
.values()
.filter(|k| k.is_key)
.filter_map(|k| k.upper_rate)
.fold(None, |acc: Option<f64>, v| {
Some(acc.map_or(v, |a| a.max(v)))
}),
) {
if upper > key_upper && upper >= 0.25 {
a.flags.push(Flag::KeyUnderperforms);
a.notes.push(format!(
"the strongest students chose option {} more often than the key ({:.0}% \
against {:.0}%). That split is the signature of two defensible readings.",
o.letter,
upper * 100.0,
key_upper * 100.0
));
}
}
if o.rate <= t.nonfunctioning {
a.flags.push(Flag::NonfunctioningDistractor);
a.notes.push(format!(
"option {} was chosen by {:.0}% of students, so it is not doing any work. \
Replace it with a plausible error students actually make.",
o.letter,
o.rate * 100.0
));
}
}
// Partial credit awarded to a non-key option is a grading-time admission of
// ambiguity, and it is the strongest such signal available.
let ambiguous = a
.options
.values()
.any(|o| !o.is_key && o.mean_credit > 0.0 && o.count > 0);
if ambiguous {
a.flags.push(Flag::Ambiguous);
let letters: Vec<String> = a
.options
.values()
.filter(|o| !o.is_key && o.mean_credit > 0.0 && o.count > 0)
.map(|o| format!("{} ({:.0}%)", o.letter, o.mean_credit * 100.0))
.collect();
a.notes.push(format!(
"partial credit was awarded at grading time to {}, which records a decision that the \
item admitted more than one reading. Rewrite the stem rather than re-deciding this \
every term.",
letters.join(", ")
));
}
// Rapid guessing, when the platform reported response times.
let times: Vec<f64> = rows
.iter()
.filter_map(|r| r.response_time_seconds)
.collect();
if times.len() >= 5 {
let rapid = times.iter().filter(|t| **t < 5.0).count() as f64 / times.len() as f64;
if rapid > 0.15 {
a.flags.push(Flag::HighRapidGuess);
a.notes.push(format!(
"{:.0}% of responses arrived in under five seconds, which is faster than the stem \
can be read. That is usually about the item's position on the form or time \
pressure, not the item.",
rapid * 100.0
));
}
}
// Did the item behave as authored? This is the one check whose meaning
// depends on where the expectation came from, so it is recorded either way
// and flagged only when the expectation had data behind it.
if let Some(d) = design {
let mut prediction = Prediction {
expected_p: d.expected_difficulty,
expected_band: d.expected_discrimination.map(|b| b.expected_band()),
calibrated,
p_error: None,
p_within: None,
band_hit: None,
notes: Vec::new(),
};
if let Some(expected) = d.expected_difficulty {
let error = a.p_value - expected;
let within = error.abs() <= t.design_tolerance;
prediction.p_error = Some(error);
prediction.p_within = Some(within);
if !within {
if calibrated {
a.flags.push(Flag::DesignMismatch);
prediction.notes.push(format!(
"this item is calibrated at about {:.0}% correct and came out at {:.0}%. \
Something changed: the cohort, the teaching, or the item.",
expected * 100.0,
a.p_value * 100.0
));
} else {
prediction.notes.push(format!(
"you predicted about {:.0}% correct and observed {:.0}%. This is the \
first data on the item, so it corrects the prediction rather than \
condemning the item.",
expected * 100.0,
a.p_value * 100.0
));
}
}
}
if let (Some(band), Some(r)) = (d.expected_discrimination, a.point_biserial) {
let (low, high) = band.expected_band();
let hit = r >= low && r <= high;
prediction.band_hit = Some(hit);
if !hit {
if calibrated {
if !a.flags.contains(&Flag::DesignMismatch) {
a.flags.push(Flag::DesignMismatch);
}
prediction.notes.push(format!(
"calibrated for {} discrimination ({low:.2} to {high:.2}), observed \
{r:.2}.",
format!("{band:?}").to_lowercase()
));
} else {
prediction.notes.push(format!(
"you predicted {} discrimination ({low:.2} to {high:.2}) and observed \
{r:.2}.",
format!("{band:?}").to_lowercase()
));
}
}
}
a.prediction = Some(prediction);
}
a.flags.sort();
a.flags.dedup();
}
/// Computes whole-test reliability.
///
/// # Arguments
///
/// * `coded` - the 0/1 response matrix.
/// * `totals` - per-student totals.
/// * `p_values` - per-item p-values.
/// * `rpbs` - per-item point-biserials that could be computed.
///
/// # Returns
///
/// The reliability summary.
fn reliability(coded: &[Vec<f64>], totals: &[f64], p_values: &[f64], rpbs: &[f64]) -> Reliability {
let n_students = coded.len();
let n_items = coded.first().map(|r| r.len()).unwrap_or(0);
// Named distinctly from the `mean` and `sd` helpers: binding `let mean =
// mean(totals)` shadows the function for the rest of the scope, which then
// makes the later `mean(p_values)` a call on an f64.
let mean_total = mean(totals);
let sd_total = sd(totals);
// KR-20. The variance terms use the population form, which is the convention
// for this coefficient and matches what other packages report.
let alpha = if n_items > 1 && sd_total > 0.0 {
let sum_pq: f64 = p_values.iter().map(|p| p * (1.0 - p)).sum();
let k = n_items as f64;
let variance = sd_total * sd_total;
Some((k / (k - 1.0)) * (1.0 - sum_pq / variance))
} else {
None
};
let sem = alpha.map(|a| sd_total * (1.0 - a).max(0.0).sqrt());
Reliability {
n_items,
n_students,
mean: mean_total,
sd: sd_total,
alpha,
sem,
mean_p: mean(p_values),
mean_point_biserial: if rpbs.is_empty() {
None
} else {
Some(mean(rpbs))
},
}
}
/// Infers the key from which options earned full credit.
///
/// Used when no assessment record is available, so that an export can be analyzed
/// before its record is written.
///
/// # Arguments
///
/// * `rows` - the responses for one item.
///
/// # Returns
///
/// The letters that appear on full-credit responses.
/// The single variant every row for one question names, if they agree.
///
/// Disagreement is not an error, it is a fact about the administration: a course
/// that draws distractors per form has two option sets under one question
/// number, and no pooled statistic describes both. Returning `None` is what
/// keeps the per-variant records from claiming otherwise.
fn one_variant(set: &ResponseSet, number: u32) -> Option<String> {
let mut seen: Option<&str> = None;
for row in set.rows.iter().filter(|r| r.item_number == number) {
let variant = row.variant.as_deref()?;
match seen {
None => seen = Some(variant),
Some(first) if first == variant => {}
Some(_) => return None,
}
}
seen.map(str::to_string)
}
fn infer_key(rows: &[&crate::responses::Response]) -> Vec<String> {
let mut out: BTreeSet<String> = BTreeSet::new();
for r in rows {
if r.credit >= 0.999 {
for letter in r.chosen() {
out.insert(letter.clone());
}
}
}
out.into_iter().collect()
}
/// The arithmetic mean, zero for an empty slice.
fn mean(v: &[f64]) -> f64 {
if v.is_empty() {
0.0
} else {
v.iter().sum::<f64>() / v.len() as f64
}
}
/// The population standard deviation.
fn sd(v: &[f64]) -> f64 {
if v.len() < 2 {
return 0.0;
}
let m = mean(v);
(v.iter().map(|x| (x - m) * (x - m)).sum::<f64>() / v.len() as f64).sqrt()
}
/// The mean of the entries at the given indices.
fn mean_of(v: &[f64], indices: &BTreeSet<usize>) -> f64 {
if indices.is_empty() {
return 0.0;
}
indices.iter().map(|i| v[*i]).sum::<f64>() / indices.len() as f64
}
/// The Pearson correlation of two equal-length vectors.
///
/// # Arguments
///
/// * `x` - the first vector.
/// * `y` - the second vector.
///
/// # Returns
///
/// The correlation, or `None` when either vector has no variance. Returning
/// `None` rather than a NaN matters: an item everyone answered correctly has no
/// correlation, and that is a meaningful result to report rather than a number to
/// propagate.
pub fn correlation(x: &[f64], y: &[f64]) -> Option<f64> {
if x.len() != y.len() || x.len() < 2 {
return None;
}
let mx = mean(x);
let my = mean(y);
let mut sxy = 0.0;
let mut sxx = 0.0;
let mut syy = 0.0;
for (xi, yi) in x.iter().zip(y.iter()) {
let dx = xi - mx;
let dy = yi - my;
sxy += dx * dy;
sxx += dx * dx;
syy += dy * dy;
}
if sxx <= f64::EPSILON || syy <= f64::EPSILON {
return None;
}
Some(sxy / (sxx * syy).sqrt())
}
#[cfg(test)]
mod tests {
use super::*;
use crate::responses::Response;
fn resp(student: &str, number: u32, letter: &str, credit: f64) -> Response {
Response {
administration_id: "C/T/a".into(),
course: "C".into(),
term: "T".into(),
assessment_id: "a".into(),
date: None,
form: None,
form_position: None,
student_key: student.into(),
sid: None,
name: None,
email: None,
section: None,
item_number: number,
item_ref: None,
item_version: None,
variant: None,
selected: if letter.is_empty() {
vec![]
} else {
vec![letter.to_string()]
},
selected_source: vec![],
eliminated: vec![],
eliminated_source: vec![],
correct: Some(credit >= 0.999),
credit,
points_possible: 1.0,
score: credit,
response_time_seconds: None,
level: None,
learning_targets: vec![],
topics: vec![],
bonus: false,
dropped: false,
dropped_full_credit: false,
}
}
/// Twelve students; item 1 discriminates, item 2 is keyed backwards.
///
/// Item 2 is only *partly* reversed, and that is load-bearing rather than
/// sloppy. The corrected point-biserial scores each item against the total of
/// the *other* items, and with four items — one of them unanimous — item 2 is
/// most of item 1's rest score. Make item 2 an exact complement of item 1 and
/// `item1 + item2 == 1` for every student, so the total collapses to
/// `2 + item4`, carries no ability signal at all, and both items come out at
/// r = -0.71. The fixture then contradicts itself: the item it calls good is
/// flagged for negative discrimination, and the reversed key is negative only
/// because it mirrors item 1 rather than because it is miskeyed.
fn sample() -> ResponseSet {
let mut set = ResponseSet::new();
for i in 0..12 {
let strong = i < 6;
let s = format!("s{i:02}");
// Item 1: strong students right, weak wrong.
set.rows.push(resp(
&s,
1,
if strong { "A" } else { "B" },
if strong { 1.0 } else { 0.0 },
));
// Item 2: weaker students do better on it, which is what a keying
// error looks like. Half the strong group and two thirds of the weak
// group get it, so it is reversed without mirroring item 1.
let missed = if strong { i < 3 } else { i < 8 };
set.rows.push(resp(
&s,
2,
if missed { "C" } else { "D" },
if missed { 0.0 } else { 1.0 },
));
// Item 3: everyone correct.
set.rows.push(resp(&s, 3, "A", 1.0));
// Item 4: a second discriminating item, so that removing any one item
// still leaves a rest score that tracks ability.
set.rows.push(resp(
&s,
4,
if strong { "A" } else { "B" },
if strong { 1.0 } else { 0.0 },
));
}
set
}
#[test]
fn correlation_returns_none_without_variance() {
assert_eq!(correlation(&[1.0, 1.0, 1.0], &[1.0, 2.0, 3.0]), None);
assert_eq!(correlation(&[1.0], &[1.0]), None);
let r = correlation(&[1.0, 2.0, 3.0], &[2.0, 4.0, 6.0]).unwrap();
assert!((r - 1.0).abs() < 1e-12, "perfect correlation is 1, got {r}");
let r = correlation(&[1.0, 2.0, 3.0], &[3.0, 2.0, 1.0]).unwrap();
assert!((r + 1.0).abs() < 1e-12);
}
#[test]
fn flags_a_reversed_key_as_negative_discrimination() {
let set = sample();
let a = analyze(&set, &Thresholds::default(), None, None);
let item2 = a.items.iter().find(|i| i.number == 2).unwrap();
assert!(
item2.point_biserial.unwrap() < 0.0,
"got {:?}",
item2.point_biserial
);
assert!(item2.flags.contains(&Flag::NegativeDiscrimination));
assert!(item2.needs_revision());
}
#[test]
fn a_good_item_is_not_flagged_for_discrimination() {
let set = sample();
let a = analyze(&set, &Thresholds::default(), None, None);
let item1 = a.items.iter().find(|i| i.number == 1).unwrap();
assert!(item1.point_biserial.unwrap() > 0.5);
assert!(!item1.flags.contains(&Flag::NegativeDiscrimination));
assert!(!item1.flags.contains(&Flag::LowDiscrimination));
}
#[test]
fn a_unanimous_item_has_no_correlation_and_is_flagged_easy() {
let set = sample();
let a = analyze(&set, &Thresholds::default(), None, None);
let item3 = a.items.iter().find(|i| i.number == 3).unwrap();
assert_eq!(item3.p_value, 1.0);
assert_eq!(item3.point_biserial, None, "no variance, so no correlation");
assert!(item3.flags.contains(&Flag::TooEasy));
}
#[test]
fn blank_responses_count_as_incorrect_but_are_reported() {
let mut set = ResponseSet::new();
for i in 0..10 {
let s = format!("s{i}");
let letter = if i < 5 { "A" } else { "" };
set.rows
.push(resp(&s, 1, letter, if i < 5 { 1.0 } else { 0.0 }));
set.rows
.push(resp(&s, 2, "A", if i < 7 { 1.0 } else { 0.0 }));
}
let a = analyze(&set, &Thresholds::default(), None, None);
let item1 = a.items.iter().find(|i| i.number == 1).unwrap();
assert_eq!(item1.p_value, 0.5);
assert_eq!(item1.blank_rate, 0.5);
assert_eq!(
item1.n_answered, 10,
"a blank is still an administered item"
);
}
#[test]
fn partial_credit_to_a_distractor_flags_ambiguity() {
let mut set = ResponseSet::new();
for i in 0..10 {
let s = format!("s{i}");
if i < 5 {
set.rows.push(resp(&s, 1, "A", 1.0));
} else {
// B earned two thirds of the points at grading time.
set.rows.push(resp(&s, 1, "B", 0.667));
}
set.rows
.push(resp(&s, 2, "A", if i % 2 == 0 { 1.0 } else { 0.0 }));
}
let a = analyze(&set, &Thresholds::default(), None, None);
let item1 = a.items.iter().find(|i| i.number == 1).unwrap();
assert!(item1.flags.contains(&Flag::Ambiguous), "{:?}", item1.flags);
assert!(item1.notes.iter().any(|n| n.contains("partial credit")));
}
#[test]
fn reliability_is_computed_and_explained() {
let set = sample();
let a = analyze(&set, &Thresholds::default(), None, None);
assert_eq!(a.reliability.n_items, 4);
assert_eq!(a.reliability.n_students, 12);
assert!(a.reliability.alpha.is_some());
let text = a.reliability.interpretation();
assert!(text.contains("KR-20"));
// A four-item test must carry the length caveat.
assert!(text.contains("test length"));
}
#[test]
fn small_samples_get_a_caution() {
let set = sample();
let a = analyze(&set, &Thresholds::default(), None, None);
assert!(a.warnings.iter().any(|w| w.contains("examinees")));
}
#[test]
fn empty_input_does_not_panic() {
let a = analyze(&ResponseSet::new(), &Thresholds::default(), None, None);
assert!(a.items.is_empty());
assert_eq!(a.reliability.alpha, None);
assert!(!a.warnings.is_empty());
}
#[test]
fn the_revise_queue_puts_blocking_flags_first() {
let set = sample();
let a = analyze(&set, &Thresholds::default(), None, None);
let queue = a.revise_queue();
assert!(!queue.is_empty());
assert_eq!(queue[0].number, 2, "the reversed key comes first");
}
}