// SPDX-License-Identifier: Prosperity-3.0.0 // Copyright Scientific Computing Studio // Source: https://git.scient.ing/education/coursebank //! Classical item analysis. //! //! The corrected point-biserial is the one to look at first. It correlates //! each student's response with their total score on the other items, which is //! the only version that answers the question you care about: did students who //! knew the material get this right? The uncorrected version includes the item in //! its own criterion and is therefore biased upward, which flatters bad items. //! A negative corrected point-biserial almost always means the key is wrong or //! the stem has a second defensible reading, and it is the single most reliable //! signal in the whole tool. //! //! The p-value is difficulty, and on its own it means very little. An item //! everyone answered correctly is not a bad item; it may be a deliberate anchor. //! An item everyone missed is only a problem if it also failed to discriminate. //! //! Distractor analysis is where poorly worded questions reveal themselves. If //! the strongest quarter of the class picked distractor B at a higher rate than //! the key, B is either the better answer or the stem is ambiguous. That is a //! different diagnosis from "hard," and it calls for rewriting rather than //! reteaching. //! //! One convention worth stating: a blank response counts as incorrect for //! difficulty and correlations, because on a scored exam it is worth zero and //! excluding it would make an item that students skipped look easier than it was. //! The blank rate is reported separately so a widely skipped item is still visible //! as such. use std::collections::{BTreeMap, BTreeSet}; use crate::assessment::AssessmentFile; use crate::catalog::Catalog; use crate::item::{Design, OptionStat}; use crate::responses::ResponseSet; use crate::taxonomy::Flag; /// Cut points for flagging items. #[derive(Debug, Clone)] pub struct Thresholds { /// A p-value above this is "too easy". pub too_easy: f64, /// A p-value below this is "too hard", but only when discrimination is also /// poor: a hard item that separates students is doing its job. pub too_hard: f64, /// A point-biserial below this is weak discrimination. pub low_discrimination: f64, /// A point-biserial below this is treated as genuinely *negative*, which is /// the blocking "check your key" finding. /// /// This is not zero, and the gap matters. In a class of twenty-four the /// standard error of a point-biserial is around 0.2, so an observed -0.004 is /// indistinguishable from no relationship at all. Alarming on it would send /// you hunting for a keying error that is not there, and after a few false /// alarms the flag stops being read. Values in the dead zone are reported as /// low discrimination instead, which is what they actually are. pub negative_discrimination: f64, /// A selection rate at or below this makes a distractor nonfunctioning. pub nonfunctioning: f64, /// How far observed difficulty may drift from the authored expectation. pub design_tolerance: f64, /// How many examinees an item's calibration needs before its recorded /// expectations are treated as evidence rather than as the author's guess. /// /// Fifty is the point at which the standard error of a proportion near 0.5 /// drops to about 0.07, which is small enough that a quarter-point miss is /// about the item rather than about the sample. pub calibrated_n: usize, /// Fraction of the class in the upper and lower comparison groups. Kelley's /// 0.27 maximizes the difference between the groups for a normal /// distribution, and it remains the convention. pub group_fraction: f64, /// Below this many examinees, statistics are reported with a caution. pub small_sample: usize, } impl Default for Thresholds { fn default() -> Thresholds { Thresholds { too_easy: 0.95, too_hard: 0.25, low_discrimination: 0.15, negative_discrimination: -0.05, nonfunctioning: 0.05, design_tolerance: 0.25, calibrated_n: 50, group_fraction: 0.27, small_sample: 100, } } } /// Statistics for one option. #[derive(Debug, Clone)] pub struct OptionAnalysis { /// The option letter. pub letter: String, /// How many students chose it. pub count: usize, /// Fraction of responses that chose it. pub rate: f64, /// Correlation between choosing this option and total score on other items. /// Should be strongly positive for the key and negative for distractors. pub point_biserial: Option, /// Selection rate in the upper group. pub upper_rate: Option, /// Selection rate in the lower group. pub lower_rate: Option, /// Whether this option is keyed. pub is_key: bool, /// Mean credit awarded to students who chose it, which exposes partial credit /// granted during grading. pub mean_credit: f64, } impl OptionAnalysis { /// Converts to the form stored on an item's calibration block. pub fn to_option_stat(&self) -> OptionStat { OptionStat { selection_rate: Some(self.rate), point_biserial: self.point_biserial, upper_group_rate: self.upper_rate, lower_group_rate: self.lower_rate, } } } /// Statistics for one item. #[derive(Debug, Clone)] pub struct ItemAnalysis { /// The question number on the form. pub number: u32, /// The item's global id, when known. pub item_ref: Option, /// The option set administered, when the rows agree on one. /// /// Within one administration an item has one variant, because a placement's /// distractors are drawn once and shared by every form — only the printed /// order differs. `None` means the rows disagreed, which happens when a /// course opts into drawing distractors per form; the statistics below then /// describe a mixture and cannot be pooled by option set. pub variant: Option, /// How many students the item was administered to. pub n: usize, /// How many gave a non-blank response. pub n_answered: usize, /// Fraction blank. pub blank_rate: f64, /// Proportion correct, counting blanks as incorrect. pub p_value: f64, /// Mean credit earned, which differs from `p_value` when partial credit was /// awarded. pub mean_credit: f64, /// Corrected item-total point-biserial. `None` when the item has no variance, /// which happens whenever everyone answered the same way. pub point_biserial: Option, /// Upper-group minus lower-group proportion correct. pub discrimination_index: Option, /// Proportion correct in the upper group. pub upper_rate: Option, /// Proportion correct in the lower group. pub lower_rate: Option, /// The keyed letters. pub key: Vec, /// Per-option statistics, by letter. pub options: BTreeMap, /// Machine-detected problems. pub flags: Vec, /// Human-readable explanations tied to the flags. pub notes: Vec, /// How the item behaved against what its author predicted, when the item /// records a prediction. pub prediction: Option, } /// An authored expectation, checked against what happened. /// /// Kept apart from [`ItemAnalysis::flags`] on purpose. Before an item has been /// administered, `design.expected_difficulty` is the author's guess, and a guess /// that turns out wrong says something about the guess rather than about the /// item. Flagging it anyway is how a report ends up with thirty /// `design_mismatch` findings and no way to see the four that matter. So the /// discrepancy is always recorded here, and it only becomes a /// [`Flag::DesignMismatch`] once the expectation has data behind it. #[derive(Debug, Clone)] pub struct Prediction { /// The difficulty the author expected. pub expected_p: Option, /// The discrimination band the author expected, as `(low, high)`. pub expected_band: Option<(f64, f64)>, /// Whether the expectation rests on a calibration with enough examinees /// behind it, rather than on the author's judgement alone. pub calibrated: bool, /// Signed difficulty error, observed minus expected. Positive means the item /// was easier than predicted. pub p_error: Option, /// Whether observed difficulty landed inside the tolerance. pub p_within: Option, /// Whether observed discrimination landed inside the expected band. pub band_hit: Option, /// What to say about it, phrased for whichever case applies. pub notes: Vec, } impl ItemAnalysis { /// Whether the item needs attention before reuse. pub fn needs_revision(&self) -> bool { self.flags.iter().any(|f| f.is_blocking()) } /// The item's most serious flag, for sorting a work queue. pub fn worst_flag(&self) -> Option { self.flags .iter() .copied() .find(|f| f.is_blocking()) .or_else(|| self.flags.first().copied()) } } /// Whole-test reliability. #[derive(Debug, Clone)] pub struct Reliability { /// Number of scored items. pub n_items: usize, /// Number of examinees. pub n_students: usize, /// Mean total score, in items correct. pub mean: f64, /// Standard deviation of total score. pub sd: f64, /// KR-20, which is Cronbach's alpha for dichotomous items. `None` when there /// is too little variance to compute it. pub alpha: Option, /// Standard error of measurement, in the same units as the total score. pub sem: Option, /// Mean p-value across items. pub mean_p: f64, /// Mean point-biserial across items that had one. pub mean_point_biserial: Option, } impl Reliability { /// A plain-language reading of the alpha value. /// /// Interpretation bands are worth printing because alpha is routinely /// over-read: it depends on test length as much as item quality, so a /// thirty-item classroom exam at 0.6 is unremarkable, not broken. /// /// # Returns /// /// A sentence about what the value means for this test. pub fn interpretation(&self) -> String { let Some(alpha) = self.alpha else { return "Reliability could not be computed: there is not enough variation in total \ scores." .to_string(); }; let band = if alpha >= 0.9 { "very high, typical of a long standardized test" } else if alpha >= 0.8 { "high" } else if alpha >= 0.7 { "acceptable for a classroom exam" } else if alpha >= 0.6 { "modest, which is common for a short exam in a small class" } else if alpha >= 0.5 { "low; treat individual student scores as rough" } else { "very low; the total score is not measuring one thing consistently" }; let mut s = format!("KR-20 is {alpha:.2} ({band})."); if let Some(sem) = self.sem { s.push_str(&format!( " The standard error of measurement is {sem:.1} items, so a student's true score \ is roughly within ±{:.0} items of what they scored.", sem * 1.96 )); } if self.n_items < 20 { s.push_str( " Alpha rises with test length, so a short test understates how well its items \ work.", ); } s } } /// The result of analyzing one administration. #[derive(Debug, Clone)] pub struct Analysis { /// Per-item statistics, in question order. pub items: Vec, /// Whole-test reliability. pub reliability: Reliability, /// Cautions about the analysis itself. pub warnings: Vec, } impl Analysis { /// Items that need revision, worst first. /// /// # Returns /// /// References to the flagged items, blocking flags before advisory ones. pub fn revise_queue(&self) -> Vec<&ItemAnalysis> { let mut out: Vec<&ItemAnalysis> = self.items.iter().filter(|i| !i.flags.is_empty()).collect(); out.sort_by(|a, b| { b.needs_revision() .cmp(&a.needs_revision()) .then_with(|| { a.point_biserial .unwrap_or(1.0) .partial_cmp(&b.point_biserial.unwrap_or(1.0)) .unwrap_or(std::cmp::Ordering::Equal) }) .then_with(|| a.number.cmp(&b.number)) }); out } } /// Runs item analysis over one administration. /// /// # Arguments /// /// * `set` - the responses. Only scored, undropped items are analyzed. /// * `t` - flag thresholds. /// * `record` - the assessment record, for keys and item references. /// * `catalog` - the loaded course, for authored expectations. /// /// # Returns /// /// The analysis, including cautions when the sample is too small to trust. pub fn analyze( set: &ResponseSet, t: &Thresholds, record: Option<&AssessmentFile>, catalog: Option<&Catalog>, ) -> Analysis { let matrix = set.matrix(false); let mut warnings = Vec::new(); if !matrix.is_analyzable() { return Analysis { items: Vec::new(), reliability: Reliability { n_items: 0, n_students: 0, mean: 0.0, sd: 0.0, alpha: None, sem: None, mean_p: 0.0, mean_point_biserial: None, }, warnings: vec![ "there are no scored responses to analyze; check that ingest matched the \ assessment record" .to_string(), ], }; } let n_students = matrix.n_students(); if n_students < t.small_sample { warnings.push(format!( "these statistics come from {n_students} examinees. Point-biserials from a class this \ size have a standard error near {:.2}, so treat anything between -0.2 and 0.2 as \ indistinguishable from zero and pool several administrations before retiring an item", 1.0 / ((n_students as f64 - 1.0).max(1.0)).sqrt() )); } // Blanks count as incorrect for scoring purposes. let coded: Vec> = matrix .coded .iter() .map(|row| row.iter().map(|c| c.unwrap_or(0) as f64).collect()) .collect(); let totals: Vec = coded.iter().map(|row| row.iter().sum()).collect(); // Upper and lower groups by total score, Kelley's fraction. let group_size = ((n_students as f64 * t.group_fraction).round() as usize).max(1); let mut order: Vec = (0..n_students).collect(); order.sort_by(|a, b| { totals[*b] .partial_cmp(&totals[*a]) .unwrap_or(std::cmp::Ordering::Equal) .then_with(|| matrix.students[*a].cmp(&matrix.students[*b])) }); let upper: BTreeSet = order.iter().take(group_size).copied().collect(); let lower: BTreeSet = order.iter().rev().take(group_size).copied().collect(); let groups_usable = n_students >= 6 && upper.is_disjoint(&lower); if !groups_usable && n_students > 0 { warnings.push(format!( "with {n_students} examinees the upper and lower comparison groups would overlap, so \ the discrimination index is omitted; the point-biserial uses the whole class and is \ reported instead" )); } // A drop changes every student's percentage, so the two ways to get it wrong // are worth saying out loud. Both are silent otherwise: the numbers simply // come out different from the platform's. if let Some(record) = record { for placement in record.items.iter().filter(|p| p.dropped) { let rows = set.for_item(placement.number); if rows.is_empty() { continue; } let all_credited = rows.iter().all(|r| r.credit >= 0.999); if placement.dropped_with_credit() && !all_credited { let short = rows.iter().filter(|r| r.credit < 0.999).count(); warnings.push(format!( "question {} is marked `dropped_as: full_credit`, but {short} of {} responses \ carry less than full credit. Either the platform was not regraded or the \ export predates the regrade; until one of those is fixed this report's \ percentages will sit below the grade of record", placement.number, rows.len() )); } if !placement.dropped_with_credit() && all_credited { warnings.push(format!( "question {} is dropped and every response carries full credit, which is what \ crediting every option on the platform looks like. It is being removed from \ the denominator here, so this report will read slightly lower than the \ platform. Set `dropped_as: full_credit` if the platform kept the point", placement.number )); } } } let mut items = Vec::new(); let mut p_values = Vec::new(); let mut rpbs = Vec::new(); for (j, number) in matrix.items.iter().enumerate() { let rows = set.for_item(*number); let x: Vec = coded.iter().map(|r| r[j]).collect(); let rest: Vec = totals .iter() .zip(x.iter()) .map(|(total, xi)| total - xi) .collect(); let n = matrix.n_students(); let n_answered = matrix.coded.iter().filter(|r| r[j].is_some()).count(); let p_value = mean(&x); let rpb = correlation(&x, &rest); let (upper_rate, lower_rate, discrimination_index) = if groups_usable { let u = mean_of(&x, &upper); let l = mean_of(&x, &lower); (Some(u), Some(l), Some(u - l)) } else { (None, None, None) }; // Keys: prefer the record, fall back to what the data says earned credit. let key: Vec = record .and_then(|r| r.placement(*number)) .map(|p| p.key.clone()) .filter(|k| !k.is_empty()) .unwrap_or_else(|| infer_key(&rows)); let mean_credit = if rows.is_empty() { 0.0 } else { rows.iter().map(|r| r.credit).sum::() / rows.len() as f64 }; // Per-option statistics. let student_index: BTreeMap<&str, usize> = matrix .students .iter() .enumerate() .map(|(i, s)| (s.as_str(), i)) .collect(); let mut chose: BTreeMap> = BTreeMap::new(); let mut credits: BTreeMap> = BTreeMap::new(); let mut blank = 0usize; for r in &rows { if r.chosen().is_empty() { blank += 1; continue; } // A multiple-response item is credited to the joined set, so that // "chose A and C" is one response pattern rather than two options. let label = r.chosen().join("+"); if let Some(&si) = student_index.get(r.student_key.as_str()) { chose.entry(label.clone()).or_default().push(si); } credits.entry(label).or_default().push(r.credit); } // Every declared option appears, even one nobody chose: a rate of zero is // the finding. let mut letters: BTreeSet = chose.keys().cloned().collect(); if let (Some(rec), Some(cat)) = (record, catalog) { if let Some(p) = rec.placement(*number) { if let Some(entry) = cat.get(&p.item) { for o in &entry.item.options { letters.insert(o.id.clone()); } } } } let responded = rows.len().max(1); let mut options = BTreeMap::new(); for letter in letters { let indices = chose.get(&letter).cloned().unwrap_or_default(); let count = indices.len(); let indicator: Vec = (0..n) .map(|i| if indices.contains(&i) { 1.0 } else { 0.0 }) .collect(); let set_indices: BTreeSet = indices.iter().copied().collect(); let cr = credits.get(&letter).cloned().unwrap_or_default(); options.insert( letter.clone(), OptionAnalysis { letter: letter.clone(), count, rate: count as f64 / responded as f64, point_biserial: correlation(&indicator, &rest), upper_rate: if groups_usable { Some(mean_of(&indicator, &upper)) } else { None }, lower_rate: if groups_usable { Some(mean_of(&indicator, &lower)) } else { None }, is_key: key.contains(&letter) || (key.len() > 1 && letter == key.join("+")), mean_credit: if cr.is_empty() { 0.0 } else { cr.iter().sum::() / cr.len() as f64 }, }, ); let _ = set_indices; } let design = record .and_then(|r| r.placement(*number)) .and_then(|p| catalog.and_then(|c| c.get(&p.item))) .and_then(|e| e.item.design.clone()); let mut analysis = ItemAnalysis { number: *number, item_ref: record .and_then(|r| r.placement(*number)) .map(|p| p.item.clone()), variant: one_variant(set, *number), n, n_answered, blank_rate: blank as f64 / responded as f64, p_value, mean_credit, point_biserial: rpb, discrimination_index, upper_rate, lower_rate, key, options, flags: Vec::new(), notes: Vec::new(), prediction: None, }; // Whether the authored expectation is evidence or a guess. An item that // has never been administered has no calibration block, and one edited // since its last calibration has a fingerprint that no longer matches. let calibrated = record .and_then(|r| r.placement(*number)) .and_then(|p| catalog.and_then(|c| c.get(&p.item))) .and_then(|entry| { let cal = entry.item.calibration.as_ref()?; let enough = cal.n_examinees.unwrap_or(0) >= t.calibrated_n; let current = match &cal.fingerprint { Some(recorded) => *recorded == entry.item.fingerprint(), // An older calibration block with no fingerprint cannot be // shown stale, so it is taken at its word. None => true, }; Some(enough && current) }) .unwrap_or(false); flag_item(&mut analysis, t, design.as_ref(), &rows, calibrated); p_values.push(p_value); if let Some(r) = rpb { rpbs.push(r); } items.push(analysis); } let reliability = reliability(&coded, &totals, &p_values, &rpbs); Analysis { items, reliability, warnings, } } /// Applies the flag rules to one item. /// /// # Arguments /// /// * `a` - the item analysis, updated in place. /// * `t` - the thresholds. /// * `design` - the authored expectation, when available. /// * `rows` - the raw responses, for partial-credit detection. /// * `calibrated` - whether that expectation rests on prior data. fn flag_item( a: &mut ItemAnalysis, t: &Thresholds, design: Option<&Design>, rows: &[&crate::responses::Response], calibrated: bool, ) { // Discrimination first: it is the finding that changes what you do. match a.point_biserial { Some(r) if r < t.negative_discrimination => { a.flags.push(Flag::NegativeDiscrimination); a.notes.push(format!( "students who did better overall did worse on this item (r = {r:.2}). Check the \ key before anything else." )); } Some(r) if r < t.low_discrimination => { a.flags.push(Flag::LowDiscrimination); a.notes.push(if r < 0.0 { format!( "this item did not separate stronger from weaker students (r = {r:.2}, which \ is indistinguishable from zero at this sample size)." ) } else { format!("this item barely separates stronger from weaker students (r = {r:.2}).") }); } None => { a.notes.push( "every student responded the same way, so this item has no variance and no \ correlation can be computed." .to_string(), ); } _ => {} } if a.p_value > t.too_easy { a.flags.push(Flag::TooEasy); a.notes.push(format!( "{:.0}% answered correctly. Fine as an opening anchor, but it carries little \ information about who knows what.", a.p_value * 100.0 )); } if a.p_value < t.too_hard { // Hard and discriminating is a good item, not a broken one. let discriminates = a .point_biserial .map(|r| r >= t.low_discrimination) .unwrap_or(false); if !discriminates { a.flags.push(Flag::TooHard); a.notes.push(format!( "only {:.0}% answered correctly, and the item did not separate students. That \ pattern usually means the stem is unclear or a prerequisite is missing, not that \ the content is hard.", a.p_value * 100.0 )); } } // Distractor analysis: the real source of "poorly worded question" findings. let key_rpb = a .options .values() .filter(|o| o.is_key) .filter_map(|o| o.point_biserial) .fold(f64::NEG_INFINITY, f64::max); for o in a.options.values() { if o.is_key { continue; } if let Some(r) = o.point_biserial { if key_rpb.is_finite() && r > key_rpb && o.rate >= 0.1 { a.flags.push(Flag::DistractorOutperformsKey); a.notes.push(format!( "option {} correlates with overall performance better than the key does \ (r = {r:.2} against {key_rpb:.2}), and {:.0}% chose it. Either it is the \ better answer or the stem admits it.", o.letter, o.rate * 100.0 )); } } if let (Some(upper), Some(key_upper)) = ( o.upper_rate, a.options .values() .filter(|k| k.is_key) .filter_map(|k| k.upper_rate) .fold(None, |acc: Option, v| { Some(acc.map_or(v, |a| a.max(v))) }), ) { if upper > key_upper && upper >= 0.25 { a.flags.push(Flag::KeyUnderperforms); a.notes.push(format!( "the strongest students chose option {} more often than the key ({:.0}% \ against {:.0}%). That split is the signature of two defensible readings.", o.letter, upper * 100.0, key_upper * 100.0 )); } } if o.rate <= t.nonfunctioning { a.flags.push(Flag::NonfunctioningDistractor); a.notes.push(format!( "option {} was chosen by {:.0}% of students, so it is not doing any work. \ Replace it with a plausible error students actually make.", o.letter, o.rate * 100.0 )); } } // Partial credit awarded to a non-key option is a grading-time admission of // ambiguity, and it is the strongest such signal available. let ambiguous = a .options .values() .any(|o| !o.is_key && o.mean_credit > 0.0 && o.count > 0); if ambiguous { a.flags.push(Flag::Ambiguous); let letters: Vec = a .options .values() .filter(|o| !o.is_key && o.mean_credit > 0.0 && o.count > 0) .map(|o| format!("{} ({:.0}%)", o.letter, o.mean_credit * 100.0)) .collect(); a.notes.push(format!( "partial credit was awarded at grading time to {}, which records a decision that the \ item admitted more than one reading. Rewrite the stem rather than re-deciding this \ every term.", letters.join(", ") )); } // Rapid guessing, when the platform reported response times. let times: Vec = rows .iter() .filter_map(|r| r.response_time_seconds) .collect(); if times.len() >= 5 { let rapid = times.iter().filter(|t| **t < 5.0).count() as f64 / times.len() as f64; if rapid > 0.15 { a.flags.push(Flag::HighRapidGuess); a.notes.push(format!( "{:.0}% of responses arrived in under five seconds, which is faster than the stem \ can be read. That is usually about the item's position on the form or time \ pressure, not the item.", rapid * 100.0 )); } } // Did the item behave as authored? This is the one check whose meaning // depends on where the expectation came from, so it is recorded either way // and flagged only when the expectation had data behind it. if let Some(d) = design { let mut prediction = Prediction { expected_p: d.expected_difficulty, expected_band: d.expected_discrimination.map(|b| b.expected_band()), calibrated, p_error: None, p_within: None, band_hit: None, notes: Vec::new(), }; if let Some(expected) = d.expected_difficulty { let error = a.p_value - expected; let within = error.abs() <= t.design_tolerance; prediction.p_error = Some(error); prediction.p_within = Some(within); if !within { if calibrated { a.flags.push(Flag::DesignMismatch); prediction.notes.push(format!( "this item is calibrated at about {:.0}% correct and came out at {:.0}%. \ Something changed: the cohort, the teaching, or the item.", expected * 100.0, a.p_value * 100.0 )); } else { prediction.notes.push(format!( "you predicted about {:.0}% correct and observed {:.0}%. This is the \ first data on the item, so it corrects the prediction rather than \ condemning the item.", expected * 100.0, a.p_value * 100.0 )); } } } if let (Some(band), Some(r)) = (d.expected_discrimination, a.point_biserial) { let (low, high) = band.expected_band(); let hit = r >= low && r <= high; prediction.band_hit = Some(hit); if !hit { if calibrated { if !a.flags.contains(&Flag::DesignMismatch) { a.flags.push(Flag::DesignMismatch); } prediction.notes.push(format!( "calibrated for {} discrimination ({low:.2} to {high:.2}), observed \ {r:.2}.", format!("{band:?}").to_lowercase() )); } else { prediction.notes.push(format!( "you predicted {} discrimination ({low:.2} to {high:.2}) and observed \ {r:.2}.", format!("{band:?}").to_lowercase() )); } } } a.prediction = Some(prediction); } a.flags.sort(); a.flags.dedup(); } /// Computes whole-test reliability. /// /// # Arguments /// /// * `coded` - the 0/1 response matrix. /// * `totals` - per-student totals. /// * `p_values` - per-item p-values. /// * `rpbs` - per-item point-biserials that could be computed. /// /// # Returns /// /// The reliability summary. fn reliability(coded: &[Vec], totals: &[f64], p_values: &[f64], rpbs: &[f64]) -> Reliability { let n_students = coded.len(); let n_items = coded.first().map(|r| r.len()).unwrap_or(0); // Named distinctly from the `mean` and `sd` helpers: binding `let mean = // mean(totals)` shadows the function for the rest of the scope, which then // makes the later `mean(p_values)` a call on an f64. let mean_total = mean(totals); let sd_total = sd(totals); // KR-20. The variance terms use the population form, which is the convention // for this coefficient and matches what other packages report. let alpha = if n_items > 1 && sd_total > 0.0 { let sum_pq: f64 = p_values.iter().map(|p| p * (1.0 - p)).sum(); let k = n_items as f64; let variance = sd_total * sd_total; Some((k / (k - 1.0)) * (1.0 - sum_pq / variance)) } else { None }; let sem = alpha.map(|a| sd_total * (1.0 - a).max(0.0).sqrt()); Reliability { n_items, n_students, mean: mean_total, sd: sd_total, alpha, sem, mean_p: mean(p_values), mean_point_biserial: if rpbs.is_empty() { None } else { Some(mean(rpbs)) }, } } /// Infers the key from which options earned full credit. /// /// Used when no assessment record is available, so that an export can be analyzed /// before its record is written. /// /// # Arguments /// /// * `rows` - the responses for one item. /// /// # Returns /// /// The letters that appear on full-credit responses. /// The single variant every row for one question names, if they agree. /// /// Disagreement is not an error, it is a fact about the administration: a course /// that draws distractors per form has two option sets under one question /// number, and no pooled statistic describes both. Returning `None` is what /// keeps the per-variant records from claiming otherwise. fn one_variant(set: &ResponseSet, number: u32) -> Option { let mut seen: Option<&str> = None; for row in set.rows.iter().filter(|r| r.item_number == number) { let variant = row.variant.as_deref()?; match seen { None => seen = Some(variant), Some(first) if first == variant => {} Some(_) => return None, } } seen.map(str::to_string) } fn infer_key(rows: &[&crate::responses::Response]) -> Vec { let mut out: BTreeSet = BTreeSet::new(); for r in rows { if r.credit >= 0.999 { for letter in r.chosen() { out.insert(letter.clone()); } } } out.into_iter().collect() } /// The arithmetic mean, zero for an empty slice. fn mean(v: &[f64]) -> f64 { if v.is_empty() { 0.0 } else { v.iter().sum::() / v.len() as f64 } } /// The population standard deviation. fn sd(v: &[f64]) -> f64 { if v.len() < 2 { return 0.0; } let m = mean(v); (v.iter().map(|x| (x - m) * (x - m)).sum::() / v.len() as f64).sqrt() } /// The mean of the entries at the given indices. fn mean_of(v: &[f64], indices: &BTreeSet) -> f64 { if indices.is_empty() { return 0.0; } indices.iter().map(|i| v[*i]).sum::() / indices.len() as f64 } /// The Pearson correlation of two equal-length vectors. /// /// # Arguments /// /// * `x` - the first vector. /// * `y` - the second vector. /// /// # Returns /// /// The correlation, or `None` when either vector has no variance. Returning /// `None` rather than a NaN matters: an item everyone answered correctly has no /// correlation, and that is a meaningful result to report rather than a number to /// propagate. pub fn correlation(x: &[f64], y: &[f64]) -> Option { if x.len() != y.len() || x.len() < 2 { return None; } let mx = mean(x); let my = mean(y); let mut sxy = 0.0; let mut sxx = 0.0; let mut syy = 0.0; for (xi, yi) in x.iter().zip(y.iter()) { let dx = xi - mx; let dy = yi - my; sxy += dx * dy; sxx += dx * dx; syy += dy * dy; } if sxx <= f64::EPSILON || syy <= f64::EPSILON { return None; } Some(sxy / (sxx * syy).sqrt()) } #[cfg(test)] mod tests { use super::*; use crate::responses::Response; fn resp(student: &str, number: u32, letter: &str, credit: f64) -> Response { Response { administration_id: "C/T/a".into(), course: "C".into(), term: "T".into(), assessment_id: "a".into(), date: None, form: None, form_position: None, student_key: student.into(), sid: None, name: None, email: None, section: None, item_number: number, item_ref: None, item_version: None, variant: None, selected: if letter.is_empty() { vec![] } else { vec![letter.to_string()] }, selected_source: vec![], eliminated: vec![], eliminated_source: vec![], correct: Some(credit >= 0.999), credit, points_possible: 1.0, score: credit, response_time_seconds: None, level: None, learning_targets: vec![], topics: vec![], bonus: false, dropped: false, dropped_full_credit: false, } } /// Twelve students; item 1 discriminates, item 2 is keyed backwards. /// /// Item 2 is only *partly* reversed, and that is load-bearing rather than /// sloppy. The corrected point-biserial scores each item against the total of /// the *other* items, and with four items — one of them unanimous — item 2 is /// most of item 1's rest score. Make item 2 an exact complement of item 1 and /// `item1 + item2 == 1` for every student, so the total collapses to /// `2 + item4`, carries no ability signal at all, and both items come out at /// r = -0.71. The fixture then contradicts itself: the item it calls good is /// flagged for negative discrimination, and the reversed key is negative only /// because it mirrors item 1 rather than because it is miskeyed. fn sample() -> ResponseSet { let mut set = ResponseSet::new(); for i in 0..12 { let strong = i < 6; let s = format!("s{i:02}"); // Item 1: strong students right, weak wrong. set.rows.push(resp( &s, 1, if strong { "A" } else { "B" }, if strong { 1.0 } else { 0.0 }, )); // Item 2: weaker students do better on it, which is what a keying // error looks like. Half the strong group and two thirds of the weak // group get it, so it is reversed without mirroring item 1. let missed = if strong { i < 3 } else { i < 8 }; set.rows.push(resp( &s, 2, if missed { "C" } else { "D" }, if missed { 0.0 } else { 1.0 }, )); // Item 3: everyone correct. set.rows.push(resp(&s, 3, "A", 1.0)); // Item 4: a second discriminating item, so that removing any one item // still leaves a rest score that tracks ability. set.rows.push(resp( &s, 4, if strong { "A" } else { "B" }, if strong { 1.0 } else { 0.0 }, )); } set } #[test] fn correlation_returns_none_without_variance() { assert_eq!(correlation(&[1.0, 1.0, 1.0], &[1.0, 2.0, 3.0]), None); assert_eq!(correlation(&[1.0], &[1.0]), None); let r = correlation(&[1.0, 2.0, 3.0], &[2.0, 4.0, 6.0]).unwrap(); assert!((r - 1.0).abs() < 1e-12, "perfect correlation is 1, got {r}"); let r = correlation(&[1.0, 2.0, 3.0], &[3.0, 2.0, 1.0]).unwrap(); assert!((r + 1.0).abs() < 1e-12); } #[test] fn flags_a_reversed_key_as_negative_discrimination() { let set = sample(); let a = analyze(&set, &Thresholds::default(), None, None); let item2 = a.items.iter().find(|i| i.number == 2).unwrap(); assert!( item2.point_biserial.unwrap() < 0.0, "got {:?}", item2.point_biserial ); assert!(item2.flags.contains(&Flag::NegativeDiscrimination)); assert!(item2.needs_revision()); } #[test] fn a_good_item_is_not_flagged_for_discrimination() { let set = sample(); let a = analyze(&set, &Thresholds::default(), None, None); let item1 = a.items.iter().find(|i| i.number == 1).unwrap(); assert!(item1.point_biserial.unwrap() > 0.5); assert!(!item1.flags.contains(&Flag::NegativeDiscrimination)); assert!(!item1.flags.contains(&Flag::LowDiscrimination)); } #[test] fn a_unanimous_item_has_no_correlation_and_is_flagged_easy() { let set = sample(); let a = analyze(&set, &Thresholds::default(), None, None); let item3 = a.items.iter().find(|i| i.number == 3).unwrap(); assert_eq!(item3.p_value, 1.0); assert_eq!(item3.point_biserial, None, "no variance, so no correlation"); assert!(item3.flags.contains(&Flag::TooEasy)); } #[test] fn blank_responses_count_as_incorrect_but_are_reported() { let mut set = ResponseSet::new(); for i in 0..10 { let s = format!("s{i}"); let letter = if i < 5 { "A" } else { "" }; set.rows .push(resp(&s, 1, letter, if i < 5 { 1.0 } else { 0.0 })); set.rows .push(resp(&s, 2, "A", if i < 7 { 1.0 } else { 0.0 })); } let a = analyze(&set, &Thresholds::default(), None, None); let item1 = a.items.iter().find(|i| i.number == 1).unwrap(); assert_eq!(item1.p_value, 0.5); assert_eq!(item1.blank_rate, 0.5); assert_eq!( item1.n_answered, 10, "a blank is still an administered item" ); } #[test] fn partial_credit_to_a_distractor_flags_ambiguity() { let mut set = ResponseSet::new(); for i in 0..10 { let s = format!("s{i}"); if i < 5 { set.rows.push(resp(&s, 1, "A", 1.0)); } else { // B earned two thirds of the points at grading time. set.rows.push(resp(&s, 1, "B", 0.667)); } set.rows .push(resp(&s, 2, "A", if i % 2 == 0 { 1.0 } else { 0.0 })); } let a = analyze(&set, &Thresholds::default(), None, None); let item1 = a.items.iter().find(|i| i.number == 1).unwrap(); assert!(item1.flags.contains(&Flag::Ambiguous), "{:?}", item1.flags); assert!(item1.notes.iter().any(|n| n.contains("partial credit"))); } #[test] fn reliability_is_computed_and_explained() { let set = sample(); let a = analyze(&set, &Thresholds::default(), None, None); assert_eq!(a.reliability.n_items, 4); assert_eq!(a.reliability.n_students, 12); assert!(a.reliability.alpha.is_some()); let text = a.reliability.interpretation(); assert!(text.contains("KR-20")); // A four-item test must carry the length caveat. assert!(text.contains("test length")); } #[test] fn small_samples_get_a_caution() { let set = sample(); let a = analyze(&set, &Thresholds::default(), None, None); assert!(a.warnings.iter().any(|w| w.contains("examinees"))); } #[test] fn empty_input_does_not_panic() { let a = analyze(&ResponseSet::new(), &Thresholds::default(), None, None); assert!(a.items.is_empty()); assert_eq!(a.reliability.alpha, None); assert!(!a.warnings.is_empty()); } #[test] fn the_revise_queue_puts_blocking_flags_first() { let set = sample(); let a = analyze(&set, &Thresholds::default(), None, None); let queue = a.revise_queue(); assert!(!queue.is_empty()); assert_eq!(queue[0].number, 2, "the reversed key comes first"); } }