refactor: improve cohort report

This commit is contained in:
2026-09-19 22:19:25 -04:00
parent c06d5caa8c
commit 92c8a8fc75
6 changed files with 2194 additions and 311 deletions
+119 -17
View File
@@ -61,6 +61,13 @@ pub struct Thresholds {
pub nonfunctioning: f64, pub nonfunctioning: f64,
/// How far observed difficulty may drift from the authored expectation. /// How far observed difficulty may drift from the authored expectation.
pub design_tolerance: f64, pub design_tolerance: f64,
/// How many examinees an item's calibration needs before its recorded
/// expectations are treated as evidence rather than as the author's guess.
///
/// Fifty is the point at which the standard error of a proportion near 0.5
/// drops to about 0.07, which is small enough that a quarter-point miss is
/// about the item rather than about the sample.
pub calibrated_n: usize,
/// Fraction of the class in the upper and lower comparison groups. Kelley's /// Fraction of the class in the upper and lower comparison groups. Kelley's
/// 0.27 maximizes the difference between the groups for a normal /// 0.27 maximizes the difference between the groups for a normal
/// distribution, and it remains the convention. /// distribution, and it remains the convention.
@@ -78,6 +85,7 @@ impl Default for Thresholds {
negative_discrimination: -0.05, negative_discrimination: -0.05,
nonfunctioning: 0.05, nonfunctioning: 0.05,
design_tolerance: 0.25, design_tolerance: 0.25,
calibrated_n: 50,
group_fraction: 0.27, group_fraction: 0.27,
small_sample: 100, small_sample: 100,
} }
@@ -154,6 +162,38 @@ pub struct ItemAnalysis {
pub flags: Vec<Flag>, pub flags: Vec<Flag>,
/// Human-readable explanations tied to the flags. /// Human-readable explanations tied to the flags.
pub notes: Vec<String>, pub notes: Vec<String>,
/// How the item behaved against what its author predicted, when the item
/// records a prediction.
pub prediction: Option<Prediction>,
}
/// An authored expectation, checked against what happened.
///
/// Kept apart from [`ItemAnalysis::flags`] on purpose. Before an item has been
/// administered, `design.expected_difficulty` is the author's guess, and a guess
/// that turns out wrong says something about the guess rather than about the
/// item. Flagging it anyway is how a report ends up with thirty
/// `design_mismatch` findings and no way to see the four that matter. So the
/// discrepancy is always recorded here, and it only becomes a
/// [`Flag::DesignMismatch`] once the expectation has data behind it.
#[derive(Debug, Clone)]
pub struct Prediction {
/// The difficulty the author expected.
pub expected_p: Option<f64>,
/// The discrimination band the author expected, as `(low, high)`.
pub expected_band: Option<(f64, f64)>,
/// Whether the expectation rests on a calibration with enough examinees
/// behind it, rather than on the author's judgement alone.
pub calibrated: bool,
/// Signed difficulty error, observed minus expected. Positive means the item
/// was easier than predicted.
pub p_error: Option<f64>,
/// Whether observed difficulty landed inside the tolerance.
pub p_within: Option<bool>,
/// Whether observed discrimination landed inside the expected band.
pub band_hit: Option<bool>,
/// What to say about it, phrased for whichever case applies.
pub notes: Vec<String>,
} }
impl ItemAnalysis { impl ItemAnalysis {
@@ -493,9 +533,29 @@ pub fn analyze(
options, options,
flags: Vec::new(), flags: Vec::new(),
notes: Vec::new(), notes: Vec::new(),
prediction: None,
}; };
flag_item(&mut analysis, t, design.as_ref(), &rows); // Whether the authored expectation is evidence or a guess. An item that
// has never been administered has no calibration block, and one edited
// since its last calibration has a fingerprint that no longer matches.
let calibrated = record
.and_then(|r| r.placement(*number))
.and_then(|p| catalog.and_then(|c| c.get(&p.item)))
.and_then(|entry| {
let cal = entry.item.calibration.as_ref()?;
let enough = cal.n_examinees.unwrap_or(0) >= t.calibrated_n;
let current = match &cal.fingerprint {
Some(recorded) => *recorded == entry.item.fingerprint(),
// An older calibration block with no fingerprint cannot be
// shown stale, so it is taken at its word.
None => true,
};
Some(enough && current)
})
.unwrap_or(false);
flag_item(&mut analysis, t, design.as_ref(), &rows, calibrated);
p_values.push(p_value); p_values.push(p_value);
if let Some(r) = rpb { if let Some(r) = rpb {
@@ -521,11 +581,13 @@ pub fn analyze(
/// * `t` - the thresholds. /// * `t` - the thresholds.
/// * `design` - the authored expectation, when available. /// * `design` - the authored expectation, when available.
/// * `rows` - the raw responses, for partial-credit detection. /// * `rows` - the raw responses, for partial-credit detection.
/// * `calibrated` - whether that expectation rests on prior data.
fn flag_item( fn flag_item(
a: &mut ItemAnalysis, a: &mut ItemAnalysis,
t: &Thresholds, t: &Thresholds,
design: Option<&Design>, design: Option<&Design>,
rows: &[&crate::responses::Response], rows: &[&crate::responses::Response],
calibrated: bool,
) { ) {
// Discrimination first: it is the finding that changes what you do. // Discrimination first: it is the finding that changes what you do.
match a.point_biserial { match a.point_biserial {
@@ -678,31 +740,71 @@ fn flag_item(
} }
} }
// Did the item behave as authored? // Did the item behave as authored? This is the one check whose meaning
// depends on where the expectation came from, so it is recorded either way
// and flagged only when the expectation had data behind it.
if let Some(d) = design { if let Some(d) = design {
let mut prediction = Prediction {
expected_p: d.expected_difficulty,
expected_band: d.expected_discrimination.map(|b| b.expected_band()),
calibrated,
p_error: None,
p_within: None,
band_hit: None,
notes: Vec::new(),
};
if let Some(expected) = d.expected_difficulty { if let Some(expected) = d.expected_difficulty {
if (expected - a.p_value).abs() > t.design_tolerance { let error = a.p_value - expected;
a.flags.push(Flag::DesignMismatch); let within = error.abs() <= t.design_tolerance;
a.notes.push(format!( prediction.p_error = Some(error);
"you expected about {:.0}% correct and observed {:.0}%. Worth knowing whether \ prediction.p_within = Some(within);
your model of the students or the item is off.", if !within {
expected * 100.0, if calibrated {
a.p_value * 100.0 a.flags.push(Flag::DesignMismatch);
)); prediction.notes.push(format!(
"this item is calibrated at about {:.0}% correct and came out at {:.0}%. \
Something changed: the cohort, the teaching, or the item.",
expected * 100.0,
a.p_value * 100.0
));
} else {
prediction.notes.push(format!(
"you predicted about {:.0}% correct and observed {:.0}%. This is the \
first data on the item, so it corrects the prediction rather than \
condemning the item.",
expected * 100.0,
a.p_value * 100.0
));
}
} }
} }
if let (Some(band), Some(r)) = (d.expected_discrimination, a.point_biserial) { if let (Some(band), Some(r)) = (d.expected_discrimination, a.point_biserial) {
let (low, high) = band.expected_band(); let (low, high) = band.expected_band();
if r < low || r > high { let hit = r >= low && r <= high;
if !a.flags.contains(&Flag::DesignMismatch) { prediction.band_hit = Some(hit);
a.flags.push(Flag::DesignMismatch); if !hit {
if calibrated {
if !a.flags.contains(&Flag::DesignMismatch) {
a.flags.push(Flag::DesignMismatch);
}
prediction.notes.push(format!(
"calibrated for {} discrimination ({low:.2} to {high:.2}), observed \
{r:.2}.",
format!("{band:?}").to_lowercase()
));
} else {
prediction.notes.push(format!(
"you predicted {} discrimination ({low:.2} to {high:.2}) and observed \
{r:.2}.",
format!("{band:?}").to_lowercase()
));
} }
a.notes.push(format!(
"you expected {} discrimination ({low:.2} to {high:.2}) and observed {r:.2}.",
format!("{band:?}").to_lowercase()
));
} }
} }
a.prediction = Some(prediction);
} }
a.flags.sort(); a.flags.sort();
+714
View File
@@ -917,8 +917,19 @@ pub struct CohortDiagnostic {
pub objectives: Vec<CohortObjectiveRow>, pub objectives: Vec<CohortObjectiveRow>,
/// Objectives the class as a whole did not meet. /// Objectives the class as a whole did not meet.
pub gaps: Vec<CohortObjectiveRow>, pub gaps: Vec<CohortObjectiveRow>,
/// The distribution binned by the course's letter-grade scale. Empty when
/// `course.yaml` sets no scale, in which case the ten-point bins stand.
#[serde(skip_serializing_if = "Vec::is_empty")]
pub grades: Vec<GradeRow>,
/// Per-lecture class performance, worst first.
#[serde(skip_serializing_if = "Vec::is_empty")]
pub lectures: Vec<CohortLectureRow>,
/// Per-question statistics. /// Per-question statistics.
pub questions: Vec<CohortQuestionRow>, pub questions: Vec<CohortQuestionRow>,
/// What to do about each question that raised something.
pub triage: Triage,
/// How the authored expectations did.
pub predictions: PredictionSummary,
/// Questions worth revisiting before reuse, worst first. /// Questions worth revisiting before reuse, worst first.
pub revise: Vec<CohortQuestionRow>, pub revise: Vec<CohortQuestionRow>,
/// One row per form, when more than one was given. /// One row per form, when more than one was given.
@@ -1036,6 +1047,20 @@ pub struct CohortQuestionRow {
pub level: Option<u8>, pub level: Option<u8>,
/// The objectives it measured. /// The objectives it measured.
pub objectives: Vec<String>, pub objectives: Vec<String>,
/// What those objectives ask, in the course's own words.
#[serde(skip_serializing_if = "Vec::is_empty")]
pub objective_texts: Vec<String>,
/// Where the item was taught, as lecture titles and slide numbers.
#[serde(skip_serializing_if = "Vec::is_empty")]
pub taught_in: Vec<String>,
/// The lecture ids alone, for a table column where only `L1.4` fits.
#[serde(skip_serializing_if = "Vec::is_empty")]
pub lectures: Vec<String>,
/// Which difficulty band it fell in: `too easy`, `moderate`, or `hard`.
pub difficulty_band: String,
/// Which discrimination band it fell in, on the conventional cut points:
/// `excellent`, `good`, `marginal`, `poor`, or `negative`.
pub discrimination_band: String,
/// Proportion correct. /// Proportion correct.
pub p_value: f64, pub p_value: f64,
/// Corrected item-total point-biserial. /// Corrected item-total point-biserial.
@@ -1054,6 +1079,13 @@ pub struct CohortQuestionRow {
pub flags: Vec<String>, pub flags: Vec<String>,
/// What those flags mean. /// What those flags mean.
pub notes: Vec<String>, pub notes: Vec<String>,
/// How the item did against its author's expectation. Separate from `notes`
/// because an unmet prediction on an uncalibrated item is a fact about the
/// prediction.
#[serde(skip_serializing_if = "Vec::is_empty")]
pub prediction_notes: Vec<String>,
/// Whether that expectation rested on a prior calibration.
pub calibrated: bool,
/// Per-form proportion correct, when more than one form was given. A gap here /// Per-form proportion correct, when more than one form was given. A gap here
/// on one question, with the rest of the exam in step, points at that /// on one question, with the rest of the exam in step, points at that
/// question's permutation rather than at the cohort. /// question's permutation rather than at the cohort.
@@ -1106,6 +1138,159 @@ pub struct PatternRow {
pub level_means: BTreeMap<u8, f64>, pub level_means: BTreeMap<u8, f64>,
} }
/// The class's scores binned by the course's own letter-grade scale.
///
/// A ten-point histogram is the default because it needs no course
/// configuration, but nobody acts on "nineteen students in the fifties". They
/// act on "nineteen students are failing", and that sentence needs the scale
/// from `course.yaml`.
#[derive(Debug, Clone, Serialize)]
#[serde(rename_all = "kebab-case")]
pub struct GradeRow {
/// The letter.
pub letter: String,
/// The lowest percentage in the band.
pub low: f64,
/// The highest percentage in the band, which is just under the next band's
/// floor, or 100 for the top band.
pub high: f64,
/// Grade points, when the scale records them.
#[serde(skip_serializing_if = "Option::is_none")]
pub gpa: Option<f64>,
/// The attainment word, when the scale records one.
#[serde(skip_serializing_if = "Option::is_none")]
pub attainment: Option<String>,
/// The colour group, so A, A- and A+ can be tinted together.
pub group: String,
/// How many students landed in the band.
pub count: usize,
/// Their share of the class, in `0.0..=1.0`.
pub share: f64,
/// How many students are in this band or a higher one.
pub at_or_above: usize,
}
/// One lecture's showing, aggregated from the items written against it.
///
/// The objective table answers "which objective went wrong". This answers "which
/// class meeting went wrong", which is the question that maps onto next week.
#[derive(Debug, Clone, Serialize)]
#[serde(rename_all = "kebab-case")]
pub struct CohortLectureRow {
/// The lecture id.
pub lecture: String,
/// Its title.
pub title: String,
/// How many scored items traced back to it.
pub n_items: usize,
/// How many distinct objectives those items measured.
pub n_objectives: usize,
/// How many of those objectives the class did not meet.
pub n_objectives_below: usize,
/// Mean proportion correct across its items.
pub rate: f64,
/// The questions, so the row can be checked against the item table.
pub questions: Vec<u32>,
/// The worst objective under this lecture, by class rate.
#[serde(skip_serializing_if = "Option::is_none")]
pub worst_objective: Option<String>,
}
/// What to do about one question, and why.
#[derive(Debug, Clone, Serialize)]
#[serde(rename_all = "kebab-case")]
pub struct TriageRow {
/// The question number.
pub number: u32,
/// The item id.
#[serde(skip_serializing_if = "Option::is_none")]
pub item: Option<String>,
/// The level code.
#[serde(skip_serializing_if = "Option::is_none")]
pub level: Option<u8>,
/// Proportion correct.
pub p_value: f64,
/// Corrected item-total correlation.
#[serde(skip_serializing_if = "Option::is_none")]
pub point_biserial: Option<f64>,
/// Upper minus lower group.
#[serde(skip_serializing_if = "Option::is_none")]
pub discrimination: Option<f64>,
/// What the question measured, in the course's words.
#[serde(skip_serializing_if = "Vec::is_empty")]
pub objectives: Vec<String>,
/// Where it was taught.
#[serde(skip_serializing_if = "Vec::is_empty")]
pub taught_in: Vec<String>,
/// The specific option this recommendation is about, when it is about one.
#[serde(skip_serializing_if = "Option::is_none")]
pub option: Option<String>,
/// That option's share of responses.
#[serde(skip_serializing_if = "Option::is_none")]
pub option_share: Option<f64>,
/// That option's correlation with total score.
#[serde(skip_serializing_if = "Option::is_none")]
pub option_point_biserial: Option<f64>,
/// The evidence, one clause per line.
pub reasons: Vec<String>,
}
/// Every question sorted into what to do with it.
///
/// The first four lists are decisions about items and are mutually exclusive: a
/// question appears in the most severe one that fits, because there is no point
/// rewriting a distractor on an item you are about to discard. `reteach` is not
/// a decision about an item at all, so a question can appear there as well as in
/// one of the others.
#[derive(Debug, Clone, Default, Serialize)]
#[serde(rename_all = "kebab-case")]
pub struct Triage {
/// Broken: the evidence says these did not measure what they were scored on.
pub discard: Vec<TriageRow>,
/// A second defensible answer with statistical support behind it.
pub rekey: Vec<TriageRow>,
/// Weak but salvageable, worth rewriting before reuse.
pub revise: Vec<TriageRow>,
/// Sound items the class got wrong. A teaching finding, not an item finding.
pub reteach: Vec<TriageRow>,
/// Items whose low discrimination is explained by their difficulty rather
/// than by a fault. Listed so they are not mistaken for work to do.
pub bounded: Vec<TriageRow>,
/// How many questions raised nothing at all.
pub clean: usize,
}
/// How the authored expectations did against the data.
///
/// This exists so that an uncalibrated bank does not produce one
/// `design_mismatch` per item. Before an item has data, its expected difficulty
/// is a prediction by its author, and the useful summary is whether those
/// predictions run optimistic or pessimistic as a set.
#[derive(Debug, Clone, Default, Serialize)]
#[serde(rename_all = "kebab-case")]
pub struct PredictionSummary {
/// How many items recorded an expected difficulty.
pub n_predicted: usize,
/// How many of those expectations rest on a prior calibration.
pub n_calibrated: usize,
/// Mean of observed minus expected difficulty. Positive means the items came
/// out easier than predicted.
#[serde(skip_serializing_if = "Option::is_none")]
pub mean_signed_error: Option<f64>,
/// Mean absolute difficulty error, which is the size of a typical miss.
#[serde(skip_serializing_if = "Option::is_none")]
pub mean_abs_error: Option<f64>,
/// How many landed inside the tolerance.
pub n_within: usize,
/// How many items recorded an expected discrimination band.
pub n_band: usize,
/// How many of those landed inside it.
pub n_band_hit: usize,
/// The largest single surprise, as `(question, expected, observed)`.
#[serde(skip_serializing_if = "Option::is_none")]
pub biggest_surprise: Option<(u32, f64, f64)>,
}
/// Builds the class diagnostic. /// Builds the class diagnostic.
/// ///
/// # Arguments /// # Arguments
@@ -1234,6 +1419,29 @@ pub fn cohort(
item: item.item_ref.clone(), item: item.item_ref.clone(),
level: meta.and_then(|m| m.0), level: meta.and_then(|m| m.0),
objectives: meta.map(|m| m.1.clone()).unwrap_or_default(), objectives: meta.map(|m| m.1.clone()).unwrap_or_default(),
objective_texts: meta
.map(|m| m.1.iter().map(|id| course.objective_text(id)).collect())
.unwrap_or_default(),
taught_in: item
.item_ref
.as_deref()
.map(|uid| taught_in(catalog, uid))
.unwrap_or_default(),
lectures: item
.item_ref
.as_deref()
.and_then(|uid| catalog.get(uid))
.map(|entry| {
entry
.item
.sources
.iter()
.map(|source| source.lecture.clone())
.collect()
})
.unwrap_or_default(),
difficulty_band: difficulty_band(item.p_value).to_string(),
discrimination_band: discrimination_band(item.point_biserial).to_string(),
p_value: item.p_value, p_value: item.p_value,
point_biserial: item.point_biserial, point_biserial: item.point_biserial,
discrimination: item.discrimination_index, discrimination: item.discrimination_index,
@@ -1253,6 +1461,12 @@ pub fn cohort(
.collect(), .collect(),
flags: item.flags.iter().map(|f| f.as_str().to_string()).collect(), flags: item.flags.iter().map(|f| f.as_str().to_string()).collect(),
notes: item.notes.clone(), notes: item.notes.clone(),
prediction_notes: item
.prediction
.as_ref()
.map(|p| p.notes.clone())
.unwrap_or_default(),
calibrated: item.prediction.as_ref().is_some_and(|p| p.calibrated),
by_form: by_form.get(&item.number).cloned().unwrap_or_default(), by_form: by_form.get(&item.number).cloned().unwrap_or_default(),
} }
}) })
@@ -1264,10 +1478,20 @@ pub fn cohort(
.filter_map(|item| questions.iter().find(|q| q.number == item.number).cloned()) .filter_map(|item| questions.iter().find(|q| q.number == item.number).cloned())
.collect(); .collect();
let default_options = course.policy.options_per_item;
let triage = triage(&questions, threshold, default_options);
let predictions = prediction_summary(analysis);
let grades = grade_rows(&course.policy, &percents);
let lectures = lecture_rows(catalog, &questions, &objectives, threshold);
CohortDiagnostic { CohortDiagnostic {
n_students: cohort.students.len(), n_students: cohort.students.len(),
n_items: analysis.reliability.n_items, n_items: analysis.reliability.n_items,
distribution: distribution(&percents), distribution: distribution(&percents),
grades,
lectures,
triage,
predictions,
reliability: ReliabilityRow { reliability: ReliabilityRow {
alpha: analysis.reliability.alpha, alpha: analysis.reliability.alpha,
sem: analysis.reliability.sem, sem: analysis.reliability.sem,
@@ -1299,6 +1523,496 @@ pub fn cohort(
} }
} }
/// Where an item was taught, as lecture titles with slide numbers.
///
/// # Arguments
///
/// * `catalog` - the loaded course.
/// * `uid` - the item's global id.
///
/// # Returns
///
/// One entry per source the item records.
fn taught_in(catalog: &Catalog, uid: &str) -> Vec<String> {
let Some(entry) = catalog.get(uid) else {
return Vec::new();
};
entry
.item
.sources
.iter()
.map(|source| {
let title = catalog
.course
.lectures
.get(&source.lecture)
.map(|l| l.title.clone())
.unwrap_or_else(|| source.lecture.clone());
if source.slides.is_empty() {
format!("{} ({})", title, source.lecture)
} else {
let slides: Vec<String> = source.slides.iter().map(|s| s.to_string()).collect();
format!(
"{} ({}), slide{} {}",
title,
source.lecture,
if source.slides.len() == 1 { "" } else { "s" },
slides.join(", ")
)
}
})
.collect()
}
/// The difficulty band a p-value falls in.
///
/// Three bands rather than five. The only distinction that changes what you do
/// is whether the item had room to discriminate at all, and that is a question
/// about the middle versus the two ends.
fn difficulty_band(p: f64) -> &'static str {
if p >= 0.85 {
"too easy"
} else if p <= 0.35 {
"hard"
} else {
"moderate"
}
}
/// The discrimination band a point-biserial falls in.
///
/// The cut points are the conventional ones from the item-analysis literature,
/// usually attributed to Ebel: about 0.40 and above is excellent, 0.30 to 0.39
/// good, 0.20 to 0.29 marginal, and below 0.20 poor. They are rules of thumb
/// rather than laws, and they must be read next to difficulty, because an item
/// almost everyone passes or fails has little variance left to correlate with
/// anything.
fn discrimination_band(r: Option<f64>) -> &'static str {
match r {
None => "no variance",
Some(r) if r < 0.0 => "negative",
Some(r) if r < 0.20 => "poor",
Some(r) if r < 0.30 => "marginal",
Some(r) if r < 0.40 => "good",
Some(_) => "excellent",
}
}
/// Bins the class by the course's letter-grade scale.
///
/// # Arguments
///
/// * `policy` - the course policy, for its scale.
/// * `percents` - one score per student, out of 100.
///
/// # Returns
///
/// One row per band, highest first. Empty when the course sets no scale, which
/// is the signal for a report to fall back to ten-point bins.
fn grade_rows(policy: &crate::course::Policy, percents: &[f64]) -> Vec<GradeRow> {
let bands = policy.bands();
if bands.is_empty() || percents.is_empty() {
return Vec::new();
}
let n = percents.len() as f64;
let mut out: Vec<GradeRow> = Vec::with_capacity(bands.len());
let mut running = 0usize;
for (index, band) in bands.iter().enumerate() {
// The ceiling is the floor of the band above, less the smallest step a
// percentage is reported at, so the printed range reads the way a
// syllabus writes it.
let high = match index {
0 => 100.0,
_ => bands[index - 1].min - 0.1,
};
let count = percents
.iter()
.filter(|percent| {
**percent + 1e-9 >= band.min && (index == 0 || **percent < bands[index - 1].min)
})
.count();
running += count;
out.push(GradeRow {
letter: band.letter.clone(),
low: band.min,
high,
gpa: band.gpa,
attainment: band.attainment.clone(),
group: band.group_key(),
count,
share: count as f64 / n,
at_or_above: running,
});
}
out
}
/// Aggregates questions into per-lecture rows, worst first.
///
/// # Arguments
///
/// * `catalog` - the loaded course, for lecture titles and objective lectures.
/// * `questions` - the per-question rows.
/// * `objectives` - the per-objective rows, for the objective counts.
/// * `threshold` - the mastery threshold.
///
/// # Returns
///
/// One row per lecture that any scored item traced back to.
fn lecture_rows(
catalog: &Catalog,
questions: &[CohortQuestionRow],
objectives: &[CohortObjectiveRow],
threshold: f64,
) -> Vec<CohortLectureRow> {
let course = &catalog.course;
let mut items: BTreeMap<String, Vec<&CohortQuestionRow>> = BTreeMap::new();
let mut lecture_objectives: BTreeMap<String, BTreeSet<String>> = BTreeMap::new();
for question in questions {
// The same two routes the student report uses: the item knows which
// lecture it was written from, and the objective registry knows which
// lectures develop it.
let mut lectures: BTreeSet<String> = BTreeSet::new();
if let Some(entry) = question.item.as_deref().and_then(|uid| catalog.get(uid)) {
for source in &entry.item.sources {
lectures.insert(source.lecture.clone());
}
}
for objective in &question.objectives {
if let Some(entry) = course.learning_objectives.get(objective) {
lectures.extend(entry.lectures.iter().cloned());
}
}
for lecture in lectures {
items.entry(lecture.clone()).or_default().push(question);
lecture_objectives
.entry(lecture)
.or_default()
.extend(question.objectives.iter().cloned());
}
}
let mut out: Vec<CohortLectureRow> = items
.into_iter()
.map(|(lecture, rows)| {
let rate = rows.iter().map(|r| r.p_value).sum::<f64>() / rows.len() as f64;
let ids = lecture_objectives
.get(&lecture)
.cloned()
.unwrap_or_default();
let mine: Vec<&CohortObjectiveRow> =
objectives.iter().filter(|o| ids.contains(&o.id)).collect();
CohortLectureRow {
title: course
.lectures
.get(&lecture)
.map(|l| l.title.clone())
.unwrap_or_else(|| lecture.clone()),
n_items: rows.len(),
n_objectives: ids.len(),
n_objectives_below: mine.iter().filter(|o| o.rate < threshold).count(),
rate,
questions: rows.iter().map(|r| r.number).collect(),
// `objectives` arrives sorted worst first, so the first match is
// the weakest one under this lecture.
worst_objective: mine.first().map(|o| o.text.clone()),
lecture,
}
})
.collect();
out.sort_by(|a, b| {
a.rate
.partial_cmp(&b.rate)
.unwrap_or(std::cmp::Ordering::Equal)
.then_with(|| a.lecture.cmp(&b.lecture))
});
out
}
/// Summarizes how the authored expectations did.
fn prediction_summary(analysis: &Analysis) -> PredictionSummary {
let mut out = PredictionSummary::default();
let mut signed: Vec<f64> = Vec::new();
let mut biggest: Option<(u32, f64, f64)> = None;
for item in &analysis.items {
let Some(prediction) = &item.prediction else {
continue;
};
if prediction.calibrated {
out.n_calibrated += 1;
}
if let (Some(expected), Some(error)) = (prediction.expected_p, prediction.p_error) {
out.n_predicted += 1;
signed.push(error);
if prediction.p_within == Some(true) {
out.n_within += 1;
}
if biggest
.map(|(_, e, o)| (o - e).abs() < error.abs())
.unwrap_or(true)
{
biggest = Some((item.number, expected, item.p_value));
}
}
if prediction.expected_band.is_some() {
out.n_band += 1;
if prediction.band_hit == Some(true) {
out.n_band_hit += 1;
}
}
}
if !signed.is_empty() {
let n = signed.len() as f64;
out.mean_signed_error = Some(signed.iter().sum::<f64>() / n);
out.mean_abs_error = Some(signed.iter().map(|e| e.abs()).sum::<f64>() / n);
}
out.biggest_surprise = biggest;
out
}
/// Sorts every question into what to do about it.
///
/// The order of the tests is the order of severity, and the first match wins for
/// the three item decisions. `reteach` is judged separately, because "the item
/// worked and the class missed it" is not a competing diagnosis; it is a
/// different kind of finding.
///
/// # Arguments
///
/// * `questions` - the per-question rows.
/// * `threshold` - the mastery threshold, which sets what counts as a content
/// gap worth reteaching.
/// * `default_options` - the course's default option count, used for the chance
/// rate when an item's own options cannot be counted.
///
/// # Returns
///
/// The buckets.
fn triage(questions: &[CohortQuestionRow], threshold: f64, default_options: usize) -> Triage {
let mut out = Triage::default();
for question in questions {
let row = |reasons: Vec<String>, option: Option<&OptionRow>| TriageRow {
number: question.number,
item: question.item.clone(),
level: question.level,
p_value: question.p_value,
point_biserial: question.point_biserial,
discrimination: question.discrimination,
objectives: question.objective_texts.clone(),
taught_in: question.taught_in.clone(),
option: option.map(|o| o.letter.clone()),
option_share: option.map(|o| o.rate),
option_point_biserial: option.and_then(|o| o.point_biserial),
reasons,
};
let r = question.point_biserial;
let key_r = question
.options
.iter()
.filter(|o| o.is_key)
.filter_map(|o| o.point_biserial)
.fold(f64::NEG_INFINITY, f64::max);
// Count single letters only, so a multiple-response combination row such
// as `A+D` is not mistaken for a fifth option and does not deflate the
// chance rate.
let counted = question
.options
.iter()
.filter(|o| o.letter.chars().count() == 1)
.count();
let n_options = if counted >= 2 {
counted
} else {
default_options.max(2)
};
let chance = 1.0 / n_options as f64;
// The best-supported alternative: chosen by a fifth of the class or more,
// and correlating with total score at least as well as the key. The share
// matters because a defensible reading that two students found is a
// wording note, not a regrade.
let challenger = question
.options
.iter()
.filter(|o| !o.is_key && o.rate >= 0.20)
.filter(|o| o.point_biserial.unwrap_or(f64::NEG_INFINITY) > 0.0)
.filter(|o| {
!key_r.is_finite() || o.point_biserial.unwrap_or(f64::NEG_INFINITY) >= key_r
})
.max_by(|a, b| {
a.point_biserial
.unwrap_or(f64::NEG_INFINITY)
.partial_cmp(&b.point_biserial.unwrap_or(f64::NEG_INFINITY))
.unwrap_or(std::cmp::Ordering::Equal)
});
let mut placed = false;
// 1. Discard. Negative discrimination means the students who knew the
// material did worse on it, which no amount of rewording fixes after
// the fact; scores already awarded on it are noise.
if let Some(r) = r {
if r < -0.05 {
out.discard.push(row(
vec![format!(
"students who scored well overall did worse on this one (r = {r:+.2}). \
Whatever it measured, it was not what the rest of the exam measured."
)],
None,
));
placed = true;
} else if r < 0.05 && question.p_value <= chance + 0.05 {
out.discard.push(row(
vec![format!(
"{:.0}% correct against {:.0}% for guessing, and no relationship to total \
score (r = {r:+.2}). The responses are indistinguishable from random.",
question.p_value * 100.0,
chance * 100.0
)],
None,
));
placed = true;
}
}
// 2. Rekey or award partial credit.
if !placed {
if let Some(option) = challenger {
let mut reasons = vec![format!(
"option {} drew {:.0}% and tracks total score at least as well as the key \
({:+.2} against {:+.2}).",
option.letter,
option.rate * 100.0,
option.point_biserial.unwrap_or(0.0),
if key_r.is_finite() { key_r } else { 0.0 }
)];
if question.flags.iter().any(|f| f == "key_underperforms") {
reasons.push(
"the strongest students chose it more often than the key, which is the \
signature of two readings rather than of a guess."
.to_string(),
);
}
reasons.push(
"Either credit it for this administration or rewrite the stem to exclude it \
before reuse."
.to_string(),
);
out.rekey.push(row(reasons, Some(option)));
placed = true;
}
}
// 3. Revise, unless the weak discrimination is explained by difficulty.
if !placed {
let mut reasons: Vec<String> = Vec::new();
let weak = r.map(|r| r < 0.20).unwrap_or(true);
let bounded = weak && (question.p_value >= 0.85 || question.p_value <= 0.20);
if weak && !bounded {
reasons.push(format!(
"at {:.0}% correct the item had room to separate students and did not \
(r = {}).",
question.p_value * 100.0,
r.map(|r| format!("{r:+.2}"))
.unwrap_or_else(|| "n/a".into())
));
}
let dead: Vec<&OptionRow> = question
.options
.iter()
.filter(|o| o.nonfunctioning)
.collect();
if !dead.is_empty() {
reasons.push(format!(
"option{} {} drew almost nobody, so the item is really a {}-way choice.",
if dead.len() == 1 { "" } else { "s" },
dead.iter()
.map(|o| o.letter.as_str())
.collect::<Vec<_>>()
.join(", "),
n_options.saturating_sub(dead.len()).max(2)
));
}
if question.flags.iter().any(|f| f == "ambiguous") {
reasons.push(
"partial credit was awarded at grading time, which is a record that the item \
admitted more than one reading."
.to_string(),
);
}
if bounded {
out.bounded.push(row(
vec![format!(
"{:.0}% correct leaves little variance to correlate with, so r = {} is \
what this difficulty allows rather than a fault.",
question.p_value * 100.0,
r.map(|r| format!("{r:+.2}"))
.unwrap_or_else(|| "n/a".into())
)],
None,
));
placed = true;
} else if !reasons.is_empty() {
out.revise.push(row(reasons, None));
placed = true;
}
}
// 4. Reteach: the item did its job and the class still missed it. Judged
// independently of the three above.
let works = r.map(|r| r >= 0.20).unwrap_or(false);
if works && question.p_value < threshold {
out.reteach.push(row(
vec![format!(
"the item separated students cleanly (r = {}) and {:.0}% still missed it, so \
this is a gap in what the class knows rather than a fault in the question.",
r.map(|r| format!("{r:+.2}"))
.unwrap_or_else(|| "n/a".into()),
(1.0 - question.p_value) * 100.0
)],
None,
));
}
if !placed {
out.clean += 1;
}
}
// Worst first inside each bucket, so the top of every list is where to start.
for bucket in [
&mut out.discard,
&mut out.rekey,
&mut out.revise,
&mut out.bounded,
] {
bucket.sort_by(|a, b| {
a.point_biserial
.unwrap_or(1.0)
.partial_cmp(&b.point_biserial.unwrap_or(1.0))
.unwrap_or(std::cmp::Ordering::Equal)
.then_with(|| a.number.cmp(&b.number))
});
}
out.reteach.sort_by(|a, b| {
a.p_value
.partial_cmp(&b.p_value)
.unwrap_or(std::cmp::Ordering::Equal)
.then_with(|| a.number.cmp(&b.number))
});
out
}
/// Per-question proportion correct, split by form. /// Per-question proportion correct, split by form.
fn per_form_p_values(set: &ResponseSet) -> BTreeMap<u32, BTreeMap<String, f64>> { fn per_form_p_values(set: &ResponseSet) -> BTreeMap<u32, BTreeMap<String, f64>> {
let forms: BTreeSet<&str> = set.rows.iter().filter_map(|r| r.form.as_deref()).collect(); let forms: BTreeSet<&str> = set.rows.iter().filter_map(|r| r.form.as_deref()).collect();
+37
View File
@@ -261,6 +261,43 @@ fn policy_schema() -> Value {
"minimum": 1, "minimum": 1,
"description": "Below this many items on an objective, reports say 'not enough \ "description": "Below this many items on an objective, reports say 'not enough \
evidence' rather than classifying." evidence' rather than classifying."
},
"grade_scale": {
"type": "array",
"description": "Letter-grade bands. Only the lower bound of each is recorded; a \
band runs up to the next one. Set this and a class report bins \
scores by letter rather than by ten-point interval.",
"items": {
"type": "object",
"required": ["letter", "min"],
"additionalProperties": false,
"properties": {
"letter": {
"type": "string",
"description": "The letter as it appears on a transcript."
},
"min": {
"type": "number",
"minimum": 0,
"maximum": 100,
"description": "Lowest percentage earning this letter, inclusive."
},
"gpa": {
"type": "number",
"minimum": 0,
"description": "Grade points the band carries."
},
"attainment": {
"type": "string",
"description": "The attainment word attached to the band."
},
"group": {
"type": "string",
"description": "Colour group for reports; defaults to the letter's \
first character."
}
}
}
} }
} }
}) })
+177
View File
@@ -549,6 +549,105 @@ pub fn cohort_value(diagnostic: &CohortDiagnostic, config: &RenderConfig) -> Val
), ),
); );
out.insert(
"grades",
Value::Array(
diagnostic
.grades
.iter()
.map(|grade| {
let mut value = Value::dict();
value.insert("letter", Value::str(&grade.letter));
value.insert("low", Value::Float(grade.low));
value.insert("high", Value::Float(grade.high));
value.insert_some("gpa", grade.gpa.map(Value::Float));
value.insert_some("attainment", grade.attainment.as_ref().map(Value::str));
value.insert("group", Value::str(&grade.group));
value.insert("count", Value::Int(grade.count as i64));
value.insert("share", Value::Float(grade.share));
value.insert("at-or-above", Value::Int(grade.at_or_above as i64));
value
})
.collect(),
),
);
out.insert(
"lectures",
Value::Array(
diagnostic
.lectures
.iter()
.map(|lecture| {
let mut value = Value::dict();
value.insert("lecture", Value::str(&lecture.lecture));
value.insert("title", Value::str(&lecture.title));
value.insert("items", Value::Int(lecture.n_items as i64));
value.insert("objectives", Value::Int(lecture.n_objectives as i64));
value.insert(
"objectives-below",
Value::Int(lecture.n_objectives_below as i64),
);
value.insert("rate", Value::Float(lecture.rate));
value.insert(
"questions",
Value::Array(
lecture
.questions
.iter()
.map(|n| Value::Int(*n as i64))
.collect(),
),
);
value.insert_some(
"worst-objective",
lecture
.worst_objective
.as_ref()
.map(|text| markup_value(text, content)),
);
value
})
.collect(),
),
);
let triage_rows = |rows: &[crate::diagnostic::TriageRow]| -> Value {
Value::Array(rows.iter().map(|row| triage_value(row, content)).collect())
};
let mut triage = Value::dict();
triage.insert("discard", triage_rows(&diagnostic.triage.discard));
triage.insert("rekey", triage_rows(&diagnostic.triage.rekey));
triage.insert("revise", triage_rows(&diagnostic.triage.revise));
triage.insert("reteach", triage_rows(&diagnostic.triage.reteach));
triage.insert("bounded", triage_rows(&diagnostic.triage.bounded));
triage.insert("clean", Value::Int(diagnostic.triage.clean as i64));
out.insert("triage", triage);
let predictions = &diagnostic.predictions;
let mut prediction = Value::dict();
prediction.insert("predicted", Value::Int(predictions.n_predicted as i64));
prediction.insert("calibrated", Value::Int(predictions.n_calibrated as i64));
prediction.insert_some(
"mean-signed-error",
predictions.mean_signed_error.map(Value::Float),
);
prediction.insert_some(
"mean-abs-error",
predictions.mean_abs_error.map(Value::Float),
);
prediction.insert("within", Value::Int(predictions.n_within as i64));
prediction.insert("band", Value::Int(predictions.n_band as i64));
prediction.insert("band-hit", Value::Int(predictions.n_band_hit as i64));
if let Some((number, expected, observed)) = predictions.biggest_surprise {
let mut surprise = Value::dict();
surprise.insert("number", Value::Int(number as i64));
surprise.insert("expected", Value::Float(expected));
surprise.insert("observed", Value::Float(observed));
prediction.insert("biggest-surprise", surprise);
}
out.insert("predictions", prediction);
out.insert( out.insert(
"forms", "forms",
Value::Array( Value::Array(
@@ -601,6 +700,46 @@ pub fn cohort_value(diagnostic: &CohortDiagnostic, config: &RenderConfig) -> Val
out out
} }
/// One triage row as a Typst value.
fn triage_value(row: &crate::diagnostic::TriageRow, content: bool) -> Value {
let mut value = Value::dict();
value.insert("number", Value::Int(row.number as i64));
value.insert_some("item", row.item.as_ref().map(Value::str));
value.insert_some("level", row.level.map(|l| Value::Int(l as i64)));
value.insert("p", Value::Float(row.p_value));
value.insert_some("point-biserial", row.point_biserial.map(Value::Float));
value.insert_some("discrimination", row.discrimination.map(Value::Float));
value.insert(
"objectives",
Value::Array(
row.objectives
.iter()
.map(|text| markup_value(text, content))
.collect(),
),
);
value.insert(
"taught-in",
Value::Array(row.taught_in.iter().map(|t| Value::str(t)).collect()),
);
value.insert_some("option", row.option.as_ref().map(Value::str));
value.insert_some("option-share", row.option_share.map(Value::Float));
value.insert_some(
"option-point-biserial",
row.option_point_biserial.map(Value::Float),
);
value.insert(
"reasons",
Value::Array(
row.reasons
.iter()
.map(|reason| markup_value(reason, content))
.collect(),
),
);
value
}
/// One histogram bin as a Typst value. /// One histogram bin as a Typst value.
fn bin_value(bin: &Bin) -> Value { fn bin_value(bin: &Bin) -> Value {
let mut value = Value::dict(); let mut value = Value::dict();
@@ -635,6 +774,29 @@ fn cohort_question_value(question: &CohortQuestionRow, content: bool) -> Value {
"objectives", "objectives",
Value::Array(question.objectives.iter().map(|o| Value::str(o)).collect()), Value::Array(question.objectives.iter().map(|o| Value::str(o)).collect()),
); );
value.insert(
"objective-texts",
Value::Array(
question
.objective_texts
.iter()
.map(|text| markup_value(text, content))
.collect(),
),
);
value.insert(
"taught-in",
Value::Array(question.taught_in.iter().map(|t| Value::str(t)).collect()),
);
value.insert(
"lectures",
Value::Array(question.lectures.iter().map(|l| Value::str(l)).collect()),
);
value.insert("difficulty-band", Value::str(&question.difficulty_band));
value.insert(
"discrimination-band",
Value::str(&question.discrimination_band),
);
value.insert("p", Value::Float(question.p_value)); value.insert("p", Value::Float(question.p_value));
value.insert_some("point-biserial", question.point_biserial.map(Value::Float)); value.insert_some("point-biserial", question.point_biserial.map(Value::Float));
value.insert_some("discrimination", question.discrimination.map(Value::Float)); value.insert_some("discrimination", question.discrimination.map(Value::Float));
@@ -676,6 +838,17 @@ fn cohort_question_value(question: &CohortQuestionRow, content: bool) -> Value {
.collect(), .collect(),
), ),
); );
value.insert(
"prediction-notes",
Value::Array(
question
.prediction_notes
.iter()
.map(|n| markup_value(n, content))
.collect(),
),
);
value.insert("calibrated", Value::Bool(question.calibrated));
let mut by_form = Value::dict(); let mut by_form = Value::dict();
for (form, p) in &question.by_form { for (form, p) in &question.by_form {
by_form.insert(form.clone(), Value::Float(*p)); by_form.insert(form.clone(), Value::Float(*p));
@@ -953,7 +1126,11 @@ mod tests {
levels: Vec::new(), levels: Vec::new(),
objectives: Vec::new(), objectives: Vec::new(),
gaps: Vec::new(), gaps: Vec::new(),
grades: Vec::new(),
lectures: Vec::new(),
questions: Vec::new(), questions: Vec::new(),
triage: crate::diagnostic::Triage::default(),
predictions: crate::diagnostic::PredictionSummary::default(),
revise: Vec::new(), revise: Vec::new(),
forms: Vec::new(), forms: Vec::new(),
blueprint: Vec::new(), blueprint: Vec::new(),
File diff suppressed because it is too large Load Diff
+129
View File
@@ -137,6 +137,58 @@ pub struct Policy {
/// The fewest items on an objective before a report will call it mastered. /// The fewest items on an objective before a report will call it mastered.
#[serde(default = "two_usize")] #[serde(default = "two_usize")]
pub min_items_for_mastery: usize, pub min_items_for_mastery: usize,
/// The letter-grade bands, highest first or in any order.
///
/// Empty by default, because a grading scale belongs to a course rather than
/// to a tool. When it is set, a class report bins the score distribution by
/// letter instead of by ten-point interval, which is the only binning a
/// student or an instructor actually acts on.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub grade_scale: Vec<GradeBand>,
}
/// One letter-grade band.
///
/// Only the lower bound is recorded. An upper bound would be a second copy of
/// the next band's lower bound, and the two would eventually disagree: a scale
/// written as `93.0 - 96.9` leaves 96.95 in no band at all. Bands are read as
/// "this letter or better from here up", so the top band needs no ceiling.
#[derive(Debug, Clone, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct GradeBand {
/// The letter as it appears on a transcript.
pub letter: String,
/// The lowest percentage that earns it, inclusive.
pub min: f64,
/// The grade points it carries, when the course records them.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub gpa: Option<f64>,
/// The attainment word attached to the band, such as `Meritorious`.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub attainment: Option<String>,
/// A colour group, so a report can tint A bands alike without parsing
/// letters. Defaults to the letter's first character.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub group: Option<String>,
}
impl GradeBand {
/// The group a band belongs to: its own `group`, else its first character.
///
/// # Returns
///
/// An uppercase group key such as `A`.
pub fn group_key(&self) -> String {
match &self.group {
Some(group) => group.to_ascii_uppercase(),
None => self
.letter
.chars()
.next()
.map(|c| c.to_ascii_uppercase().to_string())
.unwrap_or_default(),
}
}
} }
impl Default for Policy { impl Default for Policy {
@@ -149,10 +201,44 @@ impl Default for Policy {
partial_credit_floor_level: None, partial_credit_floor_level: None,
mastery_threshold: mastery_default(), mastery_threshold: mastery_default(),
min_items_for_mastery: 2, min_items_for_mastery: 2,
grade_scale: Vec::new(),
} }
} }
} }
impl Policy {
/// The grade bands, highest lower bound first.
///
/// # Returns
///
/// The bands in descending order, empty when the course sets no scale.
pub fn bands(&self) -> Vec<&GradeBand> {
let mut out: Vec<&GradeBand> = self.grade_scale.iter().collect();
out.sort_by(|a, b| {
b.min
.partial_cmp(&a.min)
.unwrap_or(std::cmp::Ordering::Equal)
});
out
}
/// The band a percentage falls in.
///
/// # Arguments
///
/// * `percent` - a score out of 100.
///
/// # Returns
///
/// The band, or `None` when the course sets no scale or the score sits below
/// every band in it.
pub fn band_for(&self, percent: f64) -> Option<&GradeBand> {
self.bands()
.into_iter()
.find(|band| percent + 1e-9 >= band.min)
}
}
/// A unit or module of the course. /// A unit or module of the course.
#[derive(Debug, Clone, Serialize, Deserialize)] #[derive(Debug, Clone, Serialize, Deserialize)]
#[serde(deny_unknown_fields)] #[serde(deny_unknown_fields)]
@@ -622,6 +708,49 @@ impl CourseFile {
)); ));
} }
// A scale with a hole in it silently drops students into no band at all,
// and the report would show a distribution that does not sum to the
// class. Cheaper to say so here.
let mut seen_letters: BTreeMap<&str, usize> = BTreeMap::new();
let mut seen_mins: Vec<f64> = Vec::new();
for band in &self.policy.grade_scale {
*seen_letters.entry(band.letter.as_str()).or_insert(0) += 1;
if !(0.0..=100.0).contains(&band.min) {
issues.push(format!(
"policy.grade_scale: band `{}` has min {}, which is not a percentage",
band.letter, band.min
));
}
if seen_mins.iter().any(|m| (m - band.min).abs() < 1e-9) {
issues.push(format!(
"policy.grade_scale: two bands start at {}%, so the lower one is unreachable",
band.min
));
}
seen_mins.push(band.min);
}
for (letter, n) in &seen_letters {
if *n > 1 {
issues.push(format!(
"policy.grade_scale: duplicate letter `{letter}` declared {n} times"
));
}
}
if !self.policy.grade_scale.is_empty() {
let lowest = self
.policy
.bands()
.last()
.map(|b| b.min)
.unwrap_or(f64::INFINITY);
if lowest > 0.0 {
issues.push(format!(
"policy.grade_scale: the lowest band starts at {lowest}%, so a score below \
that falls in no band. Give the failing grade a min of 0."
));
}
}
let unit_ids: Vec<&String> = self.units.iter().map(|u| &u.id).collect(); let unit_ids: Vec<&String> = self.units.iter().map(|u| &u.id).collect();
let mut unit_counts: BTreeMap<&str, usize> = BTreeMap::new(); let mut unit_counts: BTreeMap<&str, usize> = BTreeMap::new();
for u in &self.units { for u in &self.units {