refactor: improve cohort report
This commit is contained in:
@@ -917,8 +917,19 @@ pub struct CohortDiagnostic {
|
||||
pub objectives: Vec<CohortObjectiveRow>,
|
||||
/// Objectives the class as a whole did not meet.
|
||||
pub gaps: Vec<CohortObjectiveRow>,
|
||||
/// The distribution binned by the course's letter-grade scale. Empty when
|
||||
/// `course.yaml` sets no scale, in which case the ten-point bins stand.
|
||||
#[serde(skip_serializing_if = "Vec::is_empty")]
|
||||
pub grades: Vec<GradeRow>,
|
||||
/// Per-lecture class performance, worst first.
|
||||
#[serde(skip_serializing_if = "Vec::is_empty")]
|
||||
pub lectures: Vec<CohortLectureRow>,
|
||||
/// Per-question statistics.
|
||||
pub questions: Vec<CohortQuestionRow>,
|
||||
/// What to do about each question that raised something.
|
||||
pub triage: Triage,
|
||||
/// How the authored expectations did.
|
||||
pub predictions: PredictionSummary,
|
||||
/// Questions worth revisiting before reuse, worst first.
|
||||
pub revise: Vec<CohortQuestionRow>,
|
||||
/// One row per form, when more than one was given.
|
||||
@@ -1036,6 +1047,20 @@ pub struct CohortQuestionRow {
|
||||
pub level: Option<u8>,
|
||||
/// The objectives it measured.
|
||||
pub objectives: Vec<String>,
|
||||
/// What those objectives ask, in the course's own words.
|
||||
#[serde(skip_serializing_if = "Vec::is_empty")]
|
||||
pub objective_texts: Vec<String>,
|
||||
/// Where the item was taught, as lecture titles and slide numbers.
|
||||
#[serde(skip_serializing_if = "Vec::is_empty")]
|
||||
pub taught_in: Vec<String>,
|
||||
/// The lecture ids alone, for a table column where only `L1.4` fits.
|
||||
#[serde(skip_serializing_if = "Vec::is_empty")]
|
||||
pub lectures: Vec<String>,
|
||||
/// Which difficulty band it fell in: `too easy`, `moderate`, or `hard`.
|
||||
pub difficulty_band: String,
|
||||
/// Which discrimination band it fell in, on the conventional cut points:
|
||||
/// `excellent`, `good`, `marginal`, `poor`, or `negative`.
|
||||
pub discrimination_band: String,
|
||||
/// Proportion correct.
|
||||
pub p_value: f64,
|
||||
/// Corrected item-total point-biserial.
|
||||
@@ -1054,6 +1079,13 @@ pub struct CohortQuestionRow {
|
||||
pub flags: Vec<String>,
|
||||
/// What those flags mean.
|
||||
pub notes: Vec<String>,
|
||||
/// How the item did against its author's expectation. Separate from `notes`
|
||||
/// because an unmet prediction on an uncalibrated item is a fact about the
|
||||
/// prediction.
|
||||
#[serde(skip_serializing_if = "Vec::is_empty")]
|
||||
pub prediction_notes: Vec<String>,
|
||||
/// Whether that expectation rested on a prior calibration.
|
||||
pub calibrated: bool,
|
||||
/// Per-form proportion correct, when more than one form was given. A gap here
|
||||
/// on one question, with the rest of the exam in step, points at that
|
||||
/// question's permutation rather than at the cohort.
|
||||
@@ -1106,6 +1138,159 @@ pub struct PatternRow {
|
||||
pub level_means: BTreeMap<u8, f64>,
|
||||
}
|
||||
|
||||
/// The class's scores binned by the course's own letter-grade scale.
|
||||
///
|
||||
/// A ten-point histogram is the default because it needs no course
|
||||
/// configuration, but nobody acts on "nineteen students in the fifties". They
|
||||
/// act on "nineteen students are failing", and that sentence needs the scale
|
||||
/// from `course.yaml`.
|
||||
#[derive(Debug, Clone, Serialize)]
|
||||
#[serde(rename_all = "kebab-case")]
|
||||
pub struct GradeRow {
|
||||
/// The letter.
|
||||
pub letter: String,
|
||||
/// The lowest percentage in the band.
|
||||
pub low: f64,
|
||||
/// The highest percentage in the band, which is just under the next band's
|
||||
/// floor, or 100 for the top band.
|
||||
pub high: f64,
|
||||
/// Grade points, when the scale records them.
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub gpa: Option<f64>,
|
||||
/// The attainment word, when the scale records one.
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub attainment: Option<String>,
|
||||
/// The colour group, so A, A- and A+ can be tinted together.
|
||||
pub group: String,
|
||||
/// How many students landed in the band.
|
||||
pub count: usize,
|
||||
/// Their share of the class, in `0.0..=1.0`.
|
||||
pub share: f64,
|
||||
/// How many students are in this band or a higher one.
|
||||
pub at_or_above: usize,
|
||||
}
|
||||
|
||||
/// One lecture's showing, aggregated from the items written against it.
|
||||
///
|
||||
/// The objective table answers "which objective went wrong". This answers "which
|
||||
/// class meeting went wrong", which is the question that maps onto next week.
|
||||
#[derive(Debug, Clone, Serialize)]
|
||||
#[serde(rename_all = "kebab-case")]
|
||||
pub struct CohortLectureRow {
|
||||
/// The lecture id.
|
||||
pub lecture: String,
|
||||
/// Its title.
|
||||
pub title: String,
|
||||
/// How many scored items traced back to it.
|
||||
pub n_items: usize,
|
||||
/// How many distinct objectives those items measured.
|
||||
pub n_objectives: usize,
|
||||
/// How many of those objectives the class did not meet.
|
||||
pub n_objectives_below: usize,
|
||||
/// Mean proportion correct across its items.
|
||||
pub rate: f64,
|
||||
/// The questions, so the row can be checked against the item table.
|
||||
pub questions: Vec<u32>,
|
||||
/// The worst objective under this lecture, by class rate.
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub worst_objective: Option<String>,
|
||||
}
|
||||
|
||||
/// What to do about one question, and why.
|
||||
#[derive(Debug, Clone, Serialize)]
|
||||
#[serde(rename_all = "kebab-case")]
|
||||
pub struct TriageRow {
|
||||
/// The question number.
|
||||
pub number: u32,
|
||||
/// The item id.
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub item: Option<String>,
|
||||
/// The level code.
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub level: Option<u8>,
|
||||
/// Proportion correct.
|
||||
pub p_value: f64,
|
||||
/// Corrected item-total correlation.
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub point_biserial: Option<f64>,
|
||||
/// Upper minus lower group.
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub discrimination: Option<f64>,
|
||||
/// What the question measured, in the course's words.
|
||||
#[serde(skip_serializing_if = "Vec::is_empty")]
|
||||
pub objectives: Vec<String>,
|
||||
/// Where it was taught.
|
||||
#[serde(skip_serializing_if = "Vec::is_empty")]
|
||||
pub taught_in: Vec<String>,
|
||||
/// The specific option this recommendation is about, when it is about one.
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub option: Option<String>,
|
||||
/// That option's share of responses.
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub option_share: Option<f64>,
|
||||
/// That option's correlation with total score.
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub option_point_biserial: Option<f64>,
|
||||
/// The evidence, one clause per line.
|
||||
pub reasons: Vec<String>,
|
||||
}
|
||||
|
||||
/// Every question sorted into what to do with it.
|
||||
///
|
||||
/// The first four lists are decisions about items and are mutually exclusive: a
|
||||
/// question appears in the most severe one that fits, because there is no point
|
||||
/// rewriting a distractor on an item you are about to discard. `reteach` is not
|
||||
/// a decision about an item at all, so a question can appear there as well as in
|
||||
/// one of the others.
|
||||
#[derive(Debug, Clone, Default, Serialize)]
|
||||
#[serde(rename_all = "kebab-case")]
|
||||
pub struct Triage {
|
||||
/// Broken: the evidence says these did not measure what they were scored on.
|
||||
pub discard: Vec<TriageRow>,
|
||||
/// A second defensible answer with statistical support behind it.
|
||||
pub rekey: Vec<TriageRow>,
|
||||
/// Weak but salvageable, worth rewriting before reuse.
|
||||
pub revise: Vec<TriageRow>,
|
||||
/// Sound items the class got wrong. A teaching finding, not an item finding.
|
||||
pub reteach: Vec<TriageRow>,
|
||||
/// Items whose low discrimination is explained by their difficulty rather
|
||||
/// than by a fault. Listed so they are not mistaken for work to do.
|
||||
pub bounded: Vec<TriageRow>,
|
||||
/// How many questions raised nothing at all.
|
||||
pub clean: usize,
|
||||
}
|
||||
|
||||
/// How the authored expectations did against the data.
|
||||
///
|
||||
/// This exists so that an uncalibrated bank does not produce one
|
||||
/// `design_mismatch` per item. Before an item has data, its expected difficulty
|
||||
/// is a prediction by its author, and the useful summary is whether those
|
||||
/// predictions run optimistic or pessimistic as a set.
|
||||
#[derive(Debug, Clone, Default, Serialize)]
|
||||
#[serde(rename_all = "kebab-case")]
|
||||
pub struct PredictionSummary {
|
||||
/// How many items recorded an expected difficulty.
|
||||
pub n_predicted: usize,
|
||||
/// How many of those expectations rest on a prior calibration.
|
||||
pub n_calibrated: usize,
|
||||
/// Mean of observed minus expected difficulty. Positive means the items came
|
||||
/// out easier than predicted.
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub mean_signed_error: Option<f64>,
|
||||
/// Mean absolute difficulty error, which is the size of a typical miss.
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub mean_abs_error: Option<f64>,
|
||||
/// How many landed inside the tolerance.
|
||||
pub n_within: usize,
|
||||
/// How many items recorded an expected discrimination band.
|
||||
pub n_band: usize,
|
||||
/// How many of those landed inside it.
|
||||
pub n_band_hit: usize,
|
||||
/// The largest single surprise, as `(question, expected, observed)`.
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub biggest_surprise: Option<(u32, f64, f64)>,
|
||||
}
|
||||
|
||||
/// Builds the class diagnostic.
|
||||
///
|
||||
/// # Arguments
|
||||
@@ -1234,6 +1419,29 @@ pub fn cohort(
|
||||
item: item.item_ref.clone(),
|
||||
level: meta.and_then(|m| m.0),
|
||||
objectives: meta.map(|m| m.1.clone()).unwrap_or_default(),
|
||||
objective_texts: meta
|
||||
.map(|m| m.1.iter().map(|id| course.objective_text(id)).collect())
|
||||
.unwrap_or_default(),
|
||||
taught_in: item
|
||||
.item_ref
|
||||
.as_deref()
|
||||
.map(|uid| taught_in(catalog, uid))
|
||||
.unwrap_or_default(),
|
||||
lectures: item
|
||||
.item_ref
|
||||
.as_deref()
|
||||
.and_then(|uid| catalog.get(uid))
|
||||
.map(|entry| {
|
||||
entry
|
||||
.item
|
||||
.sources
|
||||
.iter()
|
||||
.map(|source| source.lecture.clone())
|
||||
.collect()
|
||||
})
|
||||
.unwrap_or_default(),
|
||||
difficulty_band: difficulty_band(item.p_value).to_string(),
|
||||
discrimination_band: discrimination_band(item.point_biserial).to_string(),
|
||||
p_value: item.p_value,
|
||||
point_biserial: item.point_biserial,
|
||||
discrimination: item.discrimination_index,
|
||||
@@ -1253,6 +1461,12 @@ pub fn cohort(
|
||||
.collect(),
|
||||
flags: item.flags.iter().map(|f| f.as_str().to_string()).collect(),
|
||||
notes: item.notes.clone(),
|
||||
prediction_notes: item
|
||||
.prediction
|
||||
.as_ref()
|
||||
.map(|p| p.notes.clone())
|
||||
.unwrap_or_default(),
|
||||
calibrated: item.prediction.as_ref().is_some_and(|p| p.calibrated),
|
||||
by_form: by_form.get(&item.number).cloned().unwrap_or_default(),
|
||||
}
|
||||
})
|
||||
@@ -1264,10 +1478,20 @@ pub fn cohort(
|
||||
.filter_map(|item| questions.iter().find(|q| q.number == item.number).cloned())
|
||||
.collect();
|
||||
|
||||
let default_options = course.policy.options_per_item;
|
||||
let triage = triage(&questions, threshold, default_options);
|
||||
let predictions = prediction_summary(analysis);
|
||||
let grades = grade_rows(&course.policy, &percents);
|
||||
let lectures = lecture_rows(catalog, &questions, &objectives, threshold);
|
||||
|
||||
CohortDiagnostic {
|
||||
n_students: cohort.students.len(),
|
||||
n_items: analysis.reliability.n_items,
|
||||
distribution: distribution(&percents),
|
||||
grades,
|
||||
lectures,
|
||||
triage,
|
||||
predictions,
|
||||
reliability: ReliabilityRow {
|
||||
alpha: analysis.reliability.alpha,
|
||||
sem: analysis.reliability.sem,
|
||||
@@ -1299,6 +1523,496 @@ pub fn cohort(
|
||||
}
|
||||
}
|
||||
|
||||
/// Where an item was taught, as lecture titles with slide numbers.
|
||||
///
|
||||
/// # Arguments
|
||||
///
|
||||
/// * `catalog` - the loaded course.
|
||||
/// * `uid` - the item's global id.
|
||||
///
|
||||
/// # Returns
|
||||
///
|
||||
/// One entry per source the item records.
|
||||
fn taught_in(catalog: &Catalog, uid: &str) -> Vec<String> {
|
||||
let Some(entry) = catalog.get(uid) else {
|
||||
return Vec::new();
|
||||
};
|
||||
entry
|
||||
.item
|
||||
.sources
|
||||
.iter()
|
||||
.map(|source| {
|
||||
let title = catalog
|
||||
.course
|
||||
.lectures
|
||||
.get(&source.lecture)
|
||||
.map(|l| l.title.clone())
|
||||
.unwrap_or_else(|| source.lecture.clone());
|
||||
if source.slides.is_empty() {
|
||||
format!("{} ({})", title, source.lecture)
|
||||
} else {
|
||||
let slides: Vec<String> = source.slides.iter().map(|s| s.to_string()).collect();
|
||||
format!(
|
||||
"{} ({}), slide{} {}",
|
||||
title,
|
||||
source.lecture,
|
||||
if source.slides.len() == 1 { "" } else { "s" },
|
||||
slides.join(", ")
|
||||
)
|
||||
}
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// The difficulty band a p-value falls in.
|
||||
///
|
||||
/// Three bands rather than five. The only distinction that changes what you do
|
||||
/// is whether the item had room to discriminate at all, and that is a question
|
||||
/// about the middle versus the two ends.
|
||||
fn difficulty_band(p: f64) -> &'static str {
|
||||
if p >= 0.85 {
|
||||
"too easy"
|
||||
} else if p <= 0.35 {
|
||||
"hard"
|
||||
} else {
|
||||
"moderate"
|
||||
}
|
||||
}
|
||||
|
||||
/// The discrimination band a point-biserial falls in.
|
||||
///
|
||||
/// The cut points are the conventional ones from the item-analysis literature,
|
||||
/// usually attributed to Ebel: about 0.40 and above is excellent, 0.30 to 0.39
|
||||
/// good, 0.20 to 0.29 marginal, and below 0.20 poor. They are rules of thumb
|
||||
/// rather than laws, and they must be read next to difficulty, because an item
|
||||
/// almost everyone passes or fails has little variance left to correlate with
|
||||
/// anything.
|
||||
fn discrimination_band(r: Option<f64>) -> &'static str {
|
||||
match r {
|
||||
None => "no variance",
|
||||
Some(r) if r < 0.0 => "negative",
|
||||
Some(r) if r < 0.20 => "poor",
|
||||
Some(r) if r < 0.30 => "marginal",
|
||||
Some(r) if r < 0.40 => "good",
|
||||
Some(_) => "excellent",
|
||||
}
|
||||
}
|
||||
|
||||
/// Bins the class by the course's letter-grade scale.
|
||||
///
|
||||
/// # Arguments
|
||||
///
|
||||
/// * `policy` - the course policy, for its scale.
|
||||
/// * `percents` - one score per student, out of 100.
|
||||
///
|
||||
/// # Returns
|
||||
///
|
||||
/// One row per band, highest first. Empty when the course sets no scale, which
|
||||
/// is the signal for a report to fall back to ten-point bins.
|
||||
fn grade_rows(policy: &crate::course::Policy, percents: &[f64]) -> Vec<GradeRow> {
|
||||
let bands = policy.bands();
|
||||
if bands.is_empty() || percents.is_empty() {
|
||||
return Vec::new();
|
||||
}
|
||||
|
||||
let n = percents.len() as f64;
|
||||
let mut out: Vec<GradeRow> = Vec::with_capacity(bands.len());
|
||||
let mut running = 0usize;
|
||||
for (index, band) in bands.iter().enumerate() {
|
||||
// The ceiling is the floor of the band above, less the smallest step a
|
||||
// percentage is reported at, so the printed range reads the way a
|
||||
// syllabus writes it.
|
||||
let high = match index {
|
||||
0 => 100.0,
|
||||
_ => bands[index - 1].min - 0.1,
|
||||
};
|
||||
let count = percents
|
||||
.iter()
|
||||
.filter(|percent| {
|
||||
**percent + 1e-9 >= band.min && (index == 0 || **percent < bands[index - 1].min)
|
||||
})
|
||||
.count();
|
||||
running += count;
|
||||
out.push(GradeRow {
|
||||
letter: band.letter.clone(),
|
||||
low: band.min,
|
||||
high,
|
||||
gpa: band.gpa,
|
||||
attainment: band.attainment.clone(),
|
||||
group: band.group_key(),
|
||||
count,
|
||||
share: count as f64 / n,
|
||||
at_or_above: running,
|
||||
});
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
/// Aggregates questions into per-lecture rows, worst first.
|
||||
///
|
||||
/// # Arguments
|
||||
///
|
||||
/// * `catalog` - the loaded course, for lecture titles and objective lectures.
|
||||
/// * `questions` - the per-question rows.
|
||||
/// * `objectives` - the per-objective rows, for the objective counts.
|
||||
/// * `threshold` - the mastery threshold.
|
||||
///
|
||||
/// # Returns
|
||||
///
|
||||
/// One row per lecture that any scored item traced back to.
|
||||
fn lecture_rows(
|
||||
catalog: &Catalog,
|
||||
questions: &[CohortQuestionRow],
|
||||
objectives: &[CohortObjectiveRow],
|
||||
threshold: f64,
|
||||
) -> Vec<CohortLectureRow> {
|
||||
let course = &catalog.course;
|
||||
let mut items: BTreeMap<String, Vec<&CohortQuestionRow>> = BTreeMap::new();
|
||||
let mut lecture_objectives: BTreeMap<String, BTreeSet<String>> = BTreeMap::new();
|
||||
|
||||
for question in questions {
|
||||
// The same two routes the student report uses: the item knows which
|
||||
// lecture it was written from, and the objective registry knows which
|
||||
// lectures develop it.
|
||||
let mut lectures: BTreeSet<String> = BTreeSet::new();
|
||||
if let Some(entry) = question.item.as_deref().and_then(|uid| catalog.get(uid)) {
|
||||
for source in &entry.item.sources {
|
||||
lectures.insert(source.lecture.clone());
|
||||
}
|
||||
}
|
||||
for objective in &question.objectives {
|
||||
if let Some(entry) = course.learning_objectives.get(objective) {
|
||||
lectures.extend(entry.lectures.iter().cloned());
|
||||
}
|
||||
}
|
||||
for lecture in lectures {
|
||||
items.entry(lecture.clone()).or_default().push(question);
|
||||
lecture_objectives
|
||||
.entry(lecture)
|
||||
.or_default()
|
||||
.extend(question.objectives.iter().cloned());
|
||||
}
|
||||
}
|
||||
|
||||
let mut out: Vec<CohortLectureRow> = items
|
||||
.into_iter()
|
||||
.map(|(lecture, rows)| {
|
||||
let rate = rows.iter().map(|r| r.p_value).sum::<f64>() / rows.len() as f64;
|
||||
let ids = lecture_objectives
|
||||
.get(&lecture)
|
||||
.cloned()
|
||||
.unwrap_or_default();
|
||||
let mine: Vec<&CohortObjectiveRow> =
|
||||
objectives.iter().filter(|o| ids.contains(&o.id)).collect();
|
||||
CohortLectureRow {
|
||||
title: course
|
||||
.lectures
|
||||
.get(&lecture)
|
||||
.map(|l| l.title.clone())
|
||||
.unwrap_or_else(|| lecture.clone()),
|
||||
n_items: rows.len(),
|
||||
n_objectives: ids.len(),
|
||||
n_objectives_below: mine.iter().filter(|o| o.rate < threshold).count(),
|
||||
rate,
|
||||
questions: rows.iter().map(|r| r.number).collect(),
|
||||
// `objectives` arrives sorted worst first, so the first match is
|
||||
// the weakest one under this lecture.
|
||||
worst_objective: mine.first().map(|o| o.text.clone()),
|
||||
lecture,
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
|
||||
out.sort_by(|a, b| {
|
||||
a.rate
|
||||
.partial_cmp(&b.rate)
|
||||
.unwrap_or(std::cmp::Ordering::Equal)
|
||||
.then_with(|| a.lecture.cmp(&b.lecture))
|
||||
});
|
||||
out
|
||||
}
|
||||
|
||||
/// Summarizes how the authored expectations did.
|
||||
fn prediction_summary(analysis: &Analysis) -> PredictionSummary {
|
||||
let mut out = PredictionSummary::default();
|
||||
let mut signed: Vec<f64> = Vec::new();
|
||||
let mut biggest: Option<(u32, f64, f64)> = None;
|
||||
|
||||
for item in &analysis.items {
|
||||
let Some(prediction) = &item.prediction else {
|
||||
continue;
|
||||
};
|
||||
if prediction.calibrated {
|
||||
out.n_calibrated += 1;
|
||||
}
|
||||
if let (Some(expected), Some(error)) = (prediction.expected_p, prediction.p_error) {
|
||||
out.n_predicted += 1;
|
||||
signed.push(error);
|
||||
if prediction.p_within == Some(true) {
|
||||
out.n_within += 1;
|
||||
}
|
||||
if biggest
|
||||
.map(|(_, e, o)| (o - e).abs() < error.abs())
|
||||
.unwrap_or(true)
|
||||
{
|
||||
biggest = Some((item.number, expected, item.p_value));
|
||||
}
|
||||
}
|
||||
if prediction.expected_band.is_some() {
|
||||
out.n_band += 1;
|
||||
if prediction.band_hit == Some(true) {
|
||||
out.n_band_hit += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if !signed.is_empty() {
|
||||
let n = signed.len() as f64;
|
||||
out.mean_signed_error = Some(signed.iter().sum::<f64>() / n);
|
||||
out.mean_abs_error = Some(signed.iter().map(|e| e.abs()).sum::<f64>() / n);
|
||||
}
|
||||
out.biggest_surprise = biggest;
|
||||
out
|
||||
}
|
||||
|
||||
/// Sorts every question into what to do about it.
|
||||
///
|
||||
/// The order of the tests is the order of severity, and the first match wins for
|
||||
/// the three item decisions. `reteach` is judged separately, because "the item
|
||||
/// worked and the class missed it" is not a competing diagnosis; it is a
|
||||
/// different kind of finding.
|
||||
///
|
||||
/// # Arguments
|
||||
///
|
||||
/// * `questions` - the per-question rows.
|
||||
/// * `threshold` - the mastery threshold, which sets what counts as a content
|
||||
/// gap worth reteaching.
|
||||
/// * `default_options` - the course's default option count, used for the chance
|
||||
/// rate when an item's own options cannot be counted.
|
||||
///
|
||||
/// # Returns
|
||||
///
|
||||
/// The buckets.
|
||||
fn triage(questions: &[CohortQuestionRow], threshold: f64, default_options: usize) -> Triage {
|
||||
let mut out = Triage::default();
|
||||
|
||||
for question in questions {
|
||||
let row = |reasons: Vec<String>, option: Option<&OptionRow>| TriageRow {
|
||||
number: question.number,
|
||||
item: question.item.clone(),
|
||||
level: question.level,
|
||||
p_value: question.p_value,
|
||||
point_biserial: question.point_biserial,
|
||||
discrimination: question.discrimination,
|
||||
objectives: question.objective_texts.clone(),
|
||||
taught_in: question.taught_in.clone(),
|
||||
option: option.map(|o| o.letter.clone()),
|
||||
option_share: option.map(|o| o.rate),
|
||||
option_point_biserial: option.and_then(|o| o.point_biserial),
|
||||
reasons,
|
||||
};
|
||||
|
||||
let r = question.point_biserial;
|
||||
let key_r = question
|
||||
.options
|
||||
.iter()
|
||||
.filter(|o| o.is_key)
|
||||
.filter_map(|o| o.point_biserial)
|
||||
.fold(f64::NEG_INFINITY, f64::max);
|
||||
|
||||
// Count single letters only, so a multiple-response combination row such
|
||||
// as `A+D` is not mistaken for a fifth option and does not deflate the
|
||||
// chance rate.
|
||||
let counted = question
|
||||
.options
|
||||
.iter()
|
||||
.filter(|o| o.letter.chars().count() == 1)
|
||||
.count();
|
||||
let n_options = if counted >= 2 {
|
||||
counted
|
||||
} else {
|
||||
default_options.max(2)
|
||||
};
|
||||
let chance = 1.0 / n_options as f64;
|
||||
|
||||
// The best-supported alternative: chosen by a fifth of the class or more,
|
||||
// and correlating with total score at least as well as the key. The share
|
||||
// matters because a defensible reading that two students found is a
|
||||
// wording note, not a regrade.
|
||||
let challenger = question
|
||||
.options
|
||||
.iter()
|
||||
.filter(|o| !o.is_key && o.rate >= 0.20)
|
||||
.filter(|o| o.point_biserial.unwrap_or(f64::NEG_INFINITY) > 0.0)
|
||||
.filter(|o| {
|
||||
!key_r.is_finite() || o.point_biserial.unwrap_or(f64::NEG_INFINITY) >= key_r
|
||||
})
|
||||
.max_by(|a, b| {
|
||||
a.point_biserial
|
||||
.unwrap_or(f64::NEG_INFINITY)
|
||||
.partial_cmp(&b.point_biserial.unwrap_or(f64::NEG_INFINITY))
|
||||
.unwrap_or(std::cmp::Ordering::Equal)
|
||||
});
|
||||
|
||||
let mut placed = false;
|
||||
|
||||
// 1. Discard. Negative discrimination means the students who knew the
|
||||
// material did worse on it, which no amount of rewording fixes after
|
||||
// the fact; scores already awarded on it are noise.
|
||||
if let Some(r) = r {
|
||||
if r < -0.05 {
|
||||
out.discard.push(row(
|
||||
vec![format!(
|
||||
"students who scored well overall did worse on this one (r = {r:+.2}). \
|
||||
Whatever it measured, it was not what the rest of the exam measured."
|
||||
)],
|
||||
None,
|
||||
));
|
||||
placed = true;
|
||||
} else if r < 0.05 && question.p_value <= chance + 0.05 {
|
||||
out.discard.push(row(
|
||||
vec![format!(
|
||||
"{:.0}% correct against {:.0}% for guessing, and no relationship to total \
|
||||
score (r = {r:+.2}). The responses are indistinguishable from random.",
|
||||
question.p_value * 100.0,
|
||||
chance * 100.0
|
||||
)],
|
||||
None,
|
||||
));
|
||||
placed = true;
|
||||
}
|
||||
}
|
||||
|
||||
// 2. Rekey or award partial credit.
|
||||
if !placed {
|
||||
if let Some(option) = challenger {
|
||||
let mut reasons = vec![format!(
|
||||
"option {} drew {:.0}% and tracks total score at least as well as the key \
|
||||
({:+.2} against {:+.2}).",
|
||||
option.letter,
|
||||
option.rate * 100.0,
|
||||
option.point_biserial.unwrap_or(0.0),
|
||||
if key_r.is_finite() { key_r } else { 0.0 }
|
||||
)];
|
||||
if question.flags.iter().any(|f| f == "key_underperforms") {
|
||||
reasons.push(
|
||||
"the strongest students chose it more often than the key, which is the \
|
||||
signature of two readings rather than of a guess."
|
||||
.to_string(),
|
||||
);
|
||||
}
|
||||
reasons.push(
|
||||
"Either credit it for this administration or rewrite the stem to exclude it \
|
||||
before reuse."
|
||||
.to_string(),
|
||||
);
|
||||
out.rekey.push(row(reasons, Some(option)));
|
||||
placed = true;
|
||||
}
|
||||
}
|
||||
|
||||
// 3. Revise, unless the weak discrimination is explained by difficulty.
|
||||
if !placed {
|
||||
let mut reasons: Vec<String> = Vec::new();
|
||||
let weak = r.map(|r| r < 0.20).unwrap_or(true);
|
||||
let bounded = weak && (question.p_value >= 0.85 || question.p_value <= 0.20);
|
||||
|
||||
if weak && !bounded {
|
||||
reasons.push(format!(
|
||||
"at {:.0}% correct the item had room to separate students and did not \
|
||||
(r = {}).",
|
||||
question.p_value * 100.0,
|
||||
r.map(|r| format!("{r:+.2}"))
|
||||
.unwrap_or_else(|| "n/a".into())
|
||||
));
|
||||
}
|
||||
let dead: Vec<&OptionRow> = question
|
||||
.options
|
||||
.iter()
|
||||
.filter(|o| o.nonfunctioning)
|
||||
.collect();
|
||||
if !dead.is_empty() {
|
||||
reasons.push(format!(
|
||||
"option{} {} drew almost nobody, so the item is really a {}-way choice.",
|
||||
if dead.len() == 1 { "" } else { "s" },
|
||||
dead.iter()
|
||||
.map(|o| o.letter.as_str())
|
||||
.collect::<Vec<_>>()
|
||||
.join(", "),
|
||||
n_options.saturating_sub(dead.len()).max(2)
|
||||
));
|
||||
}
|
||||
if question.flags.iter().any(|f| f == "ambiguous") {
|
||||
reasons.push(
|
||||
"partial credit was awarded at grading time, which is a record that the item \
|
||||
admitted more than one reading."
|
||||
.to_string(),
|
||||
);
|
||||
}
|
||||
|
||||
if bounded {
|
||||
out.bounded.push(row(
|
||||
vec![format!(
|
||||
"{:.0}% correct leaves little variance to correlate with, so r = {} is \
|
||||
what this difficulty allows rather than a fault.",
|
||||
question.p_value * 100.0,
|
||||
r.map(|r| format!("{r:+.2}"))
|
||||
.unwrap_or_else(|| "n/a".into())
|
||||
)],
|
||||
None,
|
||||
));
|
||||
placed = true;
|
||||
} else if !reasons.is_empty() {
|
||||
out.revise.push(row(reasons, None));
|
||||
placed = true;
|
||||
}
|
||||
}
|
||||
|
||||
// 4. Reteach: the item did its job and the class still missed it. Judged
|
||||
// independently of the three above.
|
||||
let works = r.map(|r| r >= 0.20).unwrap_or(false);
|
||||
if works && question.p_value < threshold {
|
||||
out.reteach.push(row(
|
||||
vec![format!(
|
||||
"the item separated students cleanly (r = {}) and {:.0}% still missed it, so \
|
||||
this is a gap in what the class knows rather than a fault in the question.",
|
||||
r.map(|r| format!("{r:+.2}"))
|
||||
.unwrap_or_else(|| "n/a".into()),
|
||||
(1.0 - question.p_value) * 100.0
|
||||
)],
|
||||
None,
|
||||
));
|
||||
}
|
||||
|
||||
if !placed {
|
||||
out.clean += 1;
|
||||
}
|
||||
}
|
||||
|
||||
// Worst first inside each bucket, so the top of every list is where to start.
|
||||
for bucket in [
|
||||
&mut out.discard,
|
||||
&mut out.rekey,
|
||||
&mut out.revise,
|
||||
&mut out.bounded,
|
||||
] {
|
||||
bucket.sort_by(|a, b| {
|
||||
a.point_biserial
|
||||
.unwrap_or(1.0)
|
||||
.partial_cmp(&b.point_biserial.unwrap_or(1.0))
|
||||
.unwrap_or(std::cmp::Ordering::Equal)
|
||||
.then_with(|| a.number.cmp(&b.number))
|
||||
});
|
||||
}
|
||||
out.reteach.sort_by(|a, b| {
|
||||
a.p_value
|
||||
.partial_cmp(&b.p_value)
|
||||
.unwrap_or(std::cmp::Ordering::Equal)
|
||||
.then_with(|| a.number.cmp(&b.number))
|
||||
});
|
||||
out
|
||||
}
|
||||
|
||||
/// Per-question proportion correct, split by form.
|
||||
fn per_form_p_values(set: &ResponseSet) -> BTreeMap<u32, BTreeMap<String, f64>> {
|
||||
let forms: BTreeSet<&str> = set.rows.iter().filter_map(|r| r.form.as_deref()).collect();
|
||||
|
||||
Reference in New Issue
Block a user