diff --git a/src/analysis/classical.rs b/src/analysis/classical.rs index 846ff25..20c3b7d 100644 --- a/src/analysis/classical.rs +++ b/src/analysis/classical.rs @@ -61,6 +61,13 @@ pub struct Thresholds { pub nonfunctioning: f64, /// How far observed difficulty may drift from the authored expectation. pub design_tolerance: f64, + /// How many examinees an item's calibration needs before its recorded + /// expectations are treated as evidence rather than as the author's guess. + /// + /// Fifty is the point at which the standard error of a proportion near 0.5 + /// drops to about 0.07, which is small enough that a quarter-point miss is + /// about the item rather than about the sample. + pub calibrated_n: usize, /// Fraction of the class in the upper and lower comparison groups. Kelley's /// 0.27 maximizes the difference between the groups for a normal /// distribution, and it remains the convention. @@ -78,6 +85,7 @@ impl Default for Thresholds { negative_discrimination: -0.05, nonfunctioning: 0.05, design_tolerance: 0.25, + calibrated_n: 50, group_fraction: 0.27, small_sample: 100, } @@ -154,6 +162,38 @@ pub struct ItemAnalysis { pub flags: Vec, /// Human-readable explanations tied to the flags. pub notes: Vec, + /// How the item behaved against what its author predicted, when the item + /// records a prediction. + pub prediction: Option, +} + +/// An authored expectation, checked against what happened. +/// +/// Kept apart from [`ItemAnalysis::flags`] on purpose. Before an item has been +/// administered, `design.expected_difficulty` is the author's guess, and a guess +/// that turns out wrong says something about the guess rather than about the +/// item. Flagging it anyway is how a report ends up with thirty +/// `design_mismatch` findings and no way to see the four that matter. So the +/// discrepancy is always recorded here, and it only becomes a +/// [`Flag::DesignMismatch`] once the expectation has data behind it. +#[derive(Debug, Clone)] +pub struct Prediction { + /// The difficulty the author expected. + pub expected_p: Option, + /// The discrimination band the author expected, as `(low, high)`. + pub expected_band: Option<(f64, f64)>, + /// Whether the expectation rests on a calibration with enough examinees + /// behind it, rather than on the author's judgement alone. + pub calibrated: bool, + /// Signed difficulty error, observed minus expected. Positive means the item + /// was easier than predicted. + pub p_error: Option, + /// Whether observed difficulty landed inside the tolerance. + pub p_within: Option, + /// Whether observed discrimination landed inside the expected band. + pub band_hit: Option, + /// What to say about it, phrased for whichever case applies. + pub notes: Vec, } impl ItemAnalysis { @@ -493,9 +533,29 @@ pub fn analyze( options, flags: Vec::new(), notes: Vec::new(), + prediction: None, }; - flag_item(&mut analysis, t, design.as_ref(), &rows); + // Whether the authored expectation is evidence or a guess. An item that + // has never been administered has no calibration block, and one edited + // since its last calibration has a fingerprint that no longer matches. + let calibrated = record + .and_then(|r| r.placement(*number)) + .and_then(|p| catalog.and_then(|c| c.get(&p.item))) + .and_then(|entry| { + let cal = entry.item.calibration.as_ref()?; + let enough = cal.n_examinees.unwrap_or(0) >= t.calibrated_n; + let current = match &cal.fingerprint { + Some(recorded) => *recorded == entry.item.fingerprint(), + // An older calibration block with no fingerprint cannot be + // shown stale, so it is taken at its word. + None => true, + }; + Some(enough && current) + }) + .unwrap_or(false); + + flag_item(&mut analysis, t, design.as_ref(), &rows, calibrated); p_values.push(p_value); if let Some(r) = rpb { @@ -521,11 +581,13 @@ pub fn analyze( /// * `t` - the thresholds. /// * `design` - the authored expectation, when available. /// * `rows` - the raw responses, for partial-credit detection. +/// * `calibrated` - whether that expectation rests on prior data. fn flag_item( a: &mut ItemAnalysis, t: &Thresholds, design: Option<&Design>, rows: &[&crate::responses::Response], + calibrated: bool, ) { // Discrimination first: it is the finding that changes what you do. match a.point_biserial { @@ -678,31 +740,71 @@ fn flag_item( } } - // Did the item behave as authored? + // Did the item behave as authored? This is the one check whose meaning + // depends on where the expectation came from, so it is recorded either way + // and flagged only when the expectation had data behind it. if let Some(d) = design { + let mut prediction = Prediction { + expected_p: d.expected_difficulty, + expected_band: d.expected_discrimination.map(|b| b.expected_band()), + calibrated, + p_error: None, + p_within: None, + band_hit: None, + notes: Vec::new(), + }; + if let Some(expected) = d.expected_difficulty { - if (expected - a.p_value).abs() > t.design_tolerance { - a.flags.push(Flag::DesignMismatch); - a.notes.push(format!( - "you expected about {:.0}% correct and observed {:.0}%. Worth knowing whether \ - your model of the students or the item is off.", - expected * 100.0, - a.p_value * 100.0 - )); + let error = a.p_value - expected; + let within = error.abs() <= t.design_tolerance; + prediction.p_error = Some(error); + prediction.p_within = Some(within); + if !within { + if calibrated { + a.flags.push(Flag::DesignMismatch); + prediction.notes.push(format!( + "this item is calibrated at about {:.0}% correct and came out at {:.0}%. \ + Something changed: the cohort, the teaching, or the item.", + expected * 100.0, + a.p_value * 100.0 + )); + } else { + prediction.notes.push(format!( + "you predicted about {:.0}% correct and observed {:.0}%. This is the \ + first data on the item, so it corrects the prediction rather than \ + condemning the item.", + expected * 100.0, + a.p_value * 100.0 + )); + } } } + if let (Some(band), Some(r)) = (d.expected_discrimination, a.point_biserial) { let (low, high) = band.expected_band(); - if r < low || r > high { - if !a.flags.contains(&Flag::DesignMismatch) { - a.flags.push(Flag::DesignMismatch); + let hit = r >= low && r <= high; + prediction.band_hit = Some(hit); + if !hit { + if calibrated { + if !a.flags.contains(&Flag::DesignMismatch) { + a.flags.push(Flag::DesignMismatch); + } + prediction.notes.push(format!( + "calibrated for {} discrimination ({low:.2} to {high:.2}), observed \ + {r:.2}.", + format!("{band:?}").to_lowercase() + )); + } else { + prediction.notes.push(format!( + "you predicted {} discrimination ({low:.2} to {high:.2}) and observed \ + {r:.2}.", + format!("{band:?}").to_lowercase() + )); } - a.notes.push(format!( - "you expected {} discrimination ({low:.2} to {high:.2}) and observed {r:.2}.", - format!("{band:?}").to_lowercase() - )); } } + + a.prediction = Some(prediction); } a.flags.sort(); diff --git a/src/analysis/diagnostic.rs b/src/analysis/diagnostic.rs index 58afcd3..0c3be4e 100644 --- a/src/analysis/diagnostic.rs +++ b/src/analysis/diagnostic.rs @@ -917,8 +917,19 @@ pub struct CohortDiagnostic { pub objectives: Vec, /// Objectives the class as a whole did not meet. pub gaps: Vec, + /// The distribution binned by the course's letter-grade scale. Empty when + /// `course.yaml` sets no scale, in which case the ten-point bins stand. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub grades: Vec, + /// Per-lecture class performance, worst first. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub lectures: Vec, /// Per-question statistics. pub questions: Vec, + /// What to do about each question that raised something. + pub triage: Triage, + /// How the authored expectations did. + pub predictions: PredictionSummary, /// Questions worth revisiting before reuse, worst first. pub revise: Vec, /// One row per form, when more than one was given. @@ -1036,6 +1047,20 @@ pub struct CohortQuestionRow { pub level: Option, /// The objectives it measured. pub objectives: Vec, + /// What those objectives ask, in the course's own words. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub objective_texts: Vec, + /// Where the item was taught, as lecture titles and slide numbers. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub taught_in: Vec, + /// The lecture ids alone, for a table column where only `L1.4` fits. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub lectures: Vec, + /// Which difficulty band it fell in: `too easy`, `moderate`, or `hard`. + pub difficulty_band: String, + /// Which discrimination band it fell in, on the conventional cut points: + /// `excellent`, `good`, `marginal`, `poor`, or `negative`. + pub discrimination_band: String, /// Proportion correct. pub p_value: f64, /// Corrected item-total point-biserial. @@ -1054,6 +1079,13 @@ pub struct CohortQuestionRow { pub flags: Vec, /// What those flags mean. pub notes: Vec, + /// How the item did against its author's expectation. Separate from `notes` + /// because an unmet prediction on an uncalibrated item is a fact about the + /// prediction. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub prediction_notes: Vec, + /// Whether that expectation rested on a prior calibration. + pub calibrated: bool, /// Per-form proportion correct, when more than one form was given. A gap here /// on one question, with the rest of the exam in step, points at that /// question's permutation rather than at the cohort. @@ -1106,6 +1138,159 @@ pub struct PatternRow { pub level_means: BTreeMap, } +/// The class's scores binned by the course's own letter-grade scale. +/// +/// A ten-point histogram is the default because it needs no course +/// configuration, but nobody acts on "nineteen students in the fifties". They +/// act on "nineteen students are failing", and that sentence needs the scale +/// from `course.yaml`. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct GradeRow { + /// The letter. + pub letter: String, + /// The lowest percentage in the band. + pub low: f64, + /// The highest percentage in the band, which is just under the next band's + /// floor, or 100 for the top band. + pub high: f64, + /// Grade points, when the scale records them. + #[serde(skip_serializing_if = "Option::is_none")] + pub gpa: Option, + /// The attainment word, when the scale records one. + #[serde(skip_serializing_if = "Option::is_none")] + pub attainment: Option, + /// The colour group, so A, A- and A+ can be tinted together. + pub group: String, + /// How many students landed in the band. + pub count: usize, + /// Their share of the class, in `0.0..=1.0`. + pub share: f64, + /// How many students are in this band or a higher one. + pub at_or_above: usize, +} + +/// One lecture's showing, aggregated from the items written against it. +/// +/// The objective table answers "which objective went wrong". This answers "which +/// class meeting went wrong", which is the question that maps onto next week. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct CohortLectureRow { + /// The lecture id. + pub lecture: String, + /// Its title. + pub title: String, + /// How many scored items traced back to it. + pub n_items: usize, + /// How many distinct objectives those items measured. + pub n_objectives: usize, + /// How many of those objectives the class did not meet. + pub n_objectives_below: usize, + /// Mean proportion correct across its items. + pub rate: f64, + /// The questions, so the row can be checked against the item table. + pub questions: Vec, + /// The worst objective under this lecture, by class rate. + #[serde(skip_serializing_if = "Option::is_none")] + pub worst_objective: Option, +} + +/// What to do about one question, and why. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct TriageRow { + /// The question number. + pub number: u32, + /// The item id. + #[serde(skip_serializing_if = "Option::is_none")] + pub item: Option, + /// The level code. + #[serde(skip_serializing_if = "Option::is_none")] + pub level: Option, + /// Proportion correct. + pub p_value: f64, + /// Corrected item-total correlation. + #[serde(skip_serializing_if = "Option::is_none")] + pub point_biserial: Option, + /// Upper minus lower group. + #[serde(skip_serializing_if = "Option::is_none")] + pub discrimination: Option, + /// What the question measured, in the course's words. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub objectives: Vec, + /// Where it was taught. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub taught_in: Vec, + /// The specific option this recommendation is about, when it is about one. + #[serde(skip_serializing_if = "Option::is_none")] + pub option: Option, + /// That option's share of responses. + #[serde(skip_serializing_if = "Option::is_none")] + pub option_share: Option, + /// That option's correlation with total score. + #[serde(skip_serializing_if = "Option::is_none")] + pub option_point_biserial: Option, + /// The evidence, one clause per line. + pub reasons: Vec, +} + +/// Every question sorted into what to do with it. +/// +/// The first four lists are decisions about items and are mutually exclusive: a +/// question appears in the most severe one that fits, because there is no point +/// rewriting a distractor on an item you are about to discard. `reteach` is not +/// a decision about an item at all, so a question can appear there as well as in +/// one of the others. +#[derive(Debug, Clone, Default, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct Triage { + /// Broken: the evidence says these did not measure what they were scored on. + pub discard: Vec, + /// A second defensible answer with statistical support behind it. + pub rekey: Vec, + /// Weak but salvageable, worth rewriting before reuse. + pub revise: Vec, + /// Sound items the class got wrong. A teaching finding, not an item finding. + pub reteach: Vec, + /// Items whose low discrimination is explained by their difficulty rather + /// than by a fault. Listed so they are not mistaken for work to do. + pub bounded: Vec, + /// How many questions raised nothing at all. + pub clean: usize, +} + +/// How the authored expectations did against the data. +/// +/// This exists so that an uncalibrated bank does not produce one +/// `design_mismatch` per item. Before an item has data, its expected difficulty +/// is a prediction by its author, and the useful summary is whether those +/// predictions run optimistic or pessimistic as a set. +#[derive(Debug, Clone, Default, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct PredictionSummary { + /// How many items recorded an expected difficulty. + pub n_predicted: usize, + /// How many of those expectations rest on a prior calibration. + pub n_calibrated: usize, + /// Mean of observed minus expected difficulty. Positive means the items came + /// out easier than predicted. + #[serde(skip_serializing_if = "Option::is_none")] + pub mean_signed_error: Option, + /// Mean absolute difficulty error, which is the size of a typical miss. + #[serde(skip_serializing_if = "Option::is_none")] + pub mean_abs_error: Option, + /// How many landed inside the tolerance. + pub n_within: usize, + /// How many items recorded an expected discrimination band. + pub n_band: usize, + /// How many of those landed inside it. + pub n_band_hit: usize, + /// The largest single surprise, as `(question, expected, observed)`. + #[serde(skip_serializing_if = "Option::is_none")] + pub biggest_surprise: Option<(u32, f64, f64)>, +} + /// Builds the class diagnostic. /// /// # Arguments @@ -1234,6 +1419,29 @@ pub fn cohort( item: item.item_ref.clone(), level: meta.and_then(|m| m.0), objectives: meta.map(|m| m.1.clone()).unwrap_or_default(), + objective_texts: meta + .map(|m| m.1.iter().map(|id| course.objective_text(id)).collect()) + .unwrap_or_default(), + taught_in: item + .item_ref + .as_deref() + .map(|uid| taught_in(catalog, uid)) + .unwrap_or_default(), + lectures: item + .item_ref + .as_deref() + .and_then(|uid| catalog.get(uid)) + .map(|entry| { + entry + .item + .sources + .iter() + .map(|source| source.lecture.clone()) + .collect() + }) + .unwrap_or_default(), + difficulty_band: difficulty_band(item.p_value).to_string(), + discrimination_band: discrimination_band(item.point_biserial).to_string(), p_value: item.p_value, point_biserial: item.point_biserial, discrimination: item.discrimination_index, @@ -1253,6 +1461,12 @@ pub fn cohort( .collect(), flags: item.flags.iter().map(|f| f.as_str().to_string()).collect(), notes: item.notes.clone(), + prediction_notes: item + .prediction + .as_ref() + .map(|p| p.notes.clone()) + .unwrap_or_default(), + calibrated: item.prediction.as_ref().is_some_and(|p| p.calibrated), by_form: by_form.get(&item.number).cloned().unwrap_or_default(), } }) @@ -1264,10 +1478,20 @@ pub fn cohort( .filter_map(|item| questions.iter().find(|q| q.number == item.number).cloned()) .collect(); + let default_options = course.policy.options_per_item; + let triage = triage(&questions, threshold, default_options); + let predictions = prediction_summary(analysis); + let grades = grade_rows(&course.policy, &percents); + let lectures = lecture_rows(catalog, &questions, &objectives, threshold); + CohortDiagnostic { n_students: cohort.students.len(), n_items: analysis.reliability.n_items, distribution: distribution(&percents), + grades, + lectures, + triage, + predictions, reliability: ReliabilityRow { alpha: analysis.reliability.alpha, sem: analysis.reliability.sem, @@ -1299,6 +1523,496 @@ pub fn cohort( } } +/// Where an item was taught, as lecture titles with slide numbers. +/// +/// # Arguments +/// +/// * `catalog` - the loaded course. +/// * `uid` - the item's global id. +/// +/// # Returns +/// +/// One entry per source the item records. +fn taught_in(catalog: &Catalog, uid: &str) -> Vec { + let Some(entry) = catalog.get(uid) else { + return Vec::new(); + }; + entry + .item + .sources + .iter() + .map(|source| { + let title = catalog + .course + .lectures + .get(&source.lecture) + .map(|l| l.title.clone()) + .unwrap_or_else(|| source.lecture.clone()); + if source.slides.is_empty() { + format!("{} ({})", title, source.lecture) + } else { + let slides: Vec = source.slides.iter().map(|s| s.to_string()).collect(); + format!( + "{} ({}), slide{} {}", + title, + source.lecture, + if source.slides.len() == 1 { "" } else { "s" }, + slides.join(", ") + ) + } + }) + .collect() +} + +/// The difficulty band a p-value falls in. +/// +/// Three bands rather than five. The only distinction that changes what you do +/// is whether the item had room to discriminate at all, and that is a question +/// about the middle versus the two ends. +fn difficulty_band(p: f64) -> &'static str { + if p >= 0.85 { + "too easy" + } else if p <= 0.35 { + "hard" + } else { + "moderate" + } +} + +/// The discrimination band a point-biserial falls in. +/// +/// The cut points are the conventional ones from the item-analysis literature, +/// usually attributed to Ebel: about 0.40 and above is excellent, 0.30 to 0.39 +/// good, 0.20 to 0.29 marginal, and below 0.20 poor. They are rules of thumb +/// rather than laws, and they must be read next to difficulty, because an item +/// almost everyone passes or fails has little variance left to correlate with +/// anything. +fn discrimination_band(r: Option) -> &'static str { + match r { + None => "no variance", + Some(r) if r < 0.0 => "negative", + Some(r) if r < 0.20 => "poor", + Some(r) if r < 0.30 => "marginal", + Some(r) if r < 0.40 => "good", + Some(_) => "excellent", + } +} + +/// Bins the class by the course's letter-grade scale. +/// +/// # Arguments +/// +/// * `policy` - the course policy, for its scale. +/// * `percents` - one score per student, out of 100. +/// +/// # Returns +/// +/// One row per band, highest first. Empty when the course sets no scale, which +/// is the signal for a report to fall back to ten-point bins. +fn grade_rows(policy: &crate::course::Policy, percents: &[f64]) -> Vec { + let bands = policy.bands(); + if bands.is_empty() || percents.is_empty() { + return Vec::new(); + } + + let n = percents.len() as f64; + let mut out: Vec = Vec::with_capacity(bands.len()); + let mut running = 0usize; + for (index, band) in bands.iter().enumerate() { + // The ceiling is the floor of the band above, less the smallest step a + // percentage is reported at, so the printed range reads the way a + // syllabus writes it. + let high = match index { + 0 => 100.0, + _ => bands[index - 1].min - 0.1, + }; + let count = percents + .iter() + .filter(|percent| { + **percent + 1e-9 >= band.min && (index == 0 || **percent < bands[index - 1].min) + }) + .count(); + running += count; + out.push(GradeRow { + letter: band.letter.clone(), + low: band.min, + high, + gpa: band.gpa, + attainment: band.attainment.clone(), + group: band.group_key(), + count, + share: count as f64 / n, + at_or_above: running, + }); + } + out +} + +/// Aggregates questions into per-lecture rows, worst first. +/// +/// # Arguments +/// +/// * `catalog` - the loaded course, for lecture titles and objective lectures. +/// * `questions` - the per-question rows. +/// * `objectives` - the per-objective rows, for the objective counts. +/// * `threshold` - the mastery threshold. +/// +/// # Returns +/// +/// One row per lecture that any scored item traced back to. +fn lecture_rows( + catalog: &Catalog, + questions: &[CohortQuestionRow], + objectives: &[CohortObjectiveRow], + threshold: f64, +) -> Vec { + let course = &catalog.course; + let mut items: BTreeMap> = BTreeMap::new(); + let mut lecture_objectives: BTreeMap> = BTreeMap::new(); + + for question in questions { + // The same two routes the student report uses: the item knows which + // lecture it was written from, and the objective registry knows which + // lectures develop it. + let mut lectures: BTreeSet = BTreeSet::new(); + if let Some(entry) = question.item.as_deref().and_then(|uid| catalog.get(uid)) { + for source in &entry.item.sources { + lectures.insert(source.lecture.clone()); + } + } + for objective in &question.objectives { + if let Some(entry) = course.learning_objectives.get(objective) { + lectures.extend(entry.lectures.iter().cloned()); + } + } + for lecture in lectures { + items.entry(lecture.clone()).or_default().push(question); + lecture_objectives + .entry(lecture) + .or_default() + .extend(question.objectives.iter().cloned()); + } + } + + let mut out: Vec = items + .into_iter() + .map(|(lecture, rows)| { + let rate = rows.iter().map(|r| r.p_value).sum::() / rows.len() as f64; + let ids = lecture_objectives + .get(&lecture) + .cloned() + .unwrap_or_default(); + let mine: Vec<&CohortObjectiveRow> = + objectives.iter().filter(|o| ids.contains(&o.id)).collect(); + CohortLectureRow { + title: course + .lectures + .get(&lecture) + .map(|l| l.title.clone()) + .unwrap_or_else(|| lecture.clone()), + n_items: rows.len(), + n_objectives: ids.len(), + n_objectives_below: mine.iter().filter(|o| o.rate < threshold).count(), + rate, + questions: rows.iter().map(|r| r.number).collect(), + // `objectives` arrives sorted worst first, so the first match is + // the weakest one under this lecture. + worst_objective: mine.first().map(|o| o.text.clone()), + lecture, + } + }) + .collect(); + + out.sort_by(|a, b| { + a.rate + .partial_cmp(&b.rate) + .unwrap_or(std::cmp::Ordering::Equal) + .then_with(|| a.lecture.cmp(&b.lecture)) + }); + out +} + +/// Summarizes how the authored expectations did. +fn prediction_summary(analysis: &Analysis) -> PredictionSummary { + let mut out = PredictionSummary::default(); + let mut signed: Vec = Vec::new(); + let mut biggest: Option<(u32, f64, f64)> = None; + + for item in &analysis.items { + let Some(prediction) = &item.prediction else { + continue; + }; + if prediction.calibrated { + out.n_calibrated += 1; + } + if let (Some(expected), Some(error)) = (prediction.expected_p, prediction.p_error) { + out.n_predicted += 1; + signed.push(error); + if prediction.p_within == Some(true) { + out.n_within += 1; + } + if biggest + .map(|(_, e, o)| (o - e).abs() < error.abs()) + .unwrap_or(true) + { + biggest = Some((item.number, expected, item.p_value)); + } + } + if prediction.expected_band.is_some() { + out.n_band += 1; + if prediction.band_hit == Some(true) { + out.n_band_hit += 1; + } + } + } + + if !signed.is_empty() { + let n = signed.len() as f64; + out.mean_signed_error = Some(signed.iter().sum::() / n); + out.mean_abs_error = Some(signed.iter().map(|e| e.abs()).sum::() / n); + } + out.biggest_surprise = biggest; + out +} + +/// Sorts every question into what to do about it. +/// +/// The order of the tests is the order of severity, and the first match wins for +/// the three item decisions. `reteach` is judged separately, because "the item +/// worked and the class missed it" is not a competing diagnosis; it is a +/// different kind of finding. +/// +/// # Arguments +/// +/// * `questions` - the per-question rows. +/// * `threshold` - the mastery threshold, which sets what counts as a content +/// gap worth reteaching. +/// * `default_options` - the course's default option count, used for the chance +/// rate when an item's own options cannot be counted. +/// +/// # Returns +/// +/// The buckets. +fn triage(questions: &[CohortQuestionRow], threshold: f64, default_options: usize) -> Triage { + let mut out = Triage::default(); + + for question in questions { + let row = |reasons: Vec, option: Option<&OptionRow>| TriageRow { + number: question.number, + item: question.item.clone(), + level: question.level, + p_value: question.p_value, + point_biserial: question.point_biserial, + discrimination: question.discrimination, + objectives: question.objective_texts.clone(), + taught_in: question.taught_in.clone(), + option: option.map(|o| o.letter.clone()), + option_share: option.map(|o| o.rate), + option_point_biserial: option.and_then(|o| o.point_biserial), + reasons, + }; + + let r = question.point_biserial; + let key_r = question + .options + .iter() + .filter(|o| o.is_key) + .filter_map(|o| o.point_biserial) + .fold(f64::NEG_INFINITY, f64::max); + + // Count single letters only, so a multiple-response combination row such + // as `A+D` is not mistaken for a fifth option and does not deflate the + // chance rate. + let counted = question + .options + .iter() + .filter(|o| o.letter.chars().count() == 1) + .count(); + let n_options = if counted >= 2 { + counted + } else { + default_options.max(2) + }; + let chance = 1.0 / n_options as f64; + + // The best-supported alternative: chosen by a fifth of the class or more, + // and correlating with total score at least as well as the key. The share + // matters because a defensible reading that two students found is a + // wording note, not a regrade. + let challenger = question + .options + .iter() + .filter(|o| !o.is_key && o.rate >= 0.20) + .filter(|o| o.point_biserial.unwrap_or(f64::NEG_INFINITY) > 0.0) + .filter(|o| { + !key_r.is_finite() || o.point_biserial.unwrap_or(f64::NEG_INFINITY) >= key_r + }) + .max_by(|a, b| { + a.point_biserial + .unwrap_or(f64::NEG_INFINITY) + .partial_cmp(&b.point_biserial.unwrap_or(f64::NEG_INFINITY)) + .unwrap_or(std::cmp::Ordering::Equal) + }); + + let mut placed = false; + + // 1. Discard. Negative discrimination means the students who knew the + // material did worse on it, which no amount of rewording fixes after + // the fact; scores already awarded on it are noise. + if let Some(r) = r { + if r < -0.05 { + out.discard.push(row( + vec![format!( + "students who scored well overall did worse on this one (r = {r:+.2}). \ + Whatever it measured, it was not what the rest of the exam measured." + )], + None, + )); + placed = true; + } else if r < 0.05 && question.p_value <= chance + 0.05 { + out.discard.push(row( + vec![format!( + "{:.0}% correct against {:.0}% for guessing, and no relationship to total \ + score (r = {r:+.2}). The responses are indistinguishable from random.", + question.p_value * 100.0, + chance * 100.0 + )], + None, + )); + placed = true; + } + } + + // 2. Rekey or award partial credit. + if !placed { + if let Some(option) = challenger { + let mut reasons = vec![format!( + "option {} drew {:.0}% and tracks total score at least as well as the key \ + ({:+.2} against {:+.2}).", + option.letter, + option.rate * 100.0, + option.point_biserial.unwrap_or(0.0), + if key_r.is_finite() { key_r } else { 0.0 } + )]; + if question.flags.iter().any(|f| f == "key_underperforms") { + reasons.push( + "the strongest students chose it more often than the key, which is the \ + signature of two readings rather than of a guess." + .to_string(), + ); + } + reasons.push( + "Either credit it for this administration or rewrite the stem to exclude it \ + before reuse." + .to_string(), + ); + out.rekey.push(row(reasons, Some(option))); + placed = true; + } + } + + // 3. Revise, unless the weak discrimination is explained by difficulty. + if !placed { + let mut reasons: Vec = Vec::new(); + let weak = r.map(|r| r < 0.20).unwrap_or(true); + let bounded = weak && (question.p_value >= 0.85 || question.p_value <= 0.20); + + if weak && !bounded { + reasons.push(format!( + "at {:.0}% correct the item had room to separate students and did not \ + (r = {}).", + question.p_value * 100.0, + r.map(|r| format!("{r:+.2}")) + .unwrap_or_else(|| "n/a".into()) + )); + } + let dead: Vec<&OptionRow> = question + .options + .iter() + .filter(|o| o.nonfunctioning) + .collect(); + if !dead.is_empty() { + reasons.push(format!( + "option{} {} drew almost nobody, so the item is really a {}-way choice.", + if dead.len() == 1 { "" } else { "s" }, + dead.iter() + .map(|o| o.letter.as_str()) + .collect::>() + .join(", "), + n_options.saturating_sub(dead.len()).max(2) + )); + } + if question.flags.iter().any(|f| f == "ambiguous") { + reasons.push( + "partial credit was awarded at grading time, which is a record that the item \ + admitted more than one reading." + .to_string(), + ); + } + + if bounded { + out.bounded.push(row( + vec![format!( + "{:.0}% correct leaves little variance to correlate with, so r = {} is \ + what this difficulty allows rather than a fault.", + question.p_value * 100.0, + r.map(|r| format!("{r:+.2}")) + .unwrap_or_else(|| "n/a".into()) + )], + None, + )); + placed = true; + } else if !reasons.is_empty() { + out.revise.push(row(reasons, None)); + placed = true; + } + } + + // 4. Reteach: the item did its job and the class still missed it. Judged + // independently of the three above. + let works = r.map(|r| r >= 0.20).unwrap_or(false); + if works && question.p_value < threshold { + out.reteach.push(row( + vec![format!( + "the item separated students cleanly (r = {}) and {:.0}% still missed it, so \ + this is a gap in what the class knows rather than a fault in the question.", + r.map(|r| format!("{r:+.2}")) + .unwrap_or_else(|| "n/a".into()), + (1.0 - question.p_value) * 100.0 + )], + None, + )); + } + + if !placed { + out.clean += 1; + } + } + + // Worst first inside each bucket, so the top of every list is where to start. + for bucket in [ + &mut out.discard, + &mut out.rekey, + &mut out.revise, + &mut out.bounded, + ] { + bucket.sort_by(|a, b| { + a.point_biserial + .unwrap_or(1.0) + .partial_cmp(&b.point_biserial.unwrap_or(1.0)) + .unwrap_or(std::cmp::Ordering::Equal) + .then_with(|| a.number.cmp(&b.number)) + }); + } + out.reteach.sort_by(|a, b| { + a.p_value + .partial_cmp(&b.p_value) + .unwrap_or(std::cmp::Ordering::Equal) + .then_with(|| a.number.cmp(&b.number)) + }); + out +} + /// Per-question proportion correct, split by form. fn per_form_p_values(set: &ResponseSet) -> BTreeMap> { let forms: BTreeSet<&str> = set.rows.iter().filter_map(|r| r.form.as_deref()).collect(); diff --git a/src/authoring/jsonschema.rs b/src/authoring/jsonschema.rs index 3d9ce29..3788509 100644 --- a/src/authoring/jsonschema.rs +++ b/src/authoring/jsonschema.rs @@ -261,6 +261,43 @@ fn policy_schema() -> Value { "minimum": 1, "description": "Below this many items on an objective, reports say 'not enough \ evidence' rather than classifying." + }, + "grade_scale": { + "type": "array", + "description": "Letter-grade bands. Only the lower bound of each is recorded; a \ + band runs up to the next one. Set this and a class report bins \ + scores by letter rather than by ten-point interval.", + "items": { + "type": "object", + "required": ["letter", "min"], + "additionalProperties": false, + "properties": { + "letter": { + "type": "string", + "description": "The letter as it appears on a transcript." + }, + "min": { + "type": "number", + "minimum": 0, + "maximum": 100, + "description": "Lowest percentage earning this letter, inclusive." + }, + "gpa": { + "type": "number", + "minimum": 0, + "description": "Grade points the band carries." + }, + "attainment": { + "type": "string", + "description": "The attainment word attached to the band." + }, + "group": { + "type": "string", + "description": "Colour group for reports; defaults to the letter's \ + first character." + } + } + } } } }) diff --git a/src/export/typst/diagnostic.rs b/src/export/typst/diagnostic.rs index 0eba690..08ac1db 100644 --- a/src/export/typst/diagnostic.rs +++ b/src/export/typst/diagnostic.rs @@ -549,6 +549,105 @@ pub fn cohort_value(diagnostic: &CohortDiagnostic, config: &RenderConfig) -> Val ), ); + out.insert( + "grades", + Value::Array( + diagnostic + .grades + .iter() + .map(|grade| { + let mut value = Value::dict(); + value.insert("letter", Value::str(&grade.letter)); + value.insert("low", Value::Float(grade.low)); + value.insert("high", Value::Float(grade.high)); + value.insert_some("gpa", grade.gpa.map(Value::Float)); + value.insert_some("attainment", grade.attainment.as_ref().map(Value::str)); + value.insert("group", Value::str(&grade.group)); + value.insert("count", Value::Int(grade.count as i64)); + value.insert("share", Value::Float(grade.share)); + value.insert("at-or-above", Value::Int(grade.at_or_above as i64)); + value + }) + .collect(), + ), + ); + + out.insert( + "lectures", + Value::Array( + diagnostic + .lectures + .iter() + .map(|lecture| { + let mut value = Value::dict(); + value.insert("lecture", Value::str(&lecture.lecture)); + value.insert("title", Value::str(&lecture.title)); + value.insert("items", Value::Int(lecture.n_items as i64)); + value.insert("objectives", Value::Int(lecture.n_objectives as i64)); + value.insert( + "objectives-below", + Value::Int(lecture.n_objectives_below as i64), + ); + value.insert("rate", Value::Float(lecture.rate)); + value.insert( + "questions", + Value::Array( + lecture + .questions + .iter() + .map(|n| Value::Int(*n as i64)) + .collect(), + ), + ); + value.insert_some( + "worst-objective", + lecture + .worst_objective + .as_ref() + .map(|text| markup_value(text, content)), + ); + value + }) + .collect(), + ), + ); + + let triage_rows = |rows: &[crate::diagnostic::TriageRow]| -> Value { + Value::Array(rows.iter().map(|row| triage_value(row, content)).collect()) + }; + let mut triage = Value::dict(); + triage.insert("discard", triage_rows(&diagnostic.triage.discard)); + triage.insert("rekey", triage_rows(&diagnostic.triage.rekey)); + triage.insert("revise", triage_rows(&diagnostic.triage.revise)); + triage.insert("reteach", triage_rows(&diagnostic.triage.reteach)); + triage.insert("bounded", triage_rows(&diagnostic.triage.bounded)); + triage.insert("clean", Value::Int(diagnostic.triage.clean as i64)); + out.insert("triage", triage); + + let predictions = &diagnostic.predictions; + let mut prediction = Value::dict(); + prediction.insert("predicted", Value::Int(predictions.n_predicted as i64)); + prediction.insert("calibrated", Value::Int(predictions.n_calibrated as i64)); + prediction.insert_some( + "mean-signed-error", + predictions.mean_signed_error.map(Value::Float), + ); + prediction.insert_some( + "mean-abs-error", + predictions.mean_abs_error.map(Value::Float), + ); + prediction.insert("within", Value::Int(predictions.n_within as i64)); + prediction.insert("band", Value::Int(predictions.n_band as i64)); + prediction.insert("band-hit", Value::Int(predictions.n_band_hit as i64)); + if let Some((number, expected, observed)) = predictions.biggest_surprise { + let mut surprise = Value::dict(); + surprise.insert("number", Value::Int(number as i64)); + surprise.insert("expected", Value::Float(expected)); + surprise.insert("observed", Value::Float(observed)); + prediction.insert("biggest-surprise", surprise); + } + out.insert("predictions", prediction); + out.insert( "forms", Value::Array( @@ -601,6 +700,46 @@ pub fn cohort_value(diagnostic: &CohortDiagnostic, config: &RenderConfig) -> Val out } +/// One triage row as a Typst value. +fn triage_value(row: &crate::diagnostic::TriageRow, content: bool) -> Value { + let mut value = Value::dict(); + value.insert("number", Value::Int(row.number as i64)); + value.insert_some("item", row.item.as_ref().map(Value::str)); + value.insert_some("level", row.level.map(|l| Value::Int(l as i64))); + value.insert("p", Value::Float(row.p_value)); + value.insert_some("point-biserial", row.point_biserial.map(Value::Float)); + value.insert_some("discrimination", row.discrimination.map(Value::Float)); + value.insert( + "objectives", + Value::Array( + row.objectives + .iter() + .map(|text| markup_value(text, content)) + .collect(), + ), + ); + value.insert( + "taught-in", + Value::Array(row.taught_in.iter().map(|t| Value::str(t)).collect()), + ); + value.insert_some("option", row.option.as_ref().map(Value::str)); + value.insert_some("option-share", row.option_share.map(Value::Float)); + value.insert_some( + "option-point-biserial", + row.option_point_biserial.map(Value::Float), + ); + value.insert( + "reasons", + Value::Array( + row.reasons + .iter() + .map(|reason| markup_value(reason, content)) + .collect(), + ), + ); + value +} + /// One histogram bin as a Typst value. fn bin_value(bin: &Bin) -> Value { let mut value = Value::dict(); @@ -635,6 +774,29 @@ fn cohort_question_value(question: &CohortQuestionRow, content: bool) -> Value { "objectives", Value::Array(question.objectives.iter().map(|o| Value::str(o)).collect()), ); + value.insert( + "objective-texts", + Value::Array( + question + .objective_texts + .iter() + .map(|text| markup_value(text, content)) + .collect(), + ), + ); + value.insert( + "taught-in", + Value::Array(question.taught_in.iter().map(|t| Value::str(t)).collect()), + ); + value.insert( + "lectures", + Value::Array(question.lectures.iter().map(|l| Value::str(l)).collect()), + ); + value.insert("difficulty-band", Value::str(&question.difficulty_band)); + value.insert( + "discrimination-band", + Value::str(&question.discrimination_band), + ); value.insert("p", Value::Float(question.p_value)); value.insert_some("point-biserial", question.point_biserial.map(Value::Float)); value.insert_some("discrimination", question.discrimination.map(Value::Float)); @@ -676,6 +838,17 @@ fn cohort_question_value(question: &CohortQuestionRow, content: bool) -> Value { .collect(), ), ); + value.insert( + "prediction-notes", + Value::Array( + question + .prediction_notes + .iter() + .map(|n| markup_value(n, content)) + .collect(), + ), + ); + value.insert("calibrated", Value::Bool(question.calibrated)); let mut by_form = Value::dict(); for (form, p) in &question.by_form { by_form.insert(form.clone(), Value::Float(*p)); @@ -953,7 +1126,11 @@ mod tests { levels: Vec::new(), objectives: Vec::new(), gaps: Vec::new(), + grades: Vec::new(), + lectures: Vec::new(), questions: Vec::new(), + triage: crate::diagnostic::Triage::default(), + predictions: crate::diagnostic::PredictionSummary::default(), revise: Vec::new(), forms: Vec::new(), blueprint: Vec::new(), diff --git a/src/export/typst/templates/cohort-report.typ b/src/export/typst/templates/cohort-report.typ index c7ce4ec..f0c5449 100644 --- a/src/export/typst/templates/cohort-report.typ +++ b/src/export/typst/templates/cohort-report.typ @@ -8,16 +8,21 @@ // // Markers: // -// // coursebank:begin meta course, assessment, class size, policy +// // coursebank:begin meta course, assessment, how many sat it, policy // // coursebank:end meta // // coursebank:begin data the class diagnostic // // coursebank:end data // // This document is for you, not for the class. It carries item statistics, the -// revision queue, and the per-option breakdown — everything the student report +// triage lists, and the per-option breakdown: everything the student report // withholds except the question text itself, which is in the bank where it -// belongs. Do not hand it out: an option table tells a reader which letter was -// keyed. +// belongs. Do not hand it out, because an option table tells a reader which +// letter was keyed. +// +// The order is deliberate. Where the class landed comes first, because that is +// the fact that changes next week. What to do with each question comes before +// the full item table, because the table is a reference and the triage lists are +// a work queue. #import "@preview/mitex:0.2.7": mi @@ -31,6 +36,7 @@ assessment: (id: "sample", title: "Sample assessment", date: "2026-01-01"), generator: (tool: "coursebank", version: "0.0.0", on: "2026-01-02"), policy: (mastery-threshold: 0.75, min-items-for-mastery: 2), + students-tested: 24, class-size: 24, extra: (:), ) @@ -41,7 +47,11 @@ students: 24, items: 36, distribution: ( - mean: 71.2, median: 73.0, sd: 11.4, min: 44.0, max: 94.0, + mean: 71.2, + median: 73.0, + sd: 11.4, + min: 44.0, + max: 94.0, bins: ( (low: 40, high: 50, count: 1), (low: 50, high: 60, count: 3), @@ -51,37 +61,111 @@ (low: 90, high: 100, count: 1), ), ), + grades: ( + (letter: "A", low: 93.0, high: 100.0, gpa: 4.0, group: "A", count: 1, share: 0.04, at-or-above: 1), + (letter: "B", low: 83.0, high: 92.9, gpa: 3.0, group: "B", count: 5, share: 0.21, at-or-above: 6), + (letter: "C", low: 73.0, high: 82.9, gpa: 2.0, group: "C", count: 10, share: 0.42, at-or-above: 16), + (letter: "D", low: 63.0, high: 72.9, gpa: 1.0, group: "D", count: 6, share: 0.25, at-or-above: 22), + (letter: "F", low: 0.0, high: 62.9, gpa: 0.0, group: "F", count: 2, share: 0.08, at-or-above: 24), + ), reliability: ( - alpha: 0.71, sem: 2.1, mean-p: 0.72, mean-point-biserial: 0.24, + alpha: 0.71, + sem: 2.1, + mean-p: 0.72, + mean-point-biserial: 0.24, interpretation: "Reliability is 0.71, acceptable for a classroom exam.", ), levels: ( (level: 1, name: "Remember", items: 6, rate: 0.91), (level: 3, name: "Apply", items: 9, rate: 0.64), ), + lectures: ( + ( + lecture: "L1.2", + title: "Entropy", + items: 4, + objectives: 3, + objectives-below: 2, + rate: 0.48, + questions: (10, 11, 12, 26), + worst-objective: [A sample objective the class struggled with.], + ), + ), objectives: ( - (id: "lo-sample-gap", text: [A sample objective the class struggled with.], - items: 3, rate: 0.41, meeting: 4, developing: 7, not-yet: 13, thin: 0, - below-threshold: true), - ), - gaps: ( - (id: "lo-sample-gap", text: [A sample objective the class struggled with.], - items: 3, rate: 0.41, meeting: 4, developing: 7, not-yet: 13, thin: 0, - below-threshold: true), + ( + id: "lo-sample-gap", + text: [A sample objective the class struggled with.], + items: 3, + rate: 0.41, + meeting: 4, + developing: 7, + not-yet: 13, + thin: 0, + below-threshold: true, + ), ), + gaps: (), questions: ( - (number: 1, item: "bank::q-sample-001", level: 1, objectives: ("lo-sample-gap",), - p: 0.42, point-biserial: 0.05, discrimination: 0.10, blank-rate: 0.0, - key: ("B",), - options: ( - (letter: "A", count: 9, rate: 0.375, is-key: false, point-biserial: 0.11, nonfunctioning: false), - (letter: "B", count: 10, rate: 0.417, is-key: true, point-biserial: 0.05, nonfunctioning: false), - (letter: "C", count: 5, rate: 0.208, is-key: false, point-biserial: -0.2, nonfunctioning: false), - (letter: "D", count: 0, rate: 0.0, is-key: false, nonfunctioning: true), - ), - flags: ("ambiguous",), - notes: ([Distractor A drew as many strong students as the key.],), - by-form: (A: 0.55, B: 0.30)), + ( + number: 1, + item: "bank::q-sample-001", + level: 1, + objectives: ("lo-sample-gap",), + objective-texts: ([A sample objective the class struggled with.],), + taught-in: ("Entropy (L1.2), slides 4, 5",), + lectures: ("L1.2",), + difficulty-band: "moderate", + discrimination-band: "poor", + p: 0.42, + point-biserial: 0.05, + discrimination: 0.10, + blank-rate: 0.0, + key: ("B",), + options: ( + (letter: "A", count: 9, rate: 0.375, is-key: false, point-biserial: 0.11, nonfunctioning: false), + (letter: "B", count: 10, rate: 0.417, is-key: true, point-biserial: 0.05, nonfunctioning: false), + (letter: "C", count: 5, rate: 0.208, is-key: false, point-biserial: -0.2, nonfunctioning: false), + (letter: "D", count: 0, rate: 0.0, is-key: false, nonfunctioning: true), + ), + flags: ("ambiguous",), + notes: ([Distractor A drew as many strong students as the key.],), + prediction-notes: ([You predicted about 60% correct and observed 42%.],), + calibrated: false, + by-form: (A: 0.55, B: 0.30), + ), + ), + triage: ( + discard: (), + rekey: ( + ( + number: 1, + item: "bank::q-sample-001", + level: 1, + p: 0.42, + point-biserial: 0.05, + discrimination: 0.1, + objectives: ([A sample objective the class struggled with.],), + taught-in: ("Entropy (L1.2), slides 4, 5",), + option: "A", + option-share: 0.375, + option-point-biserial: 0.11, + reasons: ([Option A drew 38% and tracks total score better than the key.],), + ), + ), + revise: (), + reteach: (), + bounded: (), + clean: 35, + ), + predictions: ( + predicted: 36, + calibrated: 0, + mean-signed-error: 0.14, + mean-abs-error: 0.19, + within: 21, + band: 36, + band-hit: 11, + biggest-surprise: (number: 14, expected: 0.45, observed: 0.86), ), revise: (), forms: ( @@ -103,34 +187,63 @@ #let body-font = extra.at("font", default: "Roboto") #let body-size = eval(extra.at("font-size", default: "9.5pt")) #let paper = extra.at("paper", default: "us-letter") + #let show-options = extra.at("option-tables", default: true) +#let show-triage = extra.at("triage", default: true) +#let show-guide = extra.at("stat-guide", default: true) +#let show-map = extra.at("item-map", default: true) +#let show-lectures = extra.at("lecture-table", default: true) +#let show-predictions = extra.at("prediction-check", default: true) #let threshold = cb-meta.policy.at("mastery-threshold", default: 0.75) +#let n-tested = cb-meta.at("students-tested", default: cb-meta.at("class-size", default: 0)) + +// The same scale the student report uses, so the two documents look related. +#let size-tag = 0.62em +#let size-micro = 0.72em // column heads, badges, legends +#let size-meta = 0.9em // item ids, provenance +#let size-small = 0.88em // table cells +#let size-lead = 0.94em // section explanations +#let size-h2 = 1.02em +#let size-h1 = 1.15em +#let size-sub = 0.95em +#let size-display = 1.35em +#let size-title = 1.5em + +#let step = 1.0em +#let entry-gap = 0.9em +#let prose-pad = 3.2cm +#let prose-pad-inset = prose-pad - 1cm #let ok-color = rgb("#2a9d8f") #let mid-color = rgb("#D19F1F") #let bad-color = rgb("#E24E29") +#let thin-color = luma(150) #set page( paper: paper, margin: (x: 1.7cm, y: 2.0cm), - header: text(size: 0.8em, fill: luma(110))[ + header: text(size: size-meta, fill: luma(110))[ #cb-meta.course.code · #cb-meta.assessment.title · class diagnostic #h(1fr) instructor copy ], - footer: context text(size: 0.8em, fill: luma(110))[ + footer: context text(size: size-meta, fill: luma(110))[ #h(1fr) Page #counter(page).display("1 of 1", both: true) ], ) #set text(font: body-font, size: body-size, lang: "en") #set par(justify: false, leading: 0.6em) -#show heading.where(level: 1): it => block(above: 1.4em, below: 0.7em)[ - #text(size: 1.15em, weight: "bold", fill: accent)[#it.body] +#show table: set text(number-width: "tabular") +#show heading.where(level: 1): it => block(above: 1.5em, below: entry-gap)[ + #text(size: size-h1, weight: "bold", fill: accent)[#it.body] #v(-0.45em) #line(length: 100%, stroke: 0.6pt + accent.lighten(55%)) ] +#show heading.where(level: 2): it => block(above: 1em, below: 0.4em)[ + #text(size: size-h2, weight: "bold")[#it.body] +] // ───────────────────────────────────────────────────────────────────────────── // Helpers @@ -140,18 +253,52 @@ #let pct(rate) = str(calc.round(rate * 100)) + "%" #let num(value, digits: 2) = str(calc.round(value, digits: digits)) #let signed(value) = (if value >= 0 { "+" } else { "" }) + num(value) +#let plural(n, one, many) = if n == 1 { one } else { many } + +#let th(body) = text(size: size-micro, fill: luma(95), tracking: 0.04em)[#body] + +#let explain(body) = block(below: entry-gap)[ + #pad(right: prose-pad)[#text(size: size-lead, fill: luma(95))[#body]] +] #let rate-color(rate) = { - if rate >= threshold { ok-color } - else if rate >= threshold * 0.6 { mid-color } - else { bad-color } + if rate >= threshold { ok-color } else if rate >= threshold * 0.6 { mid-color } else { bad-color } +} + +// The conventional reading of a corrected item-total correlation. Below about +// 0.20 an item is doing little sorting; negative means it sorts backwards. +#let r-color(r) = { + if r == none { thin-color } else if r < 0.0 { bad-color } else if r < 0.20 { mid-color } else { ok-color } +} + +#let group-color(group) = { + if group == "A" { ok-color } else if group == "B" { ok-color.lighten(25%) } else if group == "C" { + mid-color + } else if group == "D" { bad-color.lighten(25%) } else { bad-color } } #let badge(label, color) = box( fill: color.lighten(82%), radius: 3pt, inset: (x: 4pt, y: 2pt), -)[#text(size: 0.7em, weight: "bold", fill: color.darken(12%))[#label]] +)[#text(size: size-micro, weight: "bold", fill: color.darken(12%))[#label]] + +// Flags travel as machine names. Printed in full they wrap a table column into +// three lines; these are the same findings in the space available. +#let flag-labels = ( + negative_discrimination: "negative r", + low_discrimination: "low r", + distractor_outperforms_key: "distractor > key", + key_underperforms: "key split", + too_easy: "easy", + too_hard: "hard", + high_rapid_guess: "rapid", + nonfunctioning_distractor: "dead option", + dif_flagged: "group gap", + ambiguous: "ambiguous", + design_mismatch: "off prediction", +) +#let flag-badge(flag) = badge(flag-labels.at(flag, default: flag), bad-color) #let bar(rate, color: accent, width: 2.4cm) = { let r = calc.max(0.0, calc.min(1.0, rate)) @@ -164,46 +311,94 @@ ] } -#let stat-card(label, value, note: none) = block( - width: 100%, - fill: luma(247), - radius: 4pt, - inset: (x: 9pt, y: 8pt), -)[ - #text(size: 0.74em, fill: luma(95))[#upper(label)] - #v(0.12em) - #text(size: 1.35em, weight: "bold", fill: accent)[#value] - #if note != none [ - #v(0.08em) - #text(size: 0.76em, fill: luma(95))[#note] +#let stat-card(label, value, note: none) = { + let parts = ( + text(size: size-micro, fill: luma(95), tracking: 0.06em)[#upper(label)], + text(size: size-display, weight: "bold", fill: accent)[#value], + ) + if note != none { parts.push(text(size: size-meta, fill: luma(95))[#note]) } + block(width: 100%, fill: luma(247), radius: 4pt, inset: (x: 9pt, y: 8pt))[ + #stack(dir: ttb, spacing: step * 0.4, ..parts) ] -] +} + +// One question's identity line, used wherever a question is discussed rather +// than tabulated. +#let question-head(row, color) = { + let parts = ( + { + let bits = (text(weight: "bold", fill: color)[Question #row.number],) + let item = row.at("item", default: none) + if item != none { bits.push(text(size: size-meta, fill: luma(120))[#item]) } + let level = row.at("level", default: none) + if level != none { bits.push(text(size: size-meta, fill: luma(120))[L#level]) } + bits.join(h(0.5em)) + }, + ) + let stats = () + stats.push("p = " + num(row.p)) + let r = row.at("point-biserial", default: none) + if r != none { stats.push("r = " + signed(r)) } + let d = row.at("discrimination", default: none) + if d != none { stats.push("D = " + signed(d)) } + parts.push(text(size: size-meta, fill: luma(110))[#stats.join(" · ")]) + stack(dir: ttb, spacing: step * 0.4, ..parts) +} + +// The objective and lecture context for one question. +#let question-context(row) = { + let parts = () + // A question row carries ids in `objectives` and prose in `objective-texts`; a + // triage row carries the prose under `objectives`. One lookup covers both. + let objectives = row.at("objective-texts", default: row.at("objectives", default: ())) + if objectives.len() > 0 { + parts.push(text(size: size-meta, fill: luma(105))[ + #text(weight: "bold")[Measured:] #objectives.map(o => markup(o)).join([; ]) + ]) + } + let taught = row.at("taught-in", default: ()) + if taught.len() > 0 { + parts.push(text(size: size-meta, fill: luma(105))[ + #text(weight: "bold")[Taught in:] #taught.join("; ") + ]) + } + if parts.len() == 0 { return none } + stack(dir: ttb, spacing: step * 0.65, ..parts) +} // ───────────────────────────────────────────────────────────────────────────── // Heading and headline numbers // ───────────────────────────────────────────────────────────────────────────── -#block(below: 1em)[ - #text(size: 1.5em, weight: "bold")[#cb-meta.assessment.title — class diagnostic] - #linebreak() - #text(size: 0.95em, fill: luma(110))[ - #cb-meta.course.code · #cb-meta.course.term - #{ - let date = cb-meta.assessment.at("date", default: none) - if date != none [ · administered #date ] - } - ] +#block(below: entry-gap)[ + #stack( + dir: ttb, + spacing: step * 0.5, + text(size: size-title, weight: "bold")[#cb-meta.assessment.title · class diagnostic], + text(size: size-sub, fill: luma(110))[ + #cb-meta.course.code · #cb-meta.course.term + #{ + let date = cb-meta.assessment.at("date", default: none) + if date != none [ · administered #date ] + } + ], + ) ] #let dist = cb-data.distribution #let rel = cb-data.reliability +#let grades = cb-data.at("grades", default: ()) #grid( columns: (1fr, 1fr, 1fr, 1fr), gutter: 9pt, stat-card("students", str(cb-data.students), note: str(cb-data.items) + " scored items"), stat-card("mean", str(calc.round(dist.mean)) + "%", note: "median " + str(calc.round(dist.median)) + "%"), - stat-card("spread", "SD " + num(dist.sd, digits: 1), note: str(calc.round(dist.min)) + "–" + str(calc.round(dist.max)) + "%"), + stat-card( + "spread", + "SD " + num(dist.sd, digits: 1), + note: str(calc.round(dist.min)) + "–" + str(calc.round(dist.max)) + "%", + ), stat-card( "reliability", { @@ -217,26 +412,112 @@ ), ) -#v(0.6em) -#text(size: 0.88em, fill: luma(90))[#rel.interpretation] +#block(above: entry-gap)[ + #pad(right: prose-pad)[#text(size: size-lead, fill: luma(90))[#rel.interpretation]] +] // ───────────────────────────────────────────────────────────────────────────── -// Distribution +// Where the class landed // ───────────────────────────────────────────────────────────────────────────── #let bins = dist.at("bins", default: ()) -#let peak = if bins.len() > 0 { calc.max(..bins.map(b => b.count)) } else { 0 } -#if peak > 0 [ +#if grades.len() > 0 [ + = Where the class landed + + #explain[ + Scores binned by the course's own letter scale rather than by ten-point + interval, because the bands are what the class will see. The running column + is cumulative: how many students are at that letter or above. + ] + + #let peak = calc.max(..grades.map(g => g.count), 1) + #let height = 2.8cm + + #align(center)[ + #grid( + columns: (1fr,) * grades.len(), + column-gutter: 3pt, + ..grades + .rev() + .map(g => { + let share = g.count / peak + let color = group-color(g.at("group", default: g.letter)) + stack( + dir: ttb, + spacing: 3pt, + align(center + bottom)[ + #box(height: height)[ + #align(bottom + center)[ + #stack( + dir: ttb, + spacing: 2pt, + align(center)[ + #text(size: size-micro, weight: "bold", fill: if g.count > 0 { color.darken(20%) } else { + thin-color + })[#g.count] + ], + box( + width: 100%, + height: height * share, + fill: color.lighten(30%), + stroke: 0.5pt + color, + radius: (top: 2pt), + ), + ) + ] + ] + ], + align(center)[#text(size: size-small, weight: "bold")[#g.letter]], + align(center)[#text(size: size-tag, fill: luma(115))[#num(g.low, digits: 0)+]], + ) + }), + ) + ] + + #v(0.5em) + + // The sentence the histogram is for. Whichever band the median sits in is + // where the class is, and the share at or below the failing band is the number + // that decides whether this was an exam problem or a teaching one. + #let failing = grades.filter(g => g.at("group", default: "") == "F") + #let n-failing = if failing.len() > 0 { failing.at(0).count } else { 0 } + #let top = grades.filter(g => g.count > 0) + + #pad(right: prose-pad)[ + #text(size: size-lead)[ + The median score of #num(dist.median, digits: 0)% falls in the + #{ + let median-band = grades.filter(g => dist.median >= g.low) + if median-band.len() > 0 { [*#median-band.at(0).letter*] } else { [lowest] } + } + band. + #if n-failing > 0 [ + #n-failing of #cb-data.students students, #pct(n-failing / calc.max(cb-data.students, 1)), + are below the lowest passing band. + ] + #if top.len() > 0 [ + The highest band anyone reached is *#top.at(0).letter*. + ] + ] + ] +] else if bins.len() > 0 [ = Score distribution - #let height = 3.2cm + #explain[ + Ten-point bins, because `course.yaml` sets no `policy.grade_scale`. Add one + and this becomes a letter-grade histogram, which is the version worth + reading. + ] + + #let peak = calc.max(..bins.map(b => b.count), 1) + #let height = 3.0cm #align(center)[ #grid( columns: (1fr,) * bins.len(), - gutter: 5pt, + column-gutter: 5pt, ..bins.map(b => { - let share = if peak > 0 { b.count / peak } else { 0.0 } + let share = b.count / peak stack( dir: ttb, spacing: 3pt, @@ -252,13 +533,13 @@ ] ] ], - align(center)[#text(size: 0.72em, weight: "bold")[#b.count]], - align(center)[#text(size: 0.68em, fill: luma(110))[#b.low]], + align(center)[#text(size: size-micro, weight: "bold")[#b.count]], + align(center)[#text(size: size-tag, fill: luma(110))[#b.low]], ) }), ) ] - #align(center)[#text(size: 0.72em, fill: luma(120))[percent scored, in ten-point bins]] + #align(center)[#text(size: size-micro, fill: luma(120))[percent scored, in ten-point bins]] ] // ───────────────────────────────────────────────────────────────────────────── @@ -270,253 +551,683 @@ #if forms.len() > 1 [ = Forms - #text(size: 0.9em, fill: luma(95))[ + #explain[ Forms differ only in order, so their means should differ only by who sat them. A persistent gap points at an item whose permutation made it easier or harder, and the per-question columns further down are where to look. ] - #v(0.4em) - #table( columns: (auto, auto, auto, auto), stroke: none, align: (left, right, right, right), - inset: (x: 7pt, y: 4pt), + inset: (x: 7pt, y: 5pt), fill: (_, row) => if calc.odd(row) { luma(250) } else { white }, - table.header( - text(size: 0.78em, fill: luma(95))[FORM], - text(size: 0.78em, fill: luma(95))[STUDENTS], - text(size: 0.78em, fill: luma(95))[MEAN], - text(size: 0.78em, fill: luma(95))[SD], - ), - ..forms.map(form => ( - [*#form.id*], - [#form.students], - [#num(form.mean, digits: 1)%], - [#num(form.sd, digits: 1)], - )).flatten(), + table.header(th[FORM], th[STUDENTS], th[MEAN], th[SD]), + ..forms + .map(form => ( + [*#form.id*], + [#form.students], + [#num(form.mean, digits: 1)%], + [#num(form.sd, digits: 1)], + )) + .flatten(), ) ] // ───────────────────────────────────────────────────────────────────────────── -// Levels +// How to read the item statistics // ───────────────────────────────────────────────────────────────────────────── -#if cb-data.levels.len() > 0 [ - = By level +#if show-guide [ + = How to read the item statistics - #table( - columns: (auto, auto, auto, 1fr), - stroke: none, - align: (left + horizon, right + horizon, right + horizon, left + horizon), - inset: (x: 7pt, y: 4pt), - fill: (_, row) => if calc.odd(row) { luma(250) } else { white }, - table.header( - text(size: 0.78em, fill: luma(95))[LEVEL], - text(size: 0.78em, fill: luma(95))[ITEMS], - text(size: 0.78em, fill: luma(95))[CLASS], - text(size: 0.78em, fill: luma(95))[], - ), - ..cb-data.levels.map(level => ( - [#level.level · *#level.name*], - [#level.items], - [#pct(level.rate)], - bar(level.rate, color: rate-color(level.rate), width: 5cm), - )).flatten(), - ) -] - -// ───────────────────────────────────────────────────────────────────────────── -// Objectives -// ───────────────────────────────────────────────────────────────────────────── - -#let objectives = cb-data.at("objectives", default: ()) - -#if objectives.len() > 0 [ - = By objective, worst first - - #text(size: 0.9em, fill: luma(95))[ - The counts split the class into how many are meeting, developing, and not yet - on each objective. An objective where the class divides evenly is a different - teaching problem from one where nearly everyone is short. + #explain[ + Three numbers describe each question, and they only mean anything together. ] - #v(0.4em) - #table( - columns: (1fr, auto, auto, auto, auto, auto, auto), + columns: (auto, 1fr), stroke: none, - align: (left + horizon, right + horizon, right + horizon, right + horizon, right + horizon, right + horizon, left + horizon), - inset: (x: 6pt, y: 4pt), + align: (left + top, left + top), + inset: (x: 7pt, y: 5pt), fill: (_, row) => if calc.odd(row) { luma(250) } else { white }, - table.header( - text(size: 0.78em, fill: luma(95))[OBJECTIVE], - text(size: 0.78em, fill: luma(95))[Q], - text(size: 0.78em, fill: luma(95))[CLASS], - text(size: 0.78em, fill: luma(95))[MET], - text(size: 0.78em, fill: luma(95))[DEV], - text(size: 0.78em, fill: luma(95))[NOT], - text(size: 0.78em, fill: luma(95))[], - ), - ..objectives.map(objective => ( - { - markup(objective.text) - if objective.at("below-threshold", default: false) { - [ #badge("below " + pct(threshold), bad-color)] - } - }, - [#objective.items], - [#pct(objective.rate)], - [#objective.meeting], - [#objective.developing], - [#objective.at("not-yet", default: 0)], - bar(objective.rate, color: rate-color(objective.rate), width: 2.2cm), - )).flatten(), + table.header(th[], th[WHAT IT TELLS YOU]), + [*p* #text(size: size-meta, fill: luma(110))[difficulty]], + [ + The share who answered correctly, counting blanks as wrong. On its own it + says almost nothing: an item everyone passes may be a deliberate anchor, + and an item everyone misses is only a problem if it also failed to sort + students. + ], + + [*r* #text(size: size-meta, fill: luma(110))[discrimination]], + [ + The corrected item-total correlation: how well this question agrees with + the rest of the exam. This is the column to read first. Conventional + guidance puts 0.40 and above at excellent, 0.30 to 0.39 good, 0.20 to 0.29 + marginal, and below 0.20 poor. Negative means the students who did well + overall did worse here, which is almost always a keying error or a stem + with a second reading. + ], + + [*D* #text(size: size-meta, fill: luma(110))[upper minus lower]], + [ + The same idea computed crudely: the pass rate in the top quarter of the + class minus the pass rate in the bottom quarter. It uses only the extremes + and throws the middle away, so where D and r disagree, trust r. + ], ) -] -// ───────────────────────────────────────────────────────────────────────────── -// Items -// ───────────────────────────────────────────────────────────────────────────── - -#let questions = cb-data.at("questions", default: ()) - -#if questions.len() > 0 [ - #pagebreak(weak: true) - = Item analysis - - #text(size: 0.9em, fill: luma(95))[ - #emph[p] is the share answering correctly; #emph[r] is the corrected - item-total correlation, which should be positive and is the single most - useful column; #emph[D] is the upper group minus the lower group. Option - letters are the bank's, not any one form's. - ] - - #v(0.4em) - - #table( - columns: (auto, auto, auto, auto, auto, auto, 1fr), - stroke: none, - align: (right + horizon, right + horizon, right + horizon, right + horizon, right + horizon, right + horizon, left + horizon), - inset: (x: 5pt, y: 3.5pt), - fill: (_, row) => if calc.odd(row) { luma(250) } else { white }, - table.header( - text(size: 0.78em, fill: luma(95))[Q], - text(size: 0.78em, fill: luma(95))[LVL], - text(size: 0.78em, fill: luma(95))[KEY], - text(size: 0.78em, fill: luma(95))[p], - text(size: 0.78em, fill: luma(95))[r], - text(size: 0.78em, fill: luma(95))[D], - text(size: 0.78em, fill: luma(95))[FLAGS], - ), - ..questions.map(q => ( - [*#q.number*], - { - let level = q.at("level", default: none) - if level != none [#level] else [] - }, - [#q.at("key", default: ()).join("")], - text(fill: rate-color(q.p))[#num(q.p)], - { - let r = q.at("point-biserial", default: none) - if r == none { text(fill: luma(140))[n/a] } - else if r < 0.0 { text(fill: bad-color, weight: "bold")[#signed(r)] } - else if r < 0.15 { text(fill: mid-color)[#signed(r)] } - else { [#signed(r)] } - }, - { - let d = q.at("discrimination", default: none) - if d != none [#signed(d)] else [—] - }, - { - let flags = q.at("flags", default: ()) - let forms-gap = { - let by-form = q.at("by-form", default: (:)) - let values = by-form.values() - if values.len() > 1 and calc.max(..values) - calc.min(..values) >= 0.25 { - (badge("form gap " + pct(calc.max(..values) - calc.min(..values)), mid-color),) - } else { () } - } - stack( - dir: ltr, - spacing: 3pt, - ..flags.map(f => badge(f, bad-color)), - ..forms-gap, - ) - }, - )).flatten(), - ) -] - -// ───────────────────────────────────────────────────────────────────────────── -// The revise queue -// ───────────────────────────────────────────────────────────────────────────── - -#let revise = cb-data.at("revise", default: ()) - -#if revise.len() > 0 [ - = Before you use these again - - #for q in revise [ - #block(breakable: false, above: 0.9em, width: 100%)[ - #text(weight: "bold", fill: accent)[Question #q.number] - #{ - let item = q.at("item", default: none) - if item != none [ #text(size: 0.8em, fill: luma(120))[#item]] - } - #h(0.5em) - #stack(dir: ltr, spacing: 3pt, ..q.at("flags", default: ()).map(f => badge(f, bad-color))) - - #v(0.25em) - #for note in q.at("notes", default: ()) [ - #text(size: 0.9em)[— #markup(note)] - #linebreak() - ] - - #if show-options and q.at("options", default: ()).len() > 0 [ - #v(0.3em) - #table( - columns: (auto, auto, auto, auto, 1fr), - stroke: none, - align: (center + horizon, right + horizon, right + horizon, left + horizon, left + horizon), - inset: (x: 5pt, y: 3pt), - table.header( - text(size: 0.74em, fill: luma(95))[OPT], - text(size: 0.74em, fill: luma(95))[n], - text(size: 0.74em, fill: luma(95))[SHARE], - text(size: 0.74em, fill: luma(95))[], - text(size: 0.74em, fill: luma(95))[r], - ), - ..q.options.map(option => ( - { - if option.at("is-key", default: false) { - text(weight: "bold", fill: ok-color)[#option.letter] - } else { [#option.letter] } - }, - [#option.count], - [#pct(option.rate)], - bar( - option.rate, - color: if option.at("is-key", default: false) { ok-color } else { luma(160) }, - width: 2.6cm, - ), - { - let r = option.at("point-biserial", default: none) - if r == none { text(fill: luma(150))[—] } - else if not option.at("is-key", default: false) and r > 0.0 { - text(fill: mid-color)[#signed(r)] - } else { text(size: 0.95em)[#signed(r)] } - }, - )).flatten(), - ) + #block(above: entry-gap, breakable: false)[ + #pad(right: prose-pad)[ + #text(size: size-lead, fill: luma(95))[ + One constraint governs all of this: *discrimination is bounded by + difficulty*. An item that nearly everyone passes has almost no variance + left to correlate with anything, so a low r on a question at 90% correct + is arithmetic rather than a fault. That is why the triage below separates + questions whose weak r is explained by their difficulty from questions + that had room to sort students and did not. ] ] ] ] // ───────────────────────────────────────────────────────────────────────────── -// Blueprint and cautions +// The item map +// ───────────────────────────────────────────────────────────────────────────── + +#let questions = cb-data.at("questions", default: ()) + +#if show-map and questions.len() > 0 [ + = The exam at a glance + + #explain[ + Every question placed by difficulty and discrimination. The top-left cell is + where a well-built exam puts most of its items. The bottom row is where the + work is, and the right-hand column is where low discrimination is at least + partly explained by the ceiling. + ] + + #let d-bands = (("hard", "hard"), ("moderate", "moderate"), ("too easy", "at the ceiling")) + #let r-bands = ( + ("excellent", "excellent", ok-color), + ("good", "good", ok-color), + ("marginal", "marginal", mid-color), + ("poor", "poor", bad-color), + ("negative", "negative", bad-color), + ) + + #let in-cell(r-key, d-key) = questions.filter(q => ( + q.at("discrimination-band", default: "") == r-key and q.at("difficulty-band", default: "") == d-key + )) + + #table( + columns: (3.4cm, 1fr, 1fr, 1fr), + stroke: 0.5pt + luma(225), + align: left + top, + inset: 5pt, + fill: (x, y) => { + if y == 0 or x == 0 { white } else if in-cell(r-bands.at(y - 1).at(0), d-bands.at(x - 1).at(0)).len() > 0 { + r-bands.at(y - 1).at(2).lighten(93%) + } else { white } + }, + table.header(th[], ..d-bands.map(pair => align(center)[#th[#upper(pair.at(1))]])), + ..r-bands + .map(band => ( + { + let range = if band.at(0) == "excellent" { + "r ≥ 0.40" + } else if band.at(0) == "good" { + "0.30–0.39" + } else if band.at(0) == "marginal" { + "0.20–0.29" + } else if band.at(0) == "poor" { "0.00–0.19" } else { "r < 0" } + stack( + dir: ttb, + spacing: step * 0.3, + text(size: size-small, weight: "bold", fill: band.at(2))[#band.at(1)], + text(size: size-tag, fill: luma(120))[#range], + ) + }, + ..d-bands.map(pair => { + let inside = in-cell(band.at(0), pair.at(0)) + if inside.len() == 0 { + text(size: size-micro, fill: luma(200))[—] + } else { + text(size: size-small)[#inside.map(q => str(q.number)).join(" ")] + } + }), + )) + .flatten(), + ) + + #v(0.4em) + #pad(right: prose-pad)[ + #text(size: size-micro, fill: luma(120))[ + Difficulty: hard is 35% correct or below, at the ceiling is 85% or above. + A question with no variance at all appears in no cell. + ] + ] +] + +// ───────────────────────────────────────────────────────────────────────────── +// Triage +// ───────────────────────────────────────────────────────────────────────────── + +#let triage = cb-data.at("triage", default: (:)) + +#let triage-block(rows, title, color, lead) = { + if rows.len() == 0 { return none } + block(breakable: true, above: entry-gap, width: 100%)[ + == #title #text(size: size-meta, fill: luma(110), weight: "regular")[ + #rows.len() #plural(rows.len(), "question", "questions") + ] + + #pad(right: prose-pad)[#text(size: size-lead, fill: luma(95))[#lead]] + + #for row in rows [ + #block( + breakable: false, + above: step * 1.6, + width: 100%, + inset: (left: 0.6em), + stroke: (left: 2pt + color.lighten(45%)), + )[ + #{ + let parts = (question-head(row, color),) + let context_ = question-context(row) + if context_ != none { parts.push(context_) } + for reason in row.at("reasons", default: ()) { + parts.push(text(size: size-small)[#markup(reason)]) + } + let option = row.at("option", default: none) + if option != none { + let share = row.at("option-share", default: none) + let r = row.at("option-point-biserial", default: none) + let bits = ("option " + option,) + if share != none { bits.push("chosen by " + pct(share)) } + if r != none { bits.push("r = " + signed(r)) } + parts.push(text(size: size-meta, fill: luma(110))[ + #text(weight: "bold")[The option in question:] #bits.join(", ") + ]) + } + pad(right: prose-pad-inset)[#stack(dir: ttb, spacing: step * 0.6, ..parts)] + } + ] + ] + ] +} + +#if show-triage [ + #pagebreak(weak: true) + = What to do with each question + + #explain[ + Every question that raised something, sorted by what it asks of you. The + first three lists are decisions about items and are exclusive: there is no + point rewriting a distractor on a question you are about to discard. The + fourth is not about the items at all. + #{ + let clean = triage.at("clean", default: 0) + if clean > 0 [ + #clean #plural(clean, "question", "questions") raised nothing and are not + listed. + ] + } + ] + + #triage-block( + triage.at("discard", default: ()), + [Consider dropping from the score], + bad-color, + [ + The evidence says these did not measure what they were scored on. Dropping + a question raises every student's percentage, so it is a decision about + fairness rather than about the item: a question that sorted students + backwards contributed noise to every score it touched. + ], + ) + + #triage-block( + triage.at("rekey", default: ()), + [Consider credit for a second answer], + mid-color, + [ + A distractor here drew a substantial share of the class *and* tracked + overall performance at least as well as the key. That is the pattern of a + second defensible reading rather than a popular mistake. Either credit it + for this administration or rewrite the stem to exclude it. + ], + ) + + #triage-block( + triage.at("revise", default: ()), + [Rewrite before reusing], + mid-color, + [ + Weak but salvageable. These had room to separate students and did not, or + carry an option nobody chose, which makes a four-option question really a + three-option one. + ], + ) + + #triage-block( + triage.at("reteach", default: ()), + [Teach again], + accent, + [ + These questions worked. They separated the class cleanly, and the class + still missed them, which makes this a list about your teaching plan rather + than about your item bank. This is the list to bring to the next class + meeting. + ], + ) + + #triage-block( + triage.at("bounded", default: ()), + [No action needed], + ok-color, + [ + These carry a low-discrimination flag that their difficulty explains. They + are listed so the flag does not send you rewriting a question that is doing + exactly what an anchor item should do. + ], + ) +] + +// ───────────────────────────────────────────────────────────────────────────── +// By level +// ───────────────────────────────────────────────────────────────────────────── + +#if cb-data.levels.len() > 0 [ + + #block(breakable: false)[ + = By level + + #explain[ + Where the class sits on each kind of thinking. A drop from one level to the + next is expected; a cliff is worth reading as a gap in what was practised + rather than in what was taught. + ] + + #table( + columns: (auto, auto, auto, 1fr), + stroke: none, + align: (left + horizon, right + horizon, right + horizon, left + horizon), + inset: (x: 7pt, y: 5pt), + fill: (_, row) => if calc.odd(row) { luma(250) } else { white }, + table.header(th[LEVEL], th[ITEMS], th[CLASS], th[]), + ..cb-data + .levels + .map(level => ( + [#level.level · *#level.name*], + [#level.items], + [#pct(level.rate)], + bar(level.rate, color: rate-color(level.rate), width: 5cm), + )) + .flatten(), + ) + ] +] + +// ───────────────────────────────────────────────────────────────────────────── +// By lecture +// ───────────────────────────────────────────────────────────────────────────── + +#let lectures = cb-data.at("lectures", default: ()) + +#if show-lectures and lectures.len() > 0 [ + = By lecture, worst first + + #explain[ + Every question traces back to the lecture it was written from, so the item + statistics roll up into a view of the syllabus. This is the objective table + grouped into class meetings: one weak objective is a note, but three weak + objectives from the same lecture is a morning to reteach. + ] + + #table( + columns: (auto, 1fr, auto, auto, auto, 3.2cm), + stroke: none, + align: (left + horizon, left + horizon, right + horizon, right + horizon, right + horizon, left + horizon), + inset: (x: 6pt, y: 5pt), + fill: (_, row) => if calc.odd(row) { luma(250) } else { white }, + table.header(th[LEC], th[TITLE], th[Q], th[LO], th[CLASS], th[]), + ..lectures + .map(lecture => ( + text(size: size-small, fill: luma(110))[#lecture.lecture], + { + let parts = ([*#lecture.title*],) + let worst = lecture.at("worst-objective", default: none) + if worst != none { + parts.push(text(size: size-meta, fill: luma(110))[weakest: #markup(worst)]) + } + stack(dir: ttb, spacing: step * 0.4, ..parts) + }, + [#lecture.items], + { + let below = lecture.at("objectives-below", default: 0) + let total = lecture.at("objectives", default: 0) + if below > 0 { + text(fill: bad-color, weight: "bold")[#below/#total] + } else { + [#total] + } + }, + [#pct(lecture.rate)], + bar(lecture.rate, color: rate-color(lecture.rate), width: 3cm), + )) + .flatten(), + ) + + #v(0.4em) + #pad(right: prose-pad)[ + #text(size: size-micro, fill: luma(120))[ + LO counts the objectives the lecture's questions measured; a red fraction is + how many of them the class did not meet at #pct(threshold). + ] + ] +] + +// ───────────────────────────────────────────────────────────────────────────── +// By objective +// ───────────────────────────────────────────────────────────────────────────── + +#let objectives = cb-data.at("objectives", default: ()) + +#if objectives.len() > 0 [ + #pagebreak(weak: true) + = By objective, worst first + + #explain[ + The counts split the class into how many are meeting, developing, and not yet + on each objective. An objective where the class divides evenly is a different + teaching problem from one where nearly everyone is short. Where an objective + was measured by a single question, the split is thin evidence and the class + rate is the number to read. + ] + + #table( + columns: (1fr, auto, auto, auto, auto, auto, 2.2cm), + stroke: none, + align: ( + left + horizon, + right + horizon, + right + horizon, + right + horizon, + right + horizon, + right + horizon, + left + horizon, + ), + inset: (x: 6pt, y: 5pt), + fill: (_, row) => if calc.odd(row) { luma(250) } else { white }, + table.header(th[OBJECTIVE], th[Q], th[CLASS], th[MET], th[DEV], th[NOT], th[]), + ..objectives + .map(objective => ( + text(size: size-small)[#markup(objective.text)], + text(size: size-small, fill: luma(110))[#objective.items], + text(size: size-small, weight: "medium")[#pct(objective.rate)], + text(size: size-small)[#objective.meeting], + text(size: size-small)[#objective.developing], + text(size: size-small)[#objective.at("not-yet", default: 0)], + bar(objective.rate, color: rate-color(objective.rate), width: 2cm), + )) + .flatten(), + ) +] + +// ───────────────────────────────────────────────────────────────────────────── +// Item analysis +// ───────────────────────────────────────────────────────────────────────────── + +#if questions.len() > 0 [ + #pagebreak(weak: true) + = Item analysis + + #explain[ + One row per question, in the order they were numbered. Option letters are the + bank's, not any one form's. See the guide above for what p, r, and D mean; the + r column is colour-coded against the conventional cut points. + ] + + #table( + columns: (auto, auto, auto, auto, auto, auto, auto, 1fr), + stroke: none, + align: ( + right + horizon, + right + horizon, + left + horizon, + right + horizon, + right + horizon, + right + horizon, + left + horizon, + left + horizon, + ), + inset: (x: 5pt, y: 4pt), + fill: (_, row) => if calc.odd(row) { luma(250) } else { white }, + table.header(th[Q], th[LVL], th[KEY], th[p], th[r], th[D], th[LEC], th[FLAGS]), + ..questions + .map(q => ( + [*#q.number*], + { + let level = q.at("level", default: none) + if level != none [#level] else [] + }, + [#q.at("key", default: ()).join("")], + text(fill: rate-color(q.p))[#num(q.p)], + { + let r = q.at("point-biserial", default: none) + if r == none { + text(fill: thin-color)[n/a] + } else { + text(fill: r-color(r), weight: if r < 0.2 { "bold" } else { "regular" })[#signed(r)] + } + }, + { + let d = q.at("discrimination", default: none) + if d != none [#signed(d)] else [—] + }, + { + // The lecture id only, which is all that fits and all that is needed to + // find the question in the bank. + let ids = q.at("lectures", default: ()) + if ids.len() > 0 { + text(size: size-tag, fill: luma(110))[#ids.join(", ")] + } else { [] } + }, + { + let flags = q.at("flags", default: ()) + let forms-gap = { + let by-form = q.at("by-form", default: (:)) + let values = by-form.values() + if values.len() > 1 and calc.max(..values) - calc.min(..values) >= 0.25 { + (badge("form gap " + pct(calc.max(..values) - calc.min(..values)), mid-color),) + } else { () } + } + stack(dir: ltr, spacing: 3pt, ..flags.map(flag-badge), ..forms-gap) + }, + )) + .flatten(), + ) +] + +// ───────────────────────────────────────────────────────────────────────────── +// The revise queue, with option tables +// ───────────────────────────────────────────────────────────────────────────── + +#let revise = cb-data.at("revise", default: ()) + +#if revise.len() > 0 [ + #pagebreak(weak: true) + = The evidence, question by question + + #explain[ + The full breakdown for every flagged question: what the flags mean, and who + chose what. The r column beside each option is the correlation between + choosing that option and scoring well on the rest of the exam, which is how a + defensible distractor announces itself. + ] + + #for q in revise [ + #block(breakable: false, above: entry-gap, width: 100%)[ + #{ + let parts = ( + { + let bits = (text(weight: "bold", fill: accent)[Question #q.number],) + let item = q.at("item", default: none) + if item != none { bits.push(text(size: size-meta, fill: luma(120))[#item]) } + let flags = q.at("flags", default: ()) + if flags.len() > 0 { + bits.push(stack(dir: ltr, spacing: 3pt, ..flags.map(flag-badge))) + } + bits.join(h(0.5em)) + }, + ) + + let context_ = question-context(q) + if context_ != none { parts.push(context_) } + + for note in q.at("notes", default: ()) { + parts.push(text(size: size-small)[— #markup(note)]) + } + + let predictions = q.at("prediction-notes", default: ()) + if predictions.len() > 0 { + parts.push(block( + width: 100%, + fill: luma(249), + radius: 3pt, + inset: (x: 7pt, y: 6pt), + )[ + #stack( + dir: ttb, + spacing: step * 0.4, + text(size: size-micro, weight: "bold", fill: luma(95), tracking: 0.04em)[ + #if q.at("calibrated", default: false) [AGAINST ITS CALIBRATION] else [AGAINST YOUR PREDICTION] + ], + ..predictions.map(note => text(size: size-meta)[#markup(note)]), + ) + ]) + } + + if show-options and q.at("options", default: ()).len() > 0 { + parts.push(table( + columns: (auto, auto, auto, 2.6cm, auto), + stroke: none, + align: (center + horizon, right + horizon, right + horizon, left + horizon, right + horizon), + inset: (x: 5pt, y: 3.5pt), + table.header(th[OPT], th[n], th[SHARE], th[], th[r]), + ..q + .options + .map(option => ( + { + if option.at("is-key", default: false) { + text(weight: "bold", fill: ok-color)[#option.letter] + } else { [#option.letter] } + }, + text(size: size-small)[#option.count], + text(size: size-small)[#pct(option.rate)], + bar( + option.rate, + color: if option.at("is-key", default: false) { ok-color } else { luma(160) }, + width: 2.2cm, + ), + { + let r = option.at("point-biserial", default: none) + if r == none { + text(fill: thin-color)[—] + } else if not option.at("is-key", default: false) and r > 0.0 { + text(size: size-small, fill: mid-color)[#signed(r)] + } else { + text(size: size-small)[#signed(r)] + } + }, + )) + .flatten(), + )) + } + + stack(dir: ttb, spacing: step * 0.8, ..parts) + } + ] + ] +] + +// ───────────────────────────────────────────────────────────────────────────── +// How the predictions did +// ───────────────────────────────────────────────────────────────────────────── + +#let predictions = cb-data.at("predictions", default: (:)) + +#if show-predictions and predictions.at("predicted", default: 0) > 0 [ + = How your predictions did + + #let n = predictions.at("predicted", default: 0) + #let calibrated = predictions.at("calibrated", default: 0) + #let signed-error = predictions.at("mean-signed-error", default: none) + #let abs-error = predictions.at("mean-abs-error", default: none) + + #explain[ + #if calibrated == 0 [ + None of these #n #plural(n, "expectation", "expectations") rests on prior + data, so they are predictions rather than calibrations. A prediction that + misses is a fact about the prediction: it does not flag the item, and it is + summarised here instead of appearing #n times in the tables above. Once + `coursebank calibrate` has written statistics back into the bank, a + subsequent miss means the cohort or the teaching moved, and it will be + flagged. + ] else [ + #calibrated of #n #plural(n, "expectation", "expectations") rests on a + prior calibration. Those are the ones whose misses are flagged on the item, + because a calibrated item that moves is telling you about this cohort. The + rest are predictions, and a miss corrects the prediction. + ] + ] + + #grid( + columns: (1fr, 1fr, 1fr), + gutter: 9pt, + stat-card( + "difficulty bias", + if signed-error != none { + (if signed-error >= 0 { "+" } else { "" }) + str(calc.round(signed-error * 100)) + " pts" + } else { "n/a" }, + note: if signed-error != none and signed-error > 0 { + "items came out easier than you expected" + } else if signed-error != none { + "items came out harder than you expected" + } else { none }, + ), + stat-card( + "typical miss", + if abs-error != none { str(calc.round(abs-error * 100)) + " pts" } else { "n/a" }, + note: str(predictions.at("within", default: 0)) + " of " + str(n) + " inside tolerance", + ), + stat-card( + "discrimination band", + str(predictions.at("band-hit", default: 0)) + " / " + str(predictions.at("band", default: 0)), + note: "landed in the band you expected", + ), + ) + + #{ + let surprise = predictions.at("biggest-surprise", default: none) + if surprise != none { + block(above: entry-gap)[ + #pad(right: prose-pad)[ + #text(size: size-lead, fill: luma(95))[ + The largest single gap was question #surprise.number, predicted at + #pct(surprise.expected) and observed at #pct(surprise.observed). + ] + ] + ] + } + } +] + +// ───────────────────────────────────────────────────────────────────────────── +// Profiles, blueprint, cautions // ───────────────────────────────────────────────────────────────────────────── #let blueprint = cb-data.at("blueprint", default: ()) @@ -526,8 +1237,17 @@ #if patterns.len() > 0 [ = Response profiles + #explain[ + Students grouped by the shape of their answers rather than by their totals. + Two students on the same percentage can need opposite things. + ] + #for pattern in patterns [ - - *#pattern.label* — #pattern.students student(s) + #block(above: step)[ + *#pattern.label* #text(size: size-meta, fill: luma(110))[ + #pattern.students #plural(pattern.students, "student", "students") + ] + ] ] ] @@ -535,7 +1255,7 @@ = Where the form differs from its blueprint #for line in blueprint [ - - #line + #block(above: step)[#text(size: size-small)[· #line]] ] ] @@ -543,16 +1263,20 @@ = Cautions #for line in warnings [ - - #text(size: 0.9em)[#line] + #block(above: step)[#pad(right: prose-pad)[#text(size: size-small)[· #line]]] ] ] #v(1.2em) #line(length: 100%, stroke: 0.5pt + luma(210)) #v(0.4em) -#text(size: 0.76em, fill: luma(120))[ - Generated #cb-meta.generator.on by #cb-meta.generator.tool #cb-meta.generator.version. - With #cb-data.students students, an item statistic has a standard error of - roughly 0.2; treat single-item results as provisional and the pattern across - items as real. Instructor copy — the option tables identify the key. +#pad(right: prose-pad)[ + #text(size: size-micro, fill: luma(120))[ + Generated #cb-meta.generator.on by #cb-meta.generator.tool #cb-meta.generator.version. + With #cb-data.students students, a point-biserial carries a standard error + near #num(1.0 / calc.sqrt(calc.max(cb-data.students - 2, 1)), digits: 2), so + treat a single item's r as provisional and the pattern across items as real. + Pool several administrations before retiring an item. Instructor copy: the + option tables identify the key. + ] ] diff --git a/src/model/course.rs b/src/model/course.rs index 2dc2ef2..5e9dec4 100644 --- a/src/model/course.rs +++ b/src/model/course.rs @@ -137,6 +137,58 @@ pub struct Policy { /// The fewest items on an objective before a report will call it mastered. #[serde(default = "two_usize")] pub min_items_for_mastery: usize, + /// The letter-grade bands, highest first or in any order. + /// + /// Empty by default, because a grading scale belongs to a course rather than + /// to a tool. When it is set, a class report bins the score distribution by + /// letter instead of by ten-point interval, which is the only binning a + /// student or an instructor actually acts on. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub grade_scale: Vec, +} + +/// One letter-grade band. +/// +/// Only the lower bound is recorded. An upper bound would be a second copy of +/// the next band's lower bound, and the two would eventually disagree: a scale +/// written as `93.0 - 96.9` leaves 96.95 in no band at all. Bands are read as +/// "this letter or better from here up", so the top band needs no ceiling. +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct GradeBand { + /// The letter as it appears on a transcript. + pub letter: String, + /// The lowest percentage that earns it, inclusive. + pub min: f64, + /// The grade points it carries, when the course records them. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub gpa: Option, + /// The attainment word attached to the band, such as `Meritorious`. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub attainment: Option, + /// A colour group, so a report can tint A bands alike without parsing + /// letters. Defaults to the letter's first character. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub group: Option, +} + +impl GradeBand { + /// The group a band belongs to: its own `group`, else its first character. + /// + /// # Returns + /// + /// An uppercase group key such as `A`. + pub fn group_key(&self) -> String { + match &self.group { + Some(group) => group.to_ascii_uppercase(), + None => self + .letter + .chars() + .next() + .map(|c| c.to_ascii_uppercase().to_string()) + .unwrap_or_default(), + } + } } impl Default for Policy { @@ -149,10 +201,44 @@ impl Default for Policy { partial_credit_floor_level: None, mastery_threshold: mastery_default(), min_items_for_mastery: 2, + grade_scale: Vec::new(), } } } +impl Policy { + /// The grade bands, highest lower bound first. + /// + /// # Returns + /// + /// The bands in descending order, empty when the course sets no scale. + pub fn bands(&self) -> Vec<&GradeBand> { + let mut out: Vec<&GradeBand> = self.grade_scale.iter().collect(); + out.sort_by(|a, b| { + b.min + .partial_cmp(&a.min) + .unwrap_or(std::cmp::Ordering::Equal) + }); + out + } + + /// The band a percentage falls in. + /// + /// # Arguments + /// + /// * `percent` - a score out of 100. + /// + /// # Returns + /// + /// The band, or `None` when the course sets no scale or the score sits below + /// every band in it. + pub fn band_for(&self, percent: f64) -> Option<&GradeBand> { + self.bands() + .into_iter() + .find(|band| percent + 1e-9 >= band.min) + } +} + /// A unit or module of the course. #[derive(Debug, Clone, Serialize, Deserialize)] #[serde(deny_unknown_fields)] @@ -622,6 +708,49 @@ impl CourseFile { )); } + // A scale with a hole in it silently drops students into no band at all, + // and the report would show a distribution that does not sum to the + // class. Cheaper to say so here. + let mut seen_letters: BTreeMap<&str, usize> = BTreeMap::new(); + let mut seen_mins: Vec = Vec::new(); + for band in &self.policy.grade_scale { + *seen_letters.entry(band.letter.as_str()).or_insert(0) += 1; + if !(0.0..=100.0).contains(&band.min) { + issues.push(format!( + "policy.grade_scale: band `{}` has min {}, which is not a percentage", + band.letter, band.min + )); + } + if seen_mins.iter().any(|m| (m - band.min).abs() < 1e-9) { + issues.push(format!( + "policy.grade_scale: two bands start at {}%, so the lower one is unreachable", + band.min + )); + } + seen_mins.push(band.min); + } + for (letter, n) in &seen_letters { + if *n > 1 { + issues.push(format!( + "policy.grade_scale: duplicate letter `{letter}` declared {n} times" + )); + } + } + if !self.policy.grade_scale.is_empty() { + let lowest = self + .policy + .bands() + .last() + .map(|b| b.min) + .unwrap_or(f64::INFINITY); + if lowest > 0.0 { + issues.push(format!( + "policy.grade_scale: the lowest band starts at {lowest}%, so a score below \ + that falls in no band. Give the failing grade a min of 0." + )); + } + } + let unit_ids: Vec<&String> = self.units.iter().map(|u| &u.id).collect(); let mut unit_counts: BTreeMap<&str, usize> = BTreeMap::new(); for u in &self.units {