diff --git a/src/analysis.rs b/src/analysis.rs index 2d24d25..baa66ba 100644 --- a/src/analysis.rs +++ b/src/analysis.rs @@ -35,5 +35,6 @@ pub mod calibrate; pub mod classical; +pub mod diagnostic; pub mod irt; pub mod students; diff --git a/src/analysis/classical.rs b/src/analysis/classical.rs index a46f135..846ff25 100644 --- a/src/analysis/classical.rs +++ b/src/analysis/classical.rs @@ -406,13 +406,13 @@ pub fn analyze( let mut credits: BTreeMap> = BTreeMap::new(); let mut blank = 0usize; for r in &rows { - if r.selected.is_empty() { + if r.chosen().is_empty() { blank += 1; continue; } // A multiple-response item is credited to the joined set, so that // "chose A and C" is one response pattern rather than two options. - let label = r.selected.join("+"); + let label = r.chosen().join("+"); if let Some(&si) = student_index.get(r.student_key.as_str()) { chose.entry(label.clone()).or_default().push(si); } @@ -775,7 +775,7 @@ fn infer_key(rows: &[&crate::responses::Response]) -> Vec { let mut out: BTreeSet = BTreeSet::new(); for r in rows { if r.credit >= 0.999 { - for letter in &r.selected { + for letter in r.chosen() { out.insert(letter.clone()); } } @@ -857,6 +857,7 @@ mod tests { assessment_id: "a".into(), date: None, form: None, + form_position: None, student_key: student.into(), sid: None, name: None, @@ -870,7 +871,9 @@ mod tests { } else { vec![letter.to_string()] }, + selected_source: vec![], eliminated: vec![], + eliminated_source: vec![], correct: Some(credit >= 0.999), credit, points_possible: 1.0, diff --git a/src/analysis/diagnostic.rs b/src/analysis/diagnostic.rs new file mode 100644 index 0000000..25c6361 --- /dev/null +++ b/src/analysis/diagnostic.rs @@ -0,0 +1,1163 @@ +// SPDX-License-Identifier: Prosperity-3.0.0 +// Copyright Scientific Computing Studio +// Source: https://git.scient.ing/education/coursebank + +//! What to tell a student, and what to tell yourself. +//! +//! [`crate::students`] computes mastery. [`crate::classical`] computes item +//! statistics. Neither decides what belongs in a document, and that decision is +//! not a formatting concern: it determines what a student is able to reconstruct +//! from the page in front of them. +//! +//! So this module assembles two view models, and the interesting property of the +//! first one is what it does not contain. +//! +//! # The withheld-question invariant +//! +//! [`StudentDiagnostic`] has no field for a stem and no field for option text. +//! Not an empty one, not one gated behind a flag: the struct has no such field, so +//! no template can print what it was never given. This is the same argument +//! [`crate::typst::config::Reveal`] makes about the exam paper, for the same +//! reason — a flag is one forgotten `if` away from a bad afternoon, and a +//! diagnostic that carries the questions cannot be handed back before the makeup +//! exam is given. +//! +//! What a student does get, per missed question: the number, its level, the +//! objectives it measured, whether they answered it, and the feedback written for +//! the specific option they chose. That last part is why authoring distractors +//! carefully pays off twice. +//! +//! Two consequences of that invariant are worth stating because they are easy to +//! undo by accident: +//! +//! *Only student-facing feedback is used.* [`crate::item::Choice::student_text`] +//! falls back to `explanation`, which is instructor-facing and routinely says +//! which option is right. This module reads `feedback_student` and `misconception` +//! and nothing else. +//! +//! *Feedback is looked up by bank letter, not printed letter.* On a shuffled form +//! those differ, and looking up the printed letter returns another option's +//! misconception — confident, specific, and about a question the student did not +//! answer that way. See [`crate::decode`]. +//! +//! # What to study +//! +//! A list of missed objectives is a diagnosis, not a prescription. The study plan +//! resolves each weak objective through the course's own reading registry, so a +//! student is pointed at `KKW §6.2` with the sentence you wrote about what to take +//! from it, rather than at the name of a chapter. + +use std::collections::{BTreeMap, BTreeSet}; + +use serde::Serialize; + +use crate::assessment::AssessmentFile; +use crate::catalog::Catalog; +use crate::classical::Analysis; +use crate::course::{CourseFile, ReadingRole}; +use crate::irt::Fit; +use crate::responses::{Response, ResponseSet}; +use crate::students::{Cohort, Mastery, StudentSummary}; +use crate::taxonomy::Level; + +/// What to assemble. +#[derive(Debug, Clone)] +pub struct Options { + /// Whether to compare the student to the class. + pub comparison: bool, + /// Whether to include the per-question map. + /// + /// The map names question numbers, never their content. It is what lets a + /// student who has their paper back line the two up. + pub questions: bool, + /// Whether to include the feedback written for the option the student chose. + pub feedback: bool, + /// How many objectives to build a study plan for. + pub focus_limit: usize, + /// How many readings to list per objective. + pub readings_per_objective: usize, + /// Whether to include the IRT ability estimate. + pub ability: bool, +} + +impl Default for Options { + fn default() -> Options { + Options { + comparison: true, + questions: true, + feedback: true, + focus_limit: 4, + readings_per_objective: 2, + ability: false, + } + } +} + +/// Everything one student's diagnostic says. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct StudentDiagnostic { + /// The grouping key, which is a pseudonym when the store is pseudonymized. + pub student_key: String, + /// The display name, when identifiers were kept. + #[serde(skip_serializing_if = "Option::is_none")] + pub name: Option, + /// The institutional id, when identifiers were kept. + #[serde(skip_serializing_if = "Option::is_none")] + pub sid: Option, + /// Which form they sat. + #[serde(skip_serializing_if = "Option::is_none")] + pub form: Option, + /// Their score. + pub score: Score, + /// How they compare to the class, when comparison is on. + #[serde(skip_serializing_if = "Option::is_none")] + pub standing: Option, + /// Per-level performance. + pub levels: Vec, + /// Per-objective standing, in the course's own order. + pub objectives: Vec, + /// Objectives they are clearly meeting, worst first among the confident ones. + pub strengths: Vec, + /// Objectives to work on, worst first. + pub focus: Vec, + /// One row per question, with no question in it. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub questions: Vec, + /// What to read, grouped by objective. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub study: Vec, +} + +/// A score, with the denominators spelled out. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct Score { + /// Points earned on scored items. + pub points: f64, + /// Points available on scored items. + pub points_possible: f64, + /// Percentage on scored items. + pub percent: f64, + /// Bonus points earned. + pub bonus_points: f64, + /// Items answered correctly. + pub correct: usize, + /// Scored items administered. + pub n_items: usize, +} + +/// Where a score sits in the class. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct Standing { + /// The class mean percentage. + pub class_mean: f64, + /// The standard deviation of class percentages. + pub class_sd: f64, + /// A coarse band, never a rank. + pub band: String, + /// The IRT ability estimate, when asked for. + #[serde(skip_serializing_if = "Option::is_none")] + pub theta: Option, + /// Its standard error. + #[serde(skip_serializing_if = "Option::is_none")] + pub theta_se: Option, +} + +/// One cognitive level. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct LevelRow { + /// The level code, 1 through 5. + pub level: u8, + /// The level name. + pub name: String, + /// What that level asks of a student, in one phrase. + pub blurb: String, + /// How many items at this level. + pub n_items: usize, + /// The student's rate. + pub rate: f64, + /// The class rate, when comparison is on. + #[serde(skip_serializing_if = "Option::is_none")] + pub class_rate: Option, + /// A plain-language comparison, when comparison is on. + #[serde(skip_serializing_if = "Option::is_none")] + pub comparison: Option, +} + +/// One objective's standing. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct ObjectiveRow { + /// The objective id. + pub id: String, + /// The objective text. + pub text: String, + /// The unit it belongs to, when the course declares one. + #[serde(skip_serializing_if = "Option::is_none")] + pub unit: Option, + /// How many items measured it. + pub n_items: usize, + /// Credit earned across them. + pub credit: f64, + /// The observed rate. + pub rate: f64, + /// The lower bound of the 95% Wilson interval. + pub lower: f64, + /// The upper bound. + pub upper: f64, + /// The class rate, when comparison is on. + #[serde(skip_serializing_if = "Option::is_none")] + pub class_rate: Option, + /// The mastery classification: `meeting`, `developing`, `not yet`, or + /// `not enough evidence`. + pub status: String, + /// A compact symbol for the same thing. + pub symbol: String, + /// Whether the interval, not just the estimate, clears the threshold. + pub confident: bool, + /// Whether too few items measured it to classify at all. This is a fact about + /// the exam, and a report that says so is being honest rather than vague. + pub thin_evidence: bool, + /// The levels it was assessed at, as codes. + pub levels: Vec, +} + +/// An objective named in a list. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct ObjectiveRef { + /// The objective id. + pub id: String, + /// The objective text. + pub text: String, + /// The observed rate. + pub rate: f64, + /// How many items measured it. + pub n_items: usize, +} + +/// One question, described without being reproduced. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct QuestionRow { + /// The recorded question number, which is what is printed on the paper. + pub number: u32, + /// Where it sat on this student's form, when the forms differ. + #[serde(skip_serializing_if = "Option::is_none")] + pub position: Option, + /// The level code. + #[serde(skip_serializing_if = "Option::is_none")] + pub level: Option, + /// The objectives it measured. + pub objectives: Vec, + /// Whether it was answered correctly. + #[serde(skip_serializing_if = "Option::is_none")] + pub correct: Option, + /// Credit earned, as a fraction. + pub credit: f64, + /// Whether it was a bonus question. + #[serde(skip_serializing_if = "std::ops::Not::not")] + pub bonus: bool, + /// Whether the student left it blank. + pub blank: bool, + /// The share of the class that answered it correctly, when comparison is on. + #[serde(skip_serializing_if = "Option::is_none")] + pub class_rate: Option, + /// The feedback written for the option this student chose. + #[serde(skip_serializing_if = "Option::is_none")] + pub feedback: Option, + /// Where the material was taught. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub taught_in: Vec, +} + +/// What to read about one objective. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct StudyGroup { + /// The objective id. + pub objective: String, + /// The objective text. + pub text: String, + /// The observed rate, so the list is ordered by need. + pub rate: f64, + /// The readings. + pub readings: Vec, +} + +/// One reading to revisit. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct StudyReading { + /// A short citation, e.g. `KKW §6.2`. + pub citation: String, + /// The lecture it was assigned for. + pub lecture: String, + /// That lecture's title. + pub lecture_title: String, + /// A link, when the reference resolves to one. + #[serde(skip_serializing_if = "Option::is_none")] + pub url: Option, + /// What to take from it, which is the sentence worth quoting at someone who + /// missed the objective. + #[serde(skip_serializing_if = "Option::is_none")] + pub focus: Option, + /// What the section covers. + #[serde(skip_serializing_if = "Option::is_none")] + pub summary: Option, + /// Whether it was assigned or offered alongside. + pub supplemental: bool, +} + +/// Builds one student's diagnostic. +/// +/// # Arguments +/// +/// * `summary` - the student's computed summary. +/// * `cohort` - the class context. +/// * `catalog` - the loaded course, for objective text, feedback, and readings. +/// * `set` - the responses, for this student's per-question rows. +/// * `analysis` - the item analysis, for class rates per question. Optional. +/// * `opts` - what to include. +/// +/// # Returns +/// +/// The diagnostic. +pub fn student( + summary: &StudentSummary, + cohort: &Cohort, + catalog: &Catalog, + set: &ResponseSet, + analysis: Option<&Analysis>, + opts: &Options, +) -> StudentDiagnostic { + let course = &catalog.course; + let rows = set.for_student(&summary.student_key); + let form = rows.first().and_then(|r| r.form.clone()); + + let levels = summary + .levels + .iter() + .map(|profile| LevelRow { + level: profile.level.code(), + name: profile.level.name().to_string(), + blurb: profile.level.blurb().to_string(), + n_items: profile.n_items, + rate: profile.rate, + class_rate: opts.comparison.then_some(profile.cohort_rate), + comparison: opts.comparison.then(|| profile.comparison().to_string()), + }) + .collect(); + + let objectives: Vec = summary + .objectives + .iter() + .map(|mastery| ObjectiveRow { + id: mastery.objective.clone(), + text: mastery.text.clone(), + unit: course + .learning_objectives + .get(&mastery.objective) + .and_then(|o| o.unit.clone()), + n_items: mastery.n_items, + credit: mastery.credit, + rate: mastery.rate, + lower: mastery.wilson_lower, + upper: mastery.wilson_upper, + class_rate: opts.comparison.then_some(mastery.cohort_rate), + status: mastery.status.label().to_string(), + symbol: mastery.status.symbol().to_string(), + confident: mastery.confident, + thin_evidence: mastery.status == Mastery::NotEnoughEvidence, + levels: mastery.levels.iter().map(|l| l.code()).collect(), + }) + .collect(); + + let refs = |ids: &[String]| -> Vec { + ids.iter() + .filter_map(|id| objectives.iter().find(|o| &o.id == id)) + .map(|o| ObjectiveRef { + id: o.id.clone(), + text: o.text.clone(), + rate: o.rate, + n_items: o.n_items, + }) + .collect() + }; + + let strengths = refs(&summary.strengths); + let focus = refs(&summary.focus); + + let class_rates: BTreeMap = analysis + .map(|a| a.items.iter().map(|i| (i.number, i.p_value)).collect()) + .unwrap_or_default(); + + let questions = if opts.questions { + rows.iter() + .map(|row| question_row(row, catalog, &class_rates, opts)) + .collect() + } else { + Vec::new() + }; + + let study = focus + .iter() + .take(opts.focus_limit) + .map(|objective| StudyGroup { + objective: objective.id.clone(), + text: objective.text.clone(), + rate: objective.rate, + readings: readings_for(course, &objective.id, opts.readings_per_objective), + }) + .filter(|group| !group.readings.is_empty()) + .collect(); + + StudentDiagnostic { + student_key: summary.student_key.clone(), + name: summary.name.clone(), + sid: summary.sid.clone(), + form, + score: Score { + points: summary.points, + points_possible: summary.points_possible, + percent: summary.percent, + bonus_points: summary.bonus_points, + correct: summary.correct, + n_items: summary.n_items, + }, + standing: opts.comparison.then(|| Standing { + class_mean: cohort.mean_percent, + class_sd: cohort.sd_percent, + band: summary.band.clone(), + theta: opts.ability.then_some(summary.theta).flatten(), + theta_se: opts.ability.then_some(summary.theta_se).flatten(), + }), + levels, + objectives, + strengths, + focus, + questions, + study, + } +} + +/// Builds one question's row. +/// +/// The feedback lookup uses [`Response::chosen`], which prefers the bank letters +/// written at ingest. Falling back to the printed letters is right for an +/// unshuffled form and wrong for a shuffled one, which is why ingest translates +/// rather than leaving it to here. +fn question_row( + row: &Response, + catalog: &Catalog, + class_rates: &BTreeMap, + opts: &Options, +) -> QuestionRow { + let blank = row.selected.is_empty() && row.eliminated.is_empty(); + let mut feedback = None; + let mut taught_in = Vec::new(); + + if let Some(uid) = row.item_ref.as_deref() { + if let Some(entry) = catalog.get(uid) { + if opts.feedback && row.credit < 0.999 { + if let Some(letter) = row.chosen().first() { + // `student_text` falls back to `explanation`, which is written + // for a grader and often names the right answer. A student + // report must not print it. + feedback = entry.item.option(letter).and_then(|choice| { + choice + .feedback_student + .clone() + .or_else(|| choice.misconception.clone()) + }); + } + } + for source in &entry.item.sources { + let title = catalog + .course + .lectures + .get(&source.lecture) + .map(|l| l.title.clone()) + .unwrap_or_else(|| source.lecture.clone()); + if source.slides.is_empty() { + taught_in.push(format!("{} ({})", title, source.lecture)); + } else { + let slides: Vec = source.slides.iter().map(|s| s.to_string()).collect(); + taught_in.push(format!( + "{} ({}), slide{} {}", + title, + source.lecture, + if source.slides.len() == 1 { "" } else { "s" }, + slides.join(", ") + )); + } + } + } + } + + QuestionRow { + number: row.item_number, + position: row.form_position.filter(|p| *p != row.item_number), + level: row.level.map(|l| l.code()), + objectives: row.learning_objectives.clone(), + correct: row.correct, + credit: row.credit, + bonus: row.bonus, + blank, + class_rate: opts + .comparison + .then(|| class_rates.get(&row.item_number).copied()) + .flatten(), + feedback, + taught_in, + } +} + +/// Resolves an objective to readings. +/// +/// # Arguments +/// +/// * `course` - the course registry. +/// * `objective` - the objective id. +/// * `limit` - how many readings to keep. +/// +/// # Returns +/// +/// The readings, assigned ones first. +fn readings_for(course: &CourseFile, objective: &str, limit: usize) -> Vec { + let mut out = Vec::new(); + for (lecture_id, reading) in course.readings_for_objective(objective) { + let lecture_title = course + .lectures + .get(lecture_id) + .map(|l| l.title.clone()) + .unwrap_or_else(|| lecture_id.to_string()); + + let (citation, url) = match reading.reference.as_deref() { + Some(key) => match course.references.get(key) { + Some(reference) => (reading.cite(key, reference), reading.resolve_url(reference)), + None => ( + reading + .text + .clone() + .or_else(|| reading.locator.clone()) + .unwrap_or_else(|| key.to_string()), + reading.url.clone(), + ), + }, + None => ( + reading + .text + .clone() + .or_else(|| reading.locator.clone()) + .unwrap_or_else(|| lecture_title.clone()), + reading.url.clone(), + ), + }; + + out.push(StudyReading { + citation, + lecture: lecture_id.to_string(), + lecture_title, + url, + focus: reading.focus.clone(), + summary: reading.summary.clone(), + supplemental: reading.role == ReadingRole::Supplemental, + }); + } + + // Assigned before supplemental, otherwise the order the course declares. + out.sort_by_key(|r| r.supplemental); + out.truncate(limit); + out +} + +/// Everything the class diagnostic says. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct CohortDiagnostic { + /// How many students sat it. + pub n_students: usize, + /// How many scored items. + pub n_items: usize, + /// The score distribution. + pub distribution: Distribution, + /// Whole-test reliability. + pub reliability: ReliabilityRow, + /// Per-level class performance. + pub levels: Vec, + /// Per-objective class performance, worst first. + pub objectives: Vec, + /// Objectives the class as a whole did not meet. + pub gaps: Vec, + /// Per-question statistics. + pub questions: Vec, + /// Questions worth revisiting before reuse, worst first. + pub revise: Vec, + /// One row per form, when more than one was given. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub forms: Vec, + /// Where the form built differs from the blueprint it was drawn against. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub blueprint: Vec, + /// Response profiles the class falls into. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub patterns: Vec, + /// Cautions about the analysis itself. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub warnings: Vec, +} + +/// The score distribution, binned. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct Distribution { + /// Mean percentage. + pub mean: f64, + /// Median percentage. + pub median: f64, + /// Standard deviation. + pub sd: f64, + /// Lowest percentage. + pub min: f64, + /// Highest percentage. + pub max: f64, + /// Counts in ten-point bins, from 0-9 through 90-100. + pub bins: Vec, +} + +/// One histogram bin. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct Bin { + /// Inclusive lower bound, in percent. + pub low: u32, + /// Exclusive upper bound, in percent, except the last bin which includes 100. + pub high: u32, + /// How many students fell in it. + pub count: usize, +} + +/// Reliability, flattened for a template. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct ReliabilityRow { + /// KR-20, when it could be computed. + #[serde(skip_serializing_if = "Option::is_none")] + pub alpha: Option, + /// The standard error of measurement, in items. + #[serde(skip_serializing_if = "Option::is_none")] + pub sem: Option, + /// Mean p-value across items. + pub mean_p: f64, + /// Mean point-biserial across items that had one. + #[serde(skip_serializing_if = "Option::is_none")] + pub mean_point_biserial: Option, + /// What the alpha value means for a test this length in a class this size. + pub interpretation: String, +} + +/// One level, class-wide. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct CohortLevelRow { + /// The level code. + pub level: u8, + /// The level name. + pub name: String, + /// How many items sat at this level. + pub n_items: usize, + /// The class rate. + pub rate: f64, +} + +/// One objective, class-wide. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct CohortObjectiveRow { + /// The objective id. + pub id: String, + /// The objective text. + pub text: String, + /// How many items measured it. + pub n_items: usize, + /// The class rate. + pub rate: f64, + /// How many students met it. + pub meeting: usize, + /// How many students are developing on it. + pub developing: usize, + /// How many students are not yet meeting it. + pub not_yet: usize, + /// How many had too few items to classify. + pub thin: usize, + /// Whether the class rate is below the course's mastery threshold. + pub below_threshold: bool, +} + +/// One question, class-wide. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct CohortQuestionRow { + /// The recorded question number. + pub number: u32, + /// The item's global id. + #[serde(skip_serializing_if = "Option::is_none")] + pub item: Option, + /// The level code. + #[serde(skip_serializing_if = "Option::is_none")] + pub level: Option, + /// The objectives it measured. + pub objectives: Vec, + /// Proportion correct. + pub p_value: f64, + /// Corrected item-total point-biserial. + #[serde(skip_serializing_if = "Option::is_none")] + pub point_biserial: Option, + /// Upper minus lower group proportion correct. + #[serde(skip_serializing_if = "Option::is_none")] + pub discrimination: Option, + /// Fraction who left it blank. + pub blank_rate: f64, + /// The keyed letters. + pub key: Vec, + /// Per-option selection, in letter order. + pub options: Vec, + /// Machine-detected problems. + pub flags: Vec, + /// What those flags mean. + pub notes: Vec, + /// Per-form proportion correct, when more than one form was given. A gap here + /// on one question, with the rest of the exam in step, points at that + /// question's permutation rather than at the cohort. + #[serde(skip_serializing_if = "BTreeMap::is_empty")] + pub by_form: BTreeMap, +} + +/// One option's selection statistics. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct OptionRow { + /// The bank letter. + pub letter: String, + /// How many chose it. + pub count: usize, + /// The share who chose it. + pub rate: f64, + /// Whether it is keyed. + pub is_key: bool, + /// Correlation between choosing it and scoring well elsewhere. + #[serde(skip_serializing_if = "Option::is_none")] + pub point_biserial: Option, + /// Whether it drew nobody, and is therefore doing no work. + pub nonfunctioning: bool, +} + +/// One form's summary. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct FormRow { + /// The form id. + pub id: String, + /// How many students sat it. + pub n_students: usize, + /// Their mean percentage. + pub mean: f64, + /// The standard deviation of their percentages. + pub sd: f64, +} + +/// One response profile. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct PatternRow { + /// A label describing the pattern. + pub label: String, + /// How many students fit it. + pub n_students: usize, + /// Mean rate at each level, by level code. + pub level_means: BTreeMap, +} + +/// Builds the class diagnostic. +/// +/// # Arguments +/// +/// * `analysis` - the classical item analysis. +/// * `cohort` - the per-student summaries and class rates. +/// * `catalog` - the loaded course. +/// * `record` - the assessment record, for the blueprint check. +/// * `set` - the responses, for per-form and per-option breakdowns. +/// * `fit` - an IRT fit, when one was computed. +/// +/// # Returns +/// +/// The diagnostic. +pub fn cohort( + analysis: &Analysis, + cohort: &Cohort, + catalog: &Catalog, + record: &AssessmentFile, + set: &ResponseSet, + fit: Option<&Fit>, +) -> CohortDiagnostic { + let _ = fit; + let course = &catalog.course; + let threshold = course.policy.mastery_threshold; + + let percents: Vec = cohort.students.iter().map(|s| s.percent).collect(); + + let level_counts: BTreeMap = { + let mut counts: BTreeMap> = BTreeMap::new(); + for row in set.rows.iter().filter(|r| r.counts()) { + if let Some(level) = row.level { + counts.entry(level).or_default().insert(row.item_number); + } + } + counts.into_iter().map(|(k, v)| (k, v.len())).collect() + }; + + let levels = Level::ALL + .iter() + .filter_map(|level| { + let rate = cohort.level_rates.get(level).copied()?; + Some(CohortLevelRow { + level: level.code(), + name: level.name().to_string(), + n_items: level_counts.get(level).copied().unwrap_or(0), + rate, + }) + }) + .collect(); + + // Objective counts come from the responses so that an objective assessed by + // two items is not reported as if it had one. + let mut objective_items: BTreeMap> = BTreeMap::new(); + for row in set.rows.iter().filter(|r| r.counts()) { + for objective in &row.learning_objectives { + objective_items + .entry(objective.clone()) + .or_default() + .insert(row.item_number); + } + } + + let mut objectives: Vec = cohort + .objective_rates + .iter() + .map(|(id, rate)| { + let mut meeting = 0; + let mut developing = 0; + let mut not_yet = 0; + let mut thin = 0; + for student in &cohort.students { + if let Some(row) = student.objectives.iter().find(|o| &o.objective == id) { + match row.status { + Mastery::Meeting => meeting += 1, + Mastery::Developing => developing += 1, + Mastery::NotYet => not_yet += 1, + Mastery::NotEnoughEvidence => thin += 1, + } + } + } + CohortObjectiveRow { + id: id.clone(), + text: course.objective_text(id), + n_items: objective_items.get(id).map(|s| s.len()).unwrap_or(0), + rate: *rate, + meeting, + developing, + not_yet, + thin, + below_threshold: *rate < threshold, + } + }) + .collect(); + objectives.sort_by(|a, b| { + a.rate + .partial_cmp(&b.rate) + .unwrap_or(std::cmp::Ordering::Equal) + .then_with(|| a.id.cmp(&b.id)) + }); + let gaps: Vec = objectives + .iter() + .filter(|o| o.below_threshold) + .cloned() + .collect(); + + let by_form = per_form_p_values(set); + let item_meta: BTreeMap, Vec)> = record + .items + .iter() + .map(|p| { + ( + p.number, + (p.level.map(|l| l.code()), p.learning_objectives.clone()), + ) + }) + .collect(); + + let questions: Vec = analysis + .items + .iter() + .map(|item| { + let meta = item_meta.get(&item.number); + CohortQuestionRow { + number: item.number, + item: item.item_ref.clone(), + level: meta.and_then(|m| m.0), + objectives: meta.map(|m| m.1.clone()).unwrap_or_default(), + p_value: item.p_value, + point_biserial: item.point_biserial, + discrimination: item.discrimination_index, + blank_rate: item.blank_rate, + key: item.key.clone(), + options: item + .options + .values() + .map(|option| OptionRow { + letter: option.letter.clone(), + count: option.count, + rate: option.rate, + is_key: option.is_key, + point_biserial: option.point_biserial, + nonfunctioning: !option.is_key && option.rate <= 0.05, + }) + .collect(), + flags: item.flags.iter().map(|f| f.as_str().to_string()).collect(), + notes: item.notes.clone(), + by_form: by_form.get(&item.number).cloned().unwrap_or_default(), + } + }) + .collect(); + + let revise: Vec = analysis + .revise_queue() + .iter() + .filter_map(|item| questions.iter().find(|q| q.number == item.number).cloned()) + .collect(); + + CohortDiagnostic { + n_students: cohort.students.len(), + n_items: analysis.reliability.n_items, + distribution: distribution(&percents), + reliability: ReliabilityRow { + alpha: analysis.reliability.alpha, + sem: analysis.reliability.sem, + mean_p: analysis.reliability.mean_p, + mean_point_biserial: analysis.reliability.mean_point_biserial, + interpretation: analysis.reliability.interpretation(), + }, + levels, + objectives, + gaps, + questions, + revise, + forms: form_rows(set, cohort), + blueprint: crate::select::check_blueprint(record), + patterns: cohort + .archetypes + .iter() + .map(|a| PatternRow { + label: a.label.clone(), + n_students: a.members.len(), + level_means: a + .level_means + .iter() + .map(|(level, mean)| (level.code(), *mean)) + .collect(), + }) + .collect(), + warnings: analysis.warnings.clone(), + } +} + +/// Per-question proportion correct, split by form. +fn per_form_p_values(set: &ResponseSet) -> BTreeMap> { + let forms: BTreeSet<&str> = set.rows.iter().filter_map(|r| r.form.as_deref()).collect(); + if forms.len() < 2 { + return BTreeMap::new(); + } + + let mut totals: BTreeMap<(u32, String), (usize, usize)> = BTreeMap::new(); + for row in set.rows.iter().filter(|r| r.counts()) { + let Some(form) = row.form.as_deref() else { + continue; + }; + let entry = totals + .entry((row.item_number, form.to_string())) + .or_insert((0, 0)); + entry.1 += 1; + if row.correct == Some(true) { + entry.0 += 1; + } + } + + let mut out: BTreeMap> = BTreeMap::new(); + for ((number, form), (correct, n)) in totals { + if n == 0 { + continue; + } + out.entry(number) + .or_default() + .insert(form, correct as f64 / n as f64); + } + out +} + +/// One row per form, when more than one was given. +fn form_rows(set: &ResponseSet, cohort: &Cohort) -> Vec { + let mut students_by_form: BTreeMap> = BTreeMap::new(); + for row in &set.rows { + if let Some(form) = row.form.as_deref() { + students_by_form + .entry(form.to_string()) + .or_default() + .insert(row.student_key.as_str()); + } + } + if students_by_form.len() < 2 { + return Vec::new(); + } + + let percent: BTreeMap<&str, f64> = cohort + .students + .iter() + .map(|s| (s.student_key.as_str(), s.percent)) + .collect(); + + students_by_form + .into_iter() + .map(|(form, students)| { + let values: Vec = students + .iter() + .filter_map(|s| percent.get(*s).copied()) + .collect(); + let n = values.len(); + let mean = if n == 0 { + 0.0 + } else { + values.iter().sum::() / n as f64 + }; + let sd = if n < 2 { + 0.0 + } else { + let variance = + values.iter().map(|v| (v - mean).powi(2)).sum::() / (n as f64 - 1.0); + variance.sqrt() + }; + FormRow { + id: form, + n_students: n, + mean, + sd, + } + }) + .collect() +} + +/// Bins a set of percentages into a distribution. +fn distribution(percents: &[f64]) -> Distribution { + let mut sorted: Vec = percents.to_vec(); + sorted.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal)); + + let n = sorted.len(); + let mean = if n == 0 { + 0.0 + } else { + sorted.iter().sum::() / n as f64 + }; + let median = match n { + 0 => 0.0, + _ if n % 2 == 1 => sorted[n / 2], + _ => (sorted[n / 2 - 1] + sorted[n / 2]) / 2.0, + }; + let sd = if n < 2 { + 0.0 + } else { + (sorted.iter().map(|v| (v - mean).powi(2)).sum::() / (n as f64 - 1.0)).sqrt() + }; + + let mut bins: Vec = (0..10) + .map(|i| Bin { + low: i * 10, + high: if i == 9 { 100 } else { i * 10 + 10 }, + count: 0, + }) + .collect(); + for value in &sorted { + let index = ((*value / 10.0).floor() as isize).clamp(0, 9) as usize; + bins[index].count += 1; + } + + Distribution { + mean, + median, + sd, + min: sorted.first().copied().unwrap_or(0.0), + max: sorted.last().copied().unwrap_or(0.0), + bins, + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn the_distribution_bins_a_hundred_into_the_last_bin() { + let d = distribution(&[100.0, 95.0, 0.0, 42.0]); + assert_eq!(d.bins[9].count, 2, "100 belongs with the nineties"); + assert_eq!(d.bins[0].count, 1); + assert_eq!(d.bins[4].count, 1); + assert_eq!(d.max, 100.0); + assert_eq!(d.min, 0.0); + } + + #[test] + fn the_median_averages_the_middle_pair() { + assert_eq!(distribution(&[10.0, 20.0, 30.0, 40.0]).median, 25.0); + assert_eq!(distribution(&[10.0, 20.0, 30.0]).median, 20.0); + } + + #[test] + fn an_empty_class_does_not_panic() { + let d = distribution(&[]); + assert_eq!(d.mean, 0.0); + assert_eq!(d.bins.len(), 10); + } + + #[test] + fn the_student_diagnostic_has_no_field_for_question_content() { + // A compile-time argument as much as a test: the struct has no stem and no + // option text, so no template can print either one. If a field is ever + // added, this serialization check is where the reason gets re-read. + let json = serde_json::to_string(&StudentDiagnostic { + student_key: "s-1".into(), + name: None, + sid: None, + form: None, + score: Score { + points: 1.0, + points_possible: 2.0, + percent: 50.0, + bonus_points: 0.0, + correct: 1, + n_items: 2, + }, + standing: None, + levels: Vec::new(), + objectives: Vec::new(), + strengths: Vec::new(), + focus: Vec::new(), + questions: Vec::new(), + study: Vec::new(), + }) + .unwrap(); + assert!(!json.contains("stem"), "{json}"); + assert!(!json.contains("options"), "{json}"); + } +} diff --git a/src/analysis/students.rs b/src/analysis/students.rs index dd153e3..a36b874 100644 --- a/src/analysis/students.rs +++ b/src/analysis/students.rs @@ -563,7 +563,7 @@ fn missed_items( if let Some(entry) = cat.get(uid) { // Feedback for the specific option chosen, which is the whole // point of recording per-distractor misconceptions. - if let Some(letter) = r.selected.first() { + if let Some(letter) = r.chosen().first() { if let Some(choice) = entry.item.option(letter) { misconception = choice.misconception.clone(); feedback = choice.student_text().map(|s| s.to_string()); @@ -592,7 +592,7 @@ fn missed_items( out.push(MissedItem { number: r.item_number, item_ref: r.item_ref.clone(), - selected: r.selected.clone(), + selected: r.chosen().to_vec(), credit: r.credit, level: r.level, learning_objectives: r.learning_objectives.clone(), @@ -1089,6 +1089,7 @@ mod tests { assessment_id: "a".into(), date: None, form: None, + form_position: None, student_key: student.into(), sid: None, name: None, @@ -1098,7 +1099,9 @@ mod tests { item_ref: None, item_version: None, selected: vec!["A".into()], + selected_source: vec![], eliminated: vec![], + eliminated_source: vec![], correct: Some(credit >= 0.999), credit, points_possible: 1.0, diff --git a/src/cli.rs b/src/cli.rs index 452a171..3c5a32b 100644 --- a/src/cli.rs +++ b/src/cli.rs @@ -87,6 +87,8 @@ pub(crate) enum Command { /// Write reports. #[command(subcommand)] Report(ReportCommand), + /// Freeze what was administered, with digests, before printing. + Seal(SealArgs), /// List what is in the response store. Data, } @@ -551,12 +553,23 @@ impl FormatArg { #[derive(Debug, Subcommand)] pub(crate) enum IngestCommand { - /// Read a directory of Gradescope per-question CSV exports. + /// Read one directory of Gradescope per-question CSV exports per form. + /// + /// Write each source as `FORM=DIR`. A bare path takes `--form`. Gradescope { - /// The directory holding 1.csv .. N.csv. - dir: PathBuf, + /// The directories, e.g. A=exports/e1-a B=exports/e1-b. + #[arg(value_name = "SOURCE", required = true)] + sources: Vec, #[command(flatten)] common: IngestCommon, + /// Ingest even when a directory's graded keys do not match the form it + /// was labelled with. + #[arg(long)] + allow_mismatch: bool, + /// The export numbers questions by recorded number rather than by + /// printed position. Only for a platform that is not Gradescope. + #[arg(long)] + recorded_numbers: bool, }, /// Read a Canvas "Student Analysis" CSV. Canvas { @@ -648,6 +661,24 @@ pub(crate) enum ReportCommand { /// Leave out the comparison to the class. #[arg(long)] no_comparison: bool, + /// Also write a Typst diagnostic per student. + #[arg(long)] + typst: bool, + /// Skip the Markdown reports. + #[arg(long)] + no_markdown: bool, + /// Leave the per-question map out of the Typst report. + #[arg(long)] + no_questions: bool, + /// Leave per-option feedback out of the Typst report. + #[arg(long)] + no_feedback: bool, + /// Use this template instead of the usual lookup. + #[arg(long)] + template: Option, + /// Also write each payload as JSON. + #[arg(long)] + json: bool, }, /// The instructor's item analysis. Cohort { @@ -659,9 +690,42 @@ pub(crate) enum ReportCommand { /// Output path; defaults to reports/-cohort.md. #[arg(long)] out: Option, + /// Also write a Typst class diagnostic. + #[arg(long)] + typst: bool, + /// Skip the Markdown report. + #[arg(long)] + no_markdown: bool, + /// Use this template instead of the usual lookup. + #[arg(long)] + template: Option, + /// Also write the payload as JSON. + #[arg(long)] + json: bool, }, } +#[derive(Debug, Args)] +pub(crate) struct SealArgs { + /// Assessment id. + pub(crate) id: String, + /// Check the existing seal against the course instead of writing one. + #[arg(long)] + pub(crate) check: bool, + /// Overwrite an existing seal. + #[arg(long)] + pub(crate) force: bool, + /// Only these forms; defaults to every form the record declares. + #[arg(long, value_delimiter = ',')] + pub(crate) form: Vec, + /// Store only fingerprints, not the stem and option text. + #[arg(long)] + pub(crate) no_content: bool, + /// Output path; defaults to seals/.yaml. + #[arg(long)] + pub(crate) out: Option, +} + #[cfg(test)] mod tests { use super::*; diff --git a/src/commands.rs b/src/commands.rs index 8b98b7f..af2bad0 100644 --- a/src/commands.rs +++ b/src/commands.rs @@ -70,6 +70,7 @@ pub(crate) fn run(cli: &Cli) -> Result { Command::Analyze(sub) => analysis::analyze(cli, sub), Command::Calibrate(args) => analysis::calibrate(cli, args), Command::Report(sub) => analysis::report(cli, sub), + Command::Seal(args) => analysis::seal(cli, args), Command::Data => analysis::data(cli), } } diff --git a/src/commands/analysis.rs b/src/commands/analysis.rs index 6c3b7c7..141f15f 100644 --- a/src/commands/analysis.rs +++ b/src/commands/analysis.rs @@ -4,53 +4,240 @@ //! The responses-to-report pipeline. //! -//! Once an assessment has been given, responses come back through [`ingest`], -//! statistics come out of [`analyze`] (classical, IRT, or per-student), [`calibrate`] -//! writes those statistics back onto the items, and [`report`] produces the -//! student and cohort documents. [`data`] lists what the response store holds. +//! [`seal`] freezes what was administered before the papers are printed, which is +//! what lets everything downstream translate a student's marks back into the +//! bank's own lettering. Once an assessment has been given, responses come back +//! through [`ingest`], statistics come out of [`analyze`] (classical, IRT, or +//! per-student), [`calibrate`] writes those statistics back onto the items, and +//! [`report`] produces the student and cohort documents. [`data`] lists what the +//! response store holds. use coursebank::calibrate; use coursebank::canvas; +use coursebank::catalog::Severity; use coursebank::classical::{self, Thresholds}; -use coursebank::error::Result; -use coursebank::gradescope; +use coursebank::decode::Numbering; +use coursebank::diagnostic; +use coursebank::error::{Error, Result}; +use coursebank::intake::{self, Source}; use coursebank::irt; use coursebank::layout::Layout; use coursebank::report; +use coursebank::seal::{self, SealFile}; use coursebank::store::{self, Store}; use coursebank::students; +use coursebank::typst::diagnostic as typst_diagnostic; +use coursebank::typst::{self, Variant}; use coursebank::yaml; -use crate::cli::{AnalyzeCommand, CalibrateArgs, Cli, IngestCommand, ReportCommand}; +use crate::cli::{AnalyzeCommand, CalibrateArgs, Cli, IngestCommand, ReportCommand, SealArgs}; use crate::commands::Outcome; use crate::helpers::{context, load, load_record, read_salt, responses_for, truncate}; -/// `ingest`: read a Gradescope directory or a Canvas CSV into the response store. +/// `seal`: freeze what was administered, or check the freeze. /// -/// Enriches the parsed responses against the record, optionally pseudonymizes the -/// identifiers, and — unless `--dry-run` — writes them in the chosen format. +/// Write the seal after exporting the papers and before printing them. It records +/// the stem and options of every item, the printed-letter map for every form, and +/// digests over both, so that the question "is the bank still what the students +/// saw?" has an answer six months from now. +pub(crate) fn seal(cli: &Cli, args: &SealArgs) -> Result { + let catalog = load(cli)?; + let record = load_record(&catalog, &args.id)?; + let path = args + .out + .clone() + .unwrap_or_else(|| SealFile::path(&catalog.layout, &record.assessment.id)); + + if args.check { + let existing = SealFile::load(&path).map_err(|_| { + Error::usage(format!( + "no seal at {}. Write one with `coursebank seal {}`", + path.display(), + args.id + )) + })?; + + let drift = existing.verify(&catalog, &record); + if drift.is_empty() { + println!( + "{} matches the seal written {} ({})", + args.id, + existing.seal.sealed_on, + seal::short(&existing.seal.digest) + ); + return Ok(Outcome::Ok); + } + + println!( + "{} finding(s) against the seal written {}:\n", + drift.len(), + existing.seal.sealed_on + ); + for finding in &drift { + println!(" {:<6} {}", finding.severity.label(), finding.message); + } + if drift.iter().any(|d| d.is_blocking()) { + println!( + "\nAnalysis that pools this administration with another is comparing two \ + different questions. Per-option feedback in student reports may name the wrong \ + option." + ); + } + return Ok(Outcome::Findings); + } + + if path.exists() && !args.force { + return Err(Error::usage(format!( + "{} already exists. A seal is meant to be written once, before the exam is printed; \ + pass --force only if you are re-sealing an assessment that was never administered", + path.display() + ))); + } + + let opts = seal::Options { + content: !args.no_content, + forms: args.form.clone(), + }; + let file = seal::build(&catalog, &record, &opts)?; + file.save(&path)?; + + println!( + "wrote {} ({} item(s), {} form(s), {})", + path.display(), + file.items.len(), + file.forms.len(), + seal::short(&file.seal.digest) + ); + let mut advisories = Vec::new(); + for form in &file.forms { + let permuted = form + .questions + .iter() + .filter(|q| q.options.iter().any(|o| o.printed != o.canonical)) + .count(); + println!( + " form {}: {} question(s), {} with permuted options, {}", + form.id, + form.questions.len(), + permuted, + seal::short(&form.digest) + ); + advisories.extend(seal::balance(form).notes()); + } + + // The last moment before printing is the only cheap moment to notice that a + // shuffle produced a sequence a student will read as a mistake. + if !advisories.is_empty() { + println!(); + for note in &advisories { + println!("! {note}"); + } + println!( + "\nRe-seed a form by editing its `seed:` in the record, then re-export and re-seal \ + with --force. Nothing else has to change." + ); + } + + if !cli.quiet { + println!( + "\nCommit this file. Check it any time with: coursebank seal {} --check", + args.id + ); + } + Ok(Outcome::Ok) +} + +/// `ingest`: read grading exports into the response store. +/// +/// The Gradescope path takes one directory per form and merges them into a single +/// administration, translating each form's printed letters into the bank's +/// lettering on the way in. See [`coursebank::intake`]. pub(crate) fn ingest(cli: &Cli, sub: &IngestCommand) -> Result { let catalog = load(cli)?; let (common, mut set) = match sub { - IngestCommand::Gradescope { dir, common } => { + IngestCommand::Gradescope { + sources, + common, + allow_mismatch, + recorded_numbers, + } => { let record = load_record(&catalog, &common.assessment)?; - let ctx = context(&catalog, &record, common)?; - let import = gradescope::ingest_dir(dir, &ctx)?; + let sealed = SealFile::find(&catalog.layout, &record.assessment.id)?; + + if let Some(file) = &sealed { + let drift = file.verify(&catalog, &record); + let blocking: Vec<&seal::Drift> = + drift.iter().filter(|d| d.is_blocking()).collect(); + for finding in &drift { + println!("! {} {}", finding.severity.label(), finding.message); + } + if !blocking.is_empty() { + println!( + "\nThe seal still describes the papers the students held, so ingest will \ + use it. The bank has moved since; `coursebank seal {} --check` lists \ + what.", + common.assessment + ); + } + } else if !cli.quiet { + println!( + "! no seal for {}; the option maps will be derived from the record as it \ + stands today. Write one next time with `coursebank seal {}` before printing", + common.assessment, common.assessment + ); + } + + let mut parsed = Vec::new(); + for text in sources { + parsed.push(Source::parse(text, common.form.as_deref())?); + } + for source in &parsed { + if !intake::looks_like_export(&source.dir) { + return Err(Error::usage(format!( + "{} holds no files named `1.csv` … `N.csv`. Gradescope writes one file \ + per question; point at the directory those were unzipped into", + source.dir.display() + ))); + } + } + + let date = match &common.date { + Some(text) => Some(text.parse()?), + None => None, + }; + let opts = intake::Options { + date, + numbering: if *recorded_numbers { + Numbering::Recorded + } else { + Numbering::Printed + }, + strict: !*allow_mismatch, + }; + + let result = intake::run(&catalog, &record, sealed.as_ref(), &parsed, &opts)?; + + for form in &result.forms { + println!("{}", intake::describe(form)); + } // Grading-time partial credit is an ambiguity signal worth surfacing - // right here, while the exam is fresh. - for question in &import.questions { + // while the exam is fresh, and it is per form because a regrade + // applied to one form and not the other is its own problem. + for (form, question) in result.questions() { for (letter, value, note) in question.partial_credit() { println!( - "! q{}: option {letter} earned {value} of {} points at grading time{}", + "! form {form} q{}: option {letter} earned {value} of {} points at \ + grading time{}", question.number, question.points_possible(), note.map(|n| format!(" — {n}")).unwrap_or_default() ); } } - (common, import.responses) + + (common, result.responses) } IngestCommand::Canvas { file, common } => { let record = load_record(&catalog, &common.assessment)?; @@ -93,7 +280,7 @@ pub(crate) fn ingest(cli: &Cli, sub: &IngestCommand) -> Result { println!("wrote {}", path.display()); } println!( - "\nNext: coursebank analyze items {}\n coursebank report cohort {}", + "\nNext: coursebank analyze items {}\n coursebank report cohort {} --typst", common.assessment, common.assessment ); Ok(Outcome::Ok) @@ -290,7 +477,11 @@ pub(crate) fn calibrate(cli: &Cli, args: &CalibrateArgs) -> Result { Ok(Outcome::Ok) } -/// `report`: write per-student reports or the instructor's cohort item analysis. +/// `report`: write per-student diagnostics or the class diagnostic. +/// +/// Markdown is still the default for both, because it diffs and reads in a +/// terminal. `--typst` adds a document per student, or one for the class, built +/// from the templates in `templates/` and compiled with `typst compile`. pub(crate) fn report(cli: &Cli, sub: &ReportCommand) -> Result { let catalog = load(cli)?; let store = Store::open(catalog.layout.data())?; @@ -302,6 +493,12 @@ pub(crate) fn report(cli: &Cli, sub: &ReportCommand) -> Result { out, ability, no_comparison, + typst: want_typst, + no_markdown, + no_questions, + no_feedback, + template, + json, } => { let record = load_record(&catalog, id)?; let set = responses_for(&store, &catalog, &record, false)?; @@ -311,26 +508,108 @@ pub(crate) fn report(cli: &Cli, sub: &ReportCommand) -> Result { None }; let cohort = students::summarize(&set, &catalog.course, Some(&catalog), fit.as_ref()); - - let opts = report::StudentOptions { - ability: *ability, - comparison: !no_comparison, - ..report::StudentOptions::default() - }; let dir = out .clone() .unwrap_or_else(|| catalog.layout.reports().join(id)); - let written = - report::write_all_students(&dir, &cohort, &catalog.course, &record, &opts, *html)?; + + warn_about_drift(&catalog, &record, cli.quiet); + + if !no_markdown { + let opts = report::StudentOptions { + ability: *ability, + comparison: !no_comparison, + ..report::StudentOptions::default() + }; + let written = report::write_all_students( + &dir, + &cohort, + &catalog.course, + &record, + &opts, + *html, + )?; + println!( + "wrote {} Markdown file(s) for {} student(s) in {}", + written.len(), + cohort.students.len(), + dir.display() + ); + } + + if !want_typst { + return Ok(Outcome::Ok); + } + + // The item analysis supplies the class rate per question, which is + // what makes "you missed q17, and so did most of the class" possible. + let analysis = + classical::analyze(&set, &Thresholds::default(), Some(&record), Some(&catalog)); + + let config = typst::load_config(&catalog.layout)?.resolve(Variant::StudentReport); + let meta = typst_diagnostic::Meta::new(&catalog, &record, cohort.students.len()); + let opts = diagnostic::Options { + comparison: !no_comparison, + questions: !no_questions, + feedback: !no_feedback, + ability: *ability, + ..diagnostic::Options::default() + }; + + let mut written = 0usize; + let mut used_embedded = false; + for summary in &cohort.students { + let built = + diagnostic::student(summary, &cohort, &catalog, &set, Some(&analysis), &opts); + let document = typst_diagnostic::render_student( + &catalog.layout, + &meta, + &built, + &config, + template.as_deref(), + )?; + used_embedded |= document.origin == typst::Origin::Embedded; + for warning in &document.warnings { + eprintln!("warning: {warning}"); + } + + let stem = typst_diagnostic::student_stem(id, &summary.student_key); + let path = dir.join(format!("{stem}.typ")); + yaml::write_text(&path, &document.text)?; + written += 1; + + if *json { + yaml::write_json(&dir.join(format!("{stem}.json")), &built)?; + } + } + println!( - "wrote {} file(s) for {} student(s) in {}", - written.len(), - cohort.students.len(), - dir.display() + "wrote {}", + typst_diagnostic::summary(written, cohort.students.len()) ); + if !cli.quiet { + println!( + "Compile them all with:\n for f in {}/*.typ; do typst compile \"$f\"; done", + dir.display() + ); + if used_embedded { + println!( + "These used the built-in template. To take over the layout:\n \ + coursebank template dump --variant student-report" + ); + } + } Ok(Outcome::Ok) } - ReportCommand::Cohort { id, html, out } => { + + ReportCommand::Cohort { + id, + html, + out, + typst: want_typst, + no_markdown, + template, + json, + } => { let record = load_record(&catalog, id)?; let set = responses_for(&store, &catalog, &record, false)?; let analysis = @@ -338,20 +617,58 @@ pub(crate) fn report(cli: &Cli, sub: &ReportCommand) -> Result { let fit = irt::fit(&set.matrix(false), &irt::Options::default()); let cohort = students::summarize(&set, &catalog.course, Some(&catalog), Some(&fit)); - let markdown = report::cohort(&analysis, &cohort, &catalog, &record, Some(&fit)); + warn_about_drift(&catalog, &record, cli.quiet); + let path = out .clone() .unwrap_or_else(|| catalog.layout.reports().join(format!("{id}-cohort.md"))); - yaml::write_text(&path, &markdown)?; - println!("wrote {}", path.display()); - if *html { - let html_path = path.with_extension("html"); - let title = format!("{} — item analysis", record.assessment.title); - yaml::write_text(&html_path, &report::to_html(&markdown, &title))?; - println!("wrote {}", html_path.display()); + if !no_markdown { + let markdown = report::cohort(&analysis, &cohort, &catalog, &record, Some(&fit)); + yaml::write_text(&path, &markdown)?; + println!("wrote {}", path.display()); + + if *html { + let html_path = path.with_extension("html"); + let title = format!("{} — item analysis", record.assessment.title); + yaml::write_text(&html_path, &report::to_html(&markdown, &title))?; + println!("wrote {}", html_path.display()); + } } - Ok(Outcome::Ok) + + let built = diagnostic::cohort(&analysis, &cohort, &catalog, &record, &set, Some(&fit)); + + for line in typst_diagnostic::headline(&built) { + println!("{line}"); + } + + if *want_typst { + let config = typst::load_config(&catalog.layout)?.resolve(Variant::CohortReport); + let meta = typst_diagnostic::Meta::new(&catalog, &record, cohort.students.len()); + let document = typst_diagnostic::render_cohort( + &catalog.layout, + &meta, + &built, + &config, + template.as_deref(), + )?; + for warning in &document.warnings { + eprintln!("warning: {warning}"); + } + let typst_path = path.with_extension("typ"); + yaml::write_text(&typst_path, &document.text)?; + println!("wrote {} (from {})", typst_path.display(), document.origin); + + if *json { + yaml::write_json(&path.with_extension("json"), &built)?; + } + } + + Ok(if built.revise.is_empty() { + Outcome::Ok + } else { + Outcome::Findings + }) } } } @@ -384,3 +701,37 @@ pub(crate) fn data(cli: &Cli) -> Result { } Ok(Outcome::Ok) } + +/// Says so when the bank has moved since the exam was sealed. +/// +/// A report built from a drifted bank is not merely stale: the per-option +/// feedback it prints was written for options the student may never have seen. +/// Worth one line at the top of every report run. +fn warn_about_drift( + catalog: &coursebank::catalog::Catalog, + record: &coursebank::assessment::AssessmentFile, + quiet: bool, +) { + let Ok(Some(file)) = SealFile::find(&catalog.layout, &record.assessment.id) else { + return; + }; + let drift = file.verify(catalog, record); + let serious: Vec<&seal::Drift> = drift + .iter() + .filter(|d| d.severity >= Severity::Medium) + .collect(); + if serious.is_empty() { + return; + } + eprintln!( + "warning: {} item(s) have changed since this exam was sealed; feedback in these reports \ + may describe options the students did not see. Run `coursebank seal {} --check`", + serious.len(), + record.assessment.id + ); + if !quiet { + for finding in serious.iter().take(3) { + eprintln!(" {}", finding.message); + } + } +} diff --git a/src/commands/export.rs b/src/commands/export.rs index 1a4f2e1..60450f2 100644 --- a/src/commands/export.rs +++ b/src/commands/export.rs @@ -290,7 +290,11 @@ fn pick_practice_variants(names: &[String]) -> Result> { .collect()) } -/// Resolves the `--variant` flags, defaulting to every document. +/// Resolves the `--variant` flags, defaulting to the exam set. +/// +/// The default is [`typst::Variant::EXAM`] rather than every variant: a +/// diagnostic is built from responses, not from an assessment record, so +/// `export typst` has nothing to build one out of. /// /// # Arguments /// @@ -306,7 +310,7 @@ fn pick_practice_variants(names: &[String]) -> Result> { /// Returns [`Error::Usage`] naming the valid tokens. fn pick_variants(names: &[String]) -> Result> { if names.is_empty() { - return Ok(typst::Variant::ALL.to_vec()); + return Ok(typst::Variant::EXAM.to_vec()); } let mut wanted = Vec::new(); for name in names { @@ -384,7 +388,15 @@ pub(crate) fn template(cli: &Cli, sub: &TemplateCommand) -> Result { force, stdout, } => { - let variants = pick_variants(variant)?; + // `export typst` defaults to the exam set, because a report is not + // built from an assessment record. Dumping is the opposite case: with + // no `--variant` it should hand over every template there is, + // including the two reports. + let variants = if variant.is_empty() { + typst::Variant::ALL.to_vec() + } else { + pick_variants(variant)? + }; if *stdout { for (index, v) in variants.iter().enumerate() { diff --git a/src/commands/handlers.rs b/src/commands/handlers.rs new file mode 100644 index 0000000..72ebe0c --- /dev/null +++ b/src/commands/handlers.rs @@ -0,0 +1,558 @@ +// SPDX-License-Identifier: Prosperity-3.0.0 +// Copyright Scientific Computing Studio +// Source: https://git.scient.ing/education/coursebank + +//! Replacement handlers for `src/commands/analysis.rs`. +//! +//! This file is not a module of its own: `seal` is new, and `ingest` and `report` +//! replace the functions of the same name in `commands/analysis.rs`. Paste them +//! in there, add the imports listed at the top, and delete this file. It is kept +//! separate here only so the diff against the existing file is obvious. +//! +//! Imports `commands/analysis.rs` needs on top of what it already has: +//! +//! ```ignore +//! use std::collections::BTreeMap; +//! +//! use coursebank::decode::Numbering; +//! use coursebank::diagnostic; +//! use coursebank::intake::{self, Source}; +//! use coursebank::seal::{self, SealFile}; +//! use coursebank::typst::{self, Variant}; +//! use coursebank::typst::diagnostic as typst_diagnostic; +//! +//! use crate::cli::SealArgs; +//! ``` + +use std::collections::BTreeMap; + +use coursebank::canvas; +use coursebank::catalog::Severity; +use coursebank::classical::{self, Thresholds}; +use coursebank::decode::Numbering; +use coursebank::diagnostic; +use coursebank::error::{Error, Result}; +use coursebank::intake::{self, Source}; +use coursebank::irt; +use coursebank::report; +use coursebank::seal::{self, SealFile}; +use coursebank::store::Store; +use coursebank::students; +use coursebank::typst::diagnostic as typst_diagnostic; +use coursebank::typst::{self, Variant}; +use coursebank::yaml; + +use crate::cli::{Cli, IngestCommand, ReportCommand, SealArgs}; +use crate::commands::Outcome; +use crate::helpers::{context, load, load_record, read_salt, responses_for}; + +/// `seal`: freeze what was administered, or check the freeze. +/// +/// Write the seal after exporting the papers and before printing them. It records +/// the stem and options of every item, the printed-letter map for every form, and +/// digests over both, so that the question "is the bank still what the students +/// saw?" has an answer six months from now. +pub(crate) fn seal(cli: &Cli, args: &SealArgs) -> Result { + let catalog = load(cli)?; + let record = load_record(&catalog, &args.id)?; + let path = args + .out + .clone() + .unwrap_or_else(|| SealFile::path(&catalog.layout, &record.assessment.id)); + + if args.check { + let existing = SealFile::load(&path).map_err(|_| { + Error::usage(format!( + "no seal at {}. Write one with `coursebank seal {}`", + path.display(), + args.id + )) + })?; + + let drift = existing.verify(&catalog, &record); + if drift.is_empty() { + println!( + "{} matches the seal written {} ({})", + args.id, + existing.seal.sealed_on, + seal::short(&existing.seal.digest) + ); + return Ok(Outcome::Ok); + } + + println!( + "{} finding(s) against the seal written {}:\n", + drift.len(), + existing.seal.sealed_on + ); + for finding in &drift { + println!(" {:<6} {}", finding.severity.label(), finding.message); + } + if drift.iter().any(|d| d.is_blocking()) { + println!( + "\nAnalysis that pools this administration with another is comparing two \ + different questions. Per-option feedback in student reports may name the wrong \ + option." + ); + } + return Ok(Outcome::Findings); + } + + if path.exists() && !args.force { + return Err(Error::usage(format!( + "{} already exists. A seal is meant to be written once, before the exam is printed; \ + pass --force only if you are re-sealing an assessment that was never administered", + path.display() + ))); + } + + let opts = seal::Options { + content: !args.no_content, + forms: args.form.clone(), + }; + let file = seal::build(&catalog, &record, &opts)?; + file.save(&path)?; + + println!( + "wrote {} ({} item(s), {} form(s), {})", + path.display(), + file.items.len(), + file.forms.len(), + seal::short(&file.seal.digest) + ); + let mut advisories = Vec::new(); + for form in &file.forms { + let permuted = form + .questions + .iter() + .filter(|q| q.options.iter().any(|o| o.printed != o.canonical)) + .count(); + println!( + " form {}: {} question(s), {} with permuted options, {}", + form.id, + form.questions.len(), + permuted, + seal::short(&form.digest) + ); + advisories.extend(seal::balance(form).notes()); + } + + // The last moment before printing is the only cheap moment to notice that a + // shuffle produced a sequence a student will read as a mistake. + if !advisories.is_empty() { + println!(); + for note in &advisories { + println!("! {note}"); + } + println!( + "\nRe-seed a form by editing its `seed:` in the record, then re-export and re-seal \ + with --force. Nothing else has to change." + ); + } + + if !cli.quiet { + println!( + "\nCommit this file. Check it any time with: coursebank seal {} --check", + args.id + ); + } + Ok(Outcome::Ok) +} + +/// `ingest`: read grading exports into the response store. +/// +/// The Gradescope path takes one directory per form and merges them into a single +/// administration, translating each form's printed letters into the bank's +/// lettering on the way in. See [`coursebank::intake`]. +pub(crate) fn ingest(cli: &Cli, sub: &IngestCommand) -> Result { + let catalog = load(cli)?; + + let (common, mut set) = match sub { + IngestCommand::Gradescope { + sources, + common, + allow_mismatch, + recorded_numbers, + } => { + let record = load_record(&catalog, &common.assessment)?; + let sealed = SealFile::find(&catalog.layout, &record.assessment.id)?; + + if let Some(file) = &sealed { + let drift = file.verify(&catalog, &record); + let blocking: Vec<&seal::Drift> = + drift.iter().filter(|d| d.is_blocking()).collect(); + for finding in &drift { + println!("! {} {}", finding.severity.label(), finding.message); + } + if !blocking.is_empty() { + println!( + "\nThe seal still describes the papers the students held, so ingest will \ + use it. The bank has moved since; `coursebank seal {} --check` lists \ + what.", + common.assessment + ); + } + } else if !cli.quiet { + println!( + "! no seal for {}; the option maps will be derived from the record as it \ + stands today. Write one next time with `coursebank seal {}` before printing", + common.assessment, common.assessment + ); + } + + let mut parsed = Vec::new(); + for text in sources { + parsed.push(Source::parse(text, common.form.as_deref())?); + } + for source in &parsed { + if !intake::looks_like_export(&source.dir) { + return Err(Error::usage(format!( + "{} holds no files named `1.csv` … `N.csv`. Gradescope writes one file \ + per question; point at the directory those were unzipped into", + source.dir.display() + ))); + } + } + + let date = match &common.date { + Some(text) => Some(text.parse()?), + None => None, + }; + let opts = intake::Options { + date, + numbering: if *recorded_numbers { + Numbering::Recorded + } else { + Numbering::Printed + }, + strict: !*allow_mismatch, + }; + + let result = intake::run(&catalog, &record, sealed.as_ref(), &parsed, &opts)?; + + for form in &result.forms { + println!("{}", intake::describe(form)); + } + + // Grading-time partial credit is an ambiguity signal worth surfacing + // while the exam is fresh, and it is per form because a regrade + // applied to one form and not the other is its own problem. + for (form, question) in result.questions() { + for (letter, value, note) in question.partial_credit() { + println!( + "! form {form} q{}: option {letter} earned {value} of {} points at \ + grading time{}", + question.number, + question.points_possible(), + note.map(|n| format!(" — {n}")).unwrap_or_default() + ); + } + } + + (common, result.responses) + } + IngestCommand::Canvas { file, common } => { + let record = load_record(&catalog, &common.assessment)?; + let ctx = context(&catalog, &record, common)?; + let set = canvas::ingest(file, &ctx, Some(&record), Some(&catalog))?; + (common, set) + } + }; + + let record = load_record(&catalog, &common.assessment)?; + set.enrich(&record, Some(&catalog)); + + if common.pseudonymize { + let salt = read_salt(common.salt_file.as_deref())?; + set.pseudonymize(&salt); + println!("identifiers replaced with keyed pseudonyms"); + } + + for warning in &set.warnings { + println!("! {warning}"); + } + + println!( + "\n{} response(s): {} student(s) x {} item(s)", + set.rows.len(), + set.students().len(), + set.all_items().len() + ); + + if common.dry_run { + println!("(dry run, nothing written)"); + return Ok(Outcome::Ok); + } + + let mut store = Store::open(catalog.layout.data())?; + if let Some(format) = common.format { + store = store.with_format(format.as_format())?; + } + for path in store.write(&set)? { + println!("wrote {}", path.display()); + } + println!( + "\nNext: coursebank analyze items {}\n coursebank report cohort {} --typst", + common.assessment, common.assessment + ); + Ok(Outcome::Ok) +} + +/// `report`: write per-student diagnostics or the class diagnostic. +/// +/// Markdown is still the default for both, because it diffs and reads in a +/// terminal. `--typst` adds a document per student, or one for the class, built +/// from the templates in `templates/` and compiled with `typst compile`. +pub(crate) fn report(cli: &Cli, sub: &ReportCommand) -> Result { + let catalog = load(cli)?; + let store = Store::open(catalog.layout.data())?; + + match sub { + ReportCommand::Students { + id, + html, + out, + ability, + no_comparison, + typst: want_typst, + no_markdown, + no_questions, + no_feedback, + template, + json, + } => { + let record = load_record(&catalog, id)?; + let set = responses_for(&store, &catalog, &record, false)?; + let fit = if *ability { + Some(irt::fit(&set.matrix(false), &irt::Options::default())) + } else { + None + }; + let cohort = students::summarize(&set, &catalog.course, Some(&catalog), fit.as_ref()); + let dir = out + .clone() + .unwrap_or_else(|| catalog.layout.reports().join(id)); + + warn_about_drift(&catalog, &record, cli.quiet); + + if !no_markdown { + let opts = report::StudentOptions { + ability: *ability, + comparison: !no_comparison, + ..report::StudentOptions::default() + }; + let written = report::write_all_students( + &dir, + &cohort, + &catalog.course, + &record, + &opts, + *html, + )?; + println!( + "wrote {} Markdown file(s) for {} student(s) in {}", + written.len(), + cohort.students.len(), + dir.display() + ); + } + + if !want_typst { + return Ok(Outcome::Ok); + } + + // The item analysis supplies the class rate per question, which is + // what makes "you missed q17, and so did most of the class" possible. + let analysis = + classical::analyze(&set, &Thresholds::default(), Some(&record), Some(&catalog)); + + let config = typst::load_config(&catalog.layout)?.resolve(Variant::StudentReport); + let meta = typst_diagnostic::Meta::new(&catalog, &record, cohort.students.len()); + let opts = diagnostic::Options { + comparison: !no_comparison, + questions: !no_questions, + feedback: !no_feedback, + ability: *ability, + ..diagnostic::Options::default() + }; + + let mut written = 0usize; + let mut used_embedded = false; + for summary in &cohort.students { + let built = diagnostic::student( + summary, + &cohort, + &catalog, + &set, + Some(&analysis), + &opts, + ); + let document = typst_diagnostic::render_student( + &catalog.layout, + &meta, + &built, + &config, + template.as_deref(), + )?; + used_embedded |= document.origin == typst::Origin::Embedded; + for warning in &document.warnings { + eprintln!("warning: {warning}"); + } + + let stem = typst_diagnostic::student_stem(id, &summary.student_key); + let path = dir.join(format!("{stem}.typ")); + yaml::write_text(&path, &document.text)?; + written += 1; + + if *json { + yaml::write_json(&dir.join(format!("{stem}.json")), &built)?; + } + } + + println!( + "wrote {}", + typst_diagnostic::summary(written, cohort.students.len()) + ); + if !cli.quiet { + println!( + "Compile them all with:\n for f in {}/*.typ; do typst compile \"$f\"; done", + dir.display() + ); + if used_embedded { + println!( + "These used the built-in template. To take over the layout:\n \ + coursebank template dump --variant student-report" + ); + } + } + Ok(Outcome::Ok) + } + + ReportCommand::Cohort { + id, + html, + out, + typst: want_typst, + no_markdown, + template, + json, + } => { + let record = load_record(&catalog, id)?; + let set = responses_for(&store, &catalog, &record, false)?; + let analysis = + classical::analyze(&set, &Thresholds::default(), Some(&record), Some(&catalog)); + let fit = irt::fit(&set.matrix(false), &irt::Options::default()); + let cohort = students::summarize(&set, &catalog.course, Some(&catalog), Some(&fit)); + + warn_about_drift(&catalog, &record, cli.quiet); + + let path = out + .clone() + .unwrap_or_else(|| catalog.layout.reports().join(format!("{id}-cohort.md"))); + + if !no_markdown { + let markdown = report::cohort(&analysis, &cohort, &catalog, &record, Some(&fit)); + yaml::write_text(&path, &markdown)?; + println!("wrote {}", path.display()); + + if *html { + let html_path = path.with_extension("html"); + let title = format!("{} — item analysis", record.assessment.title); + yaml::write_text(&html_path, &report::to_html(&markdown, &title))?; + println!("wrote {}", html_path.display()); + } + } + + let built = diagnostic::cohort( + &analysis, + &cohort, + &catalog, + &record, + &set, + Some(&fit), + ); + + for line in typst_diagnostic::headline(&built) { + println!("{line}"); + } + + if *want_typst { + let config = typst::load_config(&catalog.layout)?.resolve(Variant::CohortReport); + let meta = typst_diagnostic::Meta::new(&catalog, &record, cohort.students.len()); + let document = typst_diagnostic::render_cohort( + &catalog.layout, + &meta, + &built, + &config, + template.as_deref(), + )?; + for warning in &document.warnings { + eprintln!("warning: {warning}"); + } + let typst_path = path.with_extension("typ"); + yaml::write_text(&typst_path, &document.text)?; + println!("wrote {} (from {})", typst_path.display(), document.origin); + + if *json { + yaml::write_json(&path.with_extension("json"), &built)?; + } + } + + Ok(if built.revise.is_empty() { + Outcome::Ok + } else { + Outcome::Findings + }) + } + } +} + +/// Says so when the bank has moved since the exam was sealed. +/// +/// A report built from a drifted bank is not merely stale: the per-option +/// feedback it prints was written for options the student may never have seen. +/// Worth one line at the top of every report run. +fn warn_about_drift( + catalog: &coursebank::catalog::Catalog, + record: &coursebank::assessment::AssessmentFile, + quiet: bool, +) { + let Ok(Some(file)) = SealFile::find(&catalog.layout, &record.assessment.id) else { + return; + }; + let drift = file.verify(catalog, record); + let serious: Vec<&seal::Drift> = drift + .iter() + .filter(|d| d.severity >= Severity::Medium) + .collect(); + if serious.is_empty() { + return; + } + eprintln!( + "warning: {} item(s) have changed since this exam was sealed; feedback in these reports \ + may describe options the students did not see. Run `coursebank seal {} --check`", + serious.len(), + record.assessment.id + ); + if !quiet { + for finding in serious.iter().take(3) { + eprintln!(" {}", finding.message); + } + } +} + +/// Counts how many students each form was given to, for the ingest summary. +/// +/// Kept here rather than in the library because it exists to print a line. +#[allow(dead_code)] +fn students_per_form(set: &coursebank::responses::ResponseSet) -> BTreeMap { + let mut seen: BTreeMap> = BTreeMap::new(); + for row in &set.rows { + if let Some(form) = row.form.as_deref() { + seen.entry(form.to_string()) + .or_default() + .insert(row.student_key.as_str()); + } + } + seen.into_iter().map(|(k, v)| (k, v.len())).collect() +} diff --git a/src/data.rs b/src/data.rs index 1fb0417..f36152c 100644 --- a/src/data.rs +++ b/src/data.rs @@ -26,7 +26,9 @@ //! `--no-default-features` a one-file change rather than a refactor. pub mod canvas; +pub mod decode; pub mod gradescope; +pub mod intake; pub mod responses; pub mod store; pub mod store_parquet; diff --git a/src/data/canvas.rs b/src/data/canvas.rs index 16f9982..0d9e873 100644 --- a/src/data/canvas.rs +++ b/src/data/canvas.rs @@ -325,6 +325,10 @@ pub fn ingest( assessment_id: ctx.assessment_id.clone(), date: ctx.date, form: ctx.form.clone(), + // Canvas numbers questions as the record does, and its exports + // carry no printed order, so there is no printed position to + // record and no letter map to decode against. + form_position: None, student_key: student_key.clone(), sid: sid.clone(), name: name.clone(), @@ -334,7 +338,9 @@ pub fn ingest( item_ref, item_version: None, selected, + selected_source: Vec::new(), eliminated: Vec::new(), + eliminated_source: Vec::new(), correct, credit, points_possible: points, diff --git a/src/data/decode.rs b/src/data/decode.rs new file mode 100644 index 0000000..3f51421 --- /dev/null +++ b/src/data/decode.rs @@ -0,0 +1,803 @@ +// SPDX-License-Identifier: Prosperity-3.0.0 +// Copyright Scientific Computing Studio +// Source: https://git.scient.ing/education/coursebank + +//! Turning what a student marked into what a student chose. +//! +//! A grading export speaks in positions and printed letters. Question 14 is the +//! fourteenth thing on the page; option C is the third bubble. An item bank speaks +//! in ids and its own lettering. When forms shuffle, those two vocabularies +//! disagree, and every analysis downstream of the disagreement is wrong in a way +//! that looks right: +//! +//! * Pooled distractor statistics add form A's option C to form B's option C, +//! which are different sentences. The resulting table is noise with the shape of +//! data. +//! * A student report looks up the misconception recorded on option C and shows it +//! to a student who chose a different option. The feedback is confident, +//! specific, and about the wrong thing. +//! * Any `credit_overrides` written in the record's lettering are applied to +//! whoever happened to mark that letter on their form. +//! +//! None of these fail loudly. That is the argument for doing the translation once, +//! at ingest, and storing both sides of it. +//! +//! # What a decoder knows +//! +//! For one form: which recorded question number sits at each printed position, +//! which bank letter each printed letter stands for, and which printed letters are +//! keyed. It is built from a [`SealFile`] when one exists, and derived from the +//! record and the form seed when one does not. Sealed is better, and not only +//! because it is faster: a derived decoder describes the form the bank *would* +//! print today, while a sealed one describes the form that was actually printed. +//! +//! # Catching a swapped directory +//! +//! Gradescope's point-value row reveals which printed letter earned full credit on +//! every question. A decoder knows what that letter should be. Comparing them +//! across a whole directory is close to a proof of which form the directory holds: +//! agreement is near total for the right form and near chance for the wrong one. +//! [`identify_form`] uses that to refuse an ingest that names form A over a +//! directory of form B papers, which is otherwise a mistake nobody catches until +//! the item statistics look strange three weeks later. + +use std::collections::{BTreeMap, BTreeSet}; + +use crate::assessment::{AssessmentFile, Form}; +use crate::catalog::Catalog; +use crate::error::{Error, Result}; +use crate::gradescope::Question; +use crate::responses::ResponseSet; +use crate::seal::{SealFile, printed_letter}; +use crate::select; + +/// Where a decoder's mapping came from. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Provenance { + /// Read from a seal written before administration. Authoritative. + Seal, + /// Derived from the assessment record and the form seed, as of now. + Derived, +} + +impl Provenance { + /// A short label for output. + pub fn label(self) -> &'static str { + match self { + Provenance::Seal => "seal", + Provenance::Derived => "derived from the record", + } + } +} + +/// How a grading export numbers its questions. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Numbering { + /// The export counts printed positions, which is what Gradescope's `N.csv` + /// file names mean. Positions are translated to recorded numbers. + Printed, + /// The export already carries recorded question numbers, so numbers pass + /// through untouched. + Recorded, +} + +/// One question's mapping on one form. +#[derive(Debug, Clone)] +pub struct QuestionMap { + /// Printed position on the page, counting from 1. + pub position: u32, + /// The recorded question number, the join key to the record and the store. + pub number: u32, + /// The item's global id. + pub item: String, + /// Keyed letters as printed on this form. + pub printed_key: Vec, + /// Keyed letters in the bank's own lettering. + pub canonical_key: Vec, + /// Printed letter to bank letter. + pub to_canonical: BTreeMap, + /// Bank letter to printed letter. + pub to_printed: BTreeMap, +} + +impl QuestionMap { + /// The bank letter a printed letter stands for. + /// + /// # Arguments + /// + /// * `printed` - the letter as the student saw it. + /// + /// # Returns + /// + /// The bank letter, or `None` when the printed letter is not one of this + /// question's options. + pub fn canonical(&self, printed: &str) -> Option<&str> { + self.to_canonical + .get(&printed.trim().to_ascii_uppercase()) + .map(|s| s.as_str()) + } + + /// How many options this question has. + pub fn n_options(&self) -> usize { + self.to_canonical.len() + } + + /// Whether this question's options were actually permuted. + pub fn is_permuted(&self) -> bool { + self.to_canonical.iter().any(|(k, v)| k != v) + } +} + +/// One form's full mapping. +#[derive(Debug, Clone)] +pub struct FormDecoder { + /// The form id. + pub form: String, + /// Where the mapping came from. + pub provenance: Provenance, + /// Questions by printed position. + by_position: BTreeMap, + /// Questions by recorded number. + by_number: BTreeMap, +} + +impl FormDecoder { + /// Builds a decoder from the question maps. + fn assemble(form: String, provenance: Provenance, maps: Vec) -> FormDecoder { + let by_position = maps.iter().map(|m| (m.position, m.clone())).collect(); + let by_number = maps.into_iter().map(|m| (m.number, m)).collect(); + FormDecoder { + form, + provenance, + by_position, + by_number, + } + } + + /// Reads one form's mapping out of a seal. + /// + /// # Arguments + /// + /// * `seal` - the seal. + /// * `form_id` - the form to decode, matched case-insensitively. + /// + /// # Returns + /// + /// The decoder. + /// + /// # Errors + /// + /// Returns [`Error::Usage`] when the seal does not cover that form. + pub fn from_seal(seal: &SealFile, form_id: &str) -> Result { + let form = seal.form(form_id).ok_or_else(|| { + Error::usage(format!( + "the seal for `{}` does not cover form `{form_id}`; it covers {}", + seal.seal.assessment, + seal.forms + .iter() + .map(|f| f.id.as_str()) + .collect::>() + .join(", ") + )) + })?; + + let maps = form + .questions + .iter() + .map(|q| { + let mut to_canonical = BTreeMap::new(); + let mut to_printed = BTreeMap::new(); + for map in &q.options { + to_canonical.insert(map.printed.clone(), map.canonical.clone()); + to_printed.insert(map.canonical.clone(), map.printed.clone()); + } + let canonical_key: Vec = q + .printed_key + .iter() + .filter_map(|p| to_canonical.get(p).cloned()) + .collect(); + QuestionMap { + position: q.position, + number: q.number, + item: q.item.clone(), + printed_key: q.printed_key.clone(), + canonical_key, + to_canonical, + to_printed, + } + }) + .collect(); + + Ok(FormDecoder::assemble( + form.id.clone(), + Provenance::Seal, + maps, + )) + } + + /// Derives one form's mapping from the record and the bank. + /// + /// Uses the same two functions every export calls, so a derived decoder and a + /// freshly exported paper agree by construction. + /// + /// # Arguments + /// + /// * `catalog` - the loaded course. + /// * `record` - the assessment record. + /// * `form` - the form. + /// + /// # Returns + /// + /// The decoder. + /// + /// # Errors + /// + /// Returns [`Error::Unresolved`] when a placement references a missing item. + pub fn derive(catalog: &Catalog, record: &AssessmentFile, form: &Form) -> Result { + let printed: Vec<_> = select::layout(record, form) + .into_iter() + .filter(|p| !p.dropped) + .collect(); + + let mut maps = Vec::with_capacity(printed.len()); + for (index, placement) in printed.iter().enumerate() { + let entry = catalog.require(&placement.item)?; + let item = &entry.item; + let order = select::option_order(form, &placement.item, item.options.len()); + + let canonical_key: BTreeSet = if placement.key.is_empty() { + item.key_letters().into_iter().collect() + } else { + placement.key.iter().cloned().collect() + }; + + let mut to_canonical = BTreeMap::new(); + let mut to_printed = BTreeMap::new(); + let mut printed_key = Vec::new(); + for (position, source_index) in order.iter().enumerate() { + let canonical = item + .options + .get(*source_index) + .map(|c| c.id.clone()) + .unwrap_or_else(|| printed_letter(*source_index)); + let label = printed_letter(position); + if canonical_key.contains(&canonical) { + printed_key.push(label.clone()); + } + to_canonical.insert(label.clone(), canonical.clone()); + to_printed.insert(canonical, label); + } + + maps.push(QuestionMap { + position: index as u32 + 1, + number: placement.number, + item: placement.item.clone(), + printed_key, + canonical_key: canonical_key.into_iter().collect(), + to_canonical, + to_printed, + }); + } + + Ok(FormDecoder::assemble( + form.id.clone(), + Provenance::Derived, + maps, + )) + } + + /// Builds a decoder, preferring the seal. + /// + /// # Arguments + /// + /// * `seal` - the seal, when one has been written. + /// * `catalog` - the loaded course. + /// * `record` - the assessment record. + /// * `form` - the form. + /// + /// # Returns + /// + /// The decoder. + /// + /// # Errors + /// + /// As [`FormDecoder::from_seal`] and [`FormDecoder::derive`]. A seal that does + /// not cover the requested form falls back to deriving rather than failing, + /// since a form added after sealing is a real situation. + pub fn resolve( + seal: Option<&SealFile>, + catalog: &Catalog, + record: &AssessmentFile, + form: &Form, + ) -> Result { + if let Some(seal) = seal { + if seal.form(&form.id).is_some() { + return FormDecoder::from_seal(seal, &form.id); + } + } + FormDecoder::derive(catalog, record, form) + } + + /// The question at a printed position. + /// + /// # Arguments + /// + /// * `position` - the printed position, counting from 1. + pub fn at_position(&self, position: u32) -> Option<&QuestionMap> { + self.by_position.get(&position) + } + + /// The question with a recorded number. + /// + /// # Arguments + /// + /// * `number` - the recorded number. + pub fn at_number(&self, number: u32) -> Option<&QuestionMap> { + self.by_number.get(&number) + } + + /// The question an export's numbering refers to. + /// + /// # Arguments + /// + /// * `n` - the number as the export gives it. + /// * `numbering` - how the export numbers questions. + pub fn lookup(&self, n: u32, numbering: Numbering) -> Option<&QuestionMap> { + match numbering { + Numbering::Printed => self.at_position(n), + Numbering::Recorded => self.at_number(n), + } + } + + /// How many questions this form prints. + pub fn len(&self) -> usize { + self.by_position.len() + } + + /// Whether the form prints nothing, which means the record is empty. + pub fn is_empty(&self) -> bool { + self.by_position.is_empty() + } + + /// Whether any question on this form has permuted options. + /// + /// Used to decide whether to say anything about translation at all: on an + /// unshuffled form the whole mechanism is an identity map and mentioning it + /// is noise. + pub fn is_permuted(&self) -> bool { + self.by_position.values().any(|q| q.is_permuted()) + } + + /// Whether printed positions and recorded numbers disagree anywhere. + /// + /// True when items were shuffled, and also when a bonus item sits mid-record, + /// since the layout moves bonus items to the end of the paper. + pub fn is_renumbered(&self) -> bool { + self.by_position.values().any(|q| q.position != q.number) + } +} + +/// What a directory of graded questions says about which form it holds. +#[derive(Debug, Clone)] +pub struct FormFit { + /// The form id. + pub form: String, + /// Questions whose graded key matched this form's printed key. + pub matched: usize, + /// Questions that could be compared at all. + pub compared: usize, + /// Question positions where the graded key disagreed. + pub mismatches: Vec, +} + +impl FormFit { + /// The share of comparable questions that agreed. + pub fn rate(&self) -> f64 { + if self.compared == 0 { + 0.0 + } else { + self.matched as f64 / self.compared as f64 + } + } + + /// Whether the fit is good enough to proceed without a warning. + /// + /// The threshold is high on purpose. A correctly matched directory agrees on + /// every question; anything less than total agreement is either a regrade that + /// moved a key or the wrong directory, and both are worth a sentence. + pub fn is_convincing(&self) -> bool { + self.compared > 0 && self.matched == self.compared + } +} + +/// Compares a parsed Gradescope directory against one form's expected keys. +/// +/// # Arguments +/// +/// * `questions` - the parsed question files. +/// * `decoder` - the form to test against. +/// * `numbering` - how the export numbers questions. +/// +/// # Returns +/// +/// The fit. +pub fn fit_form(questions: &[Question], decoder: &FormDecoder, numbering: Numbering) -> FormFit { + let mut matched = 0usize; + let mut compared = 0usize; + let mut mismatches = Vec::new(); + + for question in questions { + let Some(map) = decoder.lookup(question.number, numbering) else { + continue; + }; + let graded: BTreeSet = question.keyed().into_iter().collect(); + if graded.is_empty() { + continue; + } + let expected: BTreeSet = map.printed_key.iter().cloned().collect(); + if expected.is_empty() { + continue; + } + compared += 1; + if graded == expected { + matched += 1; + } else { + mismatches.push(question.number); + } + } + + FormFit { + form: decoder.form.clone(), + matched, + compared, + mismatches, + } +} + +/// Ranks every candidate form against a directory. +/// +/// # Arguments +/// +/// * `questions` - the parsed question files. +/// * `decoders` - one decoder per declared form. +/// * `numbering` - how the export numbers questions. +/// +/// # Returns +/// +/// The fits, best first. +pub fn identify_form( + questions: &[Question], + decoders: &[FormDecoder], + numbering: Numbering, +) -> Vec { + let mut fits: Vec = decoders + .iter() + .map(|d| fit_form(questions, d, numbering)) + .collect(); + fits.sort_by(|a, b| { + b.rate() + .partial_cmp(&a.rate()) + .unwrap_or(std::cmp::Ordering::Equal) + .then_with(|| a.form.cmp(&b.form)) + }); + fits +} + +/// Explains a fit in a sentence, or says nothing when the fit is perfect. +/// +/// # Arguments +/// +/// * `claimed` - the form the directory was ingested as. +/// * `fits` - every form's fit, best first. +/// +/// # Returns +/// +/// A warning, or `None`. +pub fn form_warning(claimed: &str, fits: &[FormFit]) -> Option { + let mine = fits.iter().find(|f| f.form.eq_ignore_ascii_case(claimed))?; + if mine.is_convincing() { + return None; + } + if mine.compared == 0 { + return Some(format!( + "form {claimed}: the export carries no point values, so the graded keys could not be \ + checked against the form. Nothing verified this directory is form {claimed}" + )); + } + + let better = fits + .iter() + .find(|f| !f.form.eq_ignore_ascii_case(claimed) && f.rate() > mine.rate()); + + let head = format!( + "form {claimed}: the graded key matches this form on {} of {} question(s)", + mine.matched, mine.compared + ); + let where_ = if mine.mismatches.is_empty() { + String::new() + } else { + let list: Vec = mine + .mismatches + .iter() + .take(8) + .map(|n| n.to_string()) + .collect(); + format!( + " (q{}{})", + list.join(", q"), + if mine.mismatches.len() > 8 { + ", …" + } else { + "" + } + ) + }; + match better { + Some(other) => Some(format!( + "{head}{where_}, but matches form {} on {} of {}. This directory is almost certainly \ + form {}, not form {claimed}", + other.form, other.matched, other.compared, other.form + )), + None => Some(format!( + "{head}{where_}. Either those questions were regraded after printing, or the form is \ + not the one named" + )), + } +} + +/// Translates a form's responses into the bank's vocabulary. +/// +/// Rewrites, for every row whose `form` matches this decoder: +/// +/// * `item_number`, from printed position to recorded number, when the export +/// numbers by position; +/// * `form_position`, recording where the question sat on the page; +/// * `selected_source` and `eliminated_source`, the bank letters for what was +/// marked. The printed letters stay in `selected` and `eliminated`, because what +/// a student physically marked is the fact and the translation is the +/// interpretation. +/// +/// Correctness and credit are untouched. Both come from the grading platform, +/// which scored the paper the student actually held, and are already right. +/// +/// # Arguments +/// +/// * `set` - the responses to translate, in place. +/// * `decoder` - the form's mapping. +/// * `numbering` - how the export numbered questions. +/// +/// # Returns +/// +/// Warnings for anything that could not be translated. +pub fn apply(set: &mut ResponseSet, decoder: &FormDecoder, numbering: Numbering) -> Vec { + let mut warnings = Vec::new(); + let mut unmapped_positions: BTreeSet = BTreeSet::new(); + let mut unmapped_letters: BTreeSet = BTreeSet::new(); + let mut translated = 0usize; + + for row in &mut set.rows { + let belongs = row + .form + .as_deref() + .map(|f| f.eq_ignore_ascii_case(&decoder.form)) + .unwrap_or(false); + if !belongs { + continue; + } + + let Some(map) = decoder.lookup(row.item_number, numbering) else { + unmapped_positions.insert(row.item_number); + continue; + }; + + row.form_position = Some(map.position); + row.item_number = map.number; + + let mut convert = |letters: &[String]| -> Vec { + let mut out = Vec::with_capacity(letters.len()); + for letter in letters { + match map.canonical(letter) { + Some(canonical) => out.push(canonical.to_string()), + None => { + unmapped_letters.insert(format!("q{} {}", map.number, letter)); + } + } + } + out.sort(); + out + }; + + row.selected_source = convert(&row.selected); + row.eliminated_source = convert(&row.eliminated); + translated += 1; + } + + if !unmapped_positions.is_empty() { + let list: Vec = unmapped_positions.iter().map(|n| n.to_string()).collect(); + warnings.push(format!( + "form {}: question(s) {} are in the export but not on this form; they were left \ + untranslated", + decoder.form, + list.join(", ") + )); + } + if !unmapped_letters.is_empty() { + let list: Vec = unmapped_letters.iter().take(10).cloned().collect(); + warnings.push(format!( + "form {}: {} marked option(s) are not options on the printed form ({}), which usually \ + means a rubric column was added by hand in Gradescope", + decoder.form, + unmapped_letters.len(), + list.join(", ") + )); + } + if translated == 0 { + warnings.push(format!( + "form {}: no response rows carried this form id, so nothing was translated", + decoder.form + )); + } + + warnings +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::responses::{Response, administration_id}; + + fn map(position: u32, number: u32, pairs: &[(&str, &str)], key: &str) -> QuestionMap { + let mut to_canonical = BTreeMap::new(); + let mut to_printed = BTreeMap::new(); + for (printed, canonical) in pairs { + to_canonical.insert(printed.to_string(), canonical.to_string()); + to_printed.insert(canonical.to_string(), printed.to_string()); + } + let printed_key = vec![to_printed.get(key).cloned().unwrap_or_default()]; + QuestionMap { + position, + number, + item: format!("b::q-{number}"), + printed_key, + canonical_key: vec![key.to_string()], + to_canonical, + to_printed, + } + } + + fn decoder() -> FormDecoder { + FormDecoder::assemble( + "B".to_string(), + Provenance::Seal, + vec![ + // Printed A..D show bank C, A, D, B. The key is bank A, printed B. + map(1, 1, &[("A", "C"), ("B", "A"), ("C", "D"), ("D", "B")], "A"), + // A bonus item recorded as 9 but printed last, at position 2. + map(2, 9, &[("A", "B"), ("B", "A")], "B"), + ], + ) + } + + fn row(number: u32, selected: &str) -> Response { + Response { + administration_id: administration_id("C", "2026f", "e1"), + course: "C".into(), + term: "2026f".into(), + assessment_id: "e1".into(), + date: None, + form: Some("B".into()), + student_key: "s1".into(), + sid: None, + name: None, + email: None, + section: None, + item_number: number, + form_position: None, + item_ref: None, + item_version: None, + selected: vec![selected.into()], + eliminated: Vec::new(), + selected_source: Vec::new(), + eliminated_source: Vec::new(), + correct: Some(false), + credit: 0.0, + points_possible: 1.0, + score: 0.0, + response_time_seconds: None, + level: None, + learning_objectives: Vec::new(), + topics: Vec::new(), + bonus: false, + dropped: false, + } + } + + #[test] + fn printed_letters_become_bank_letters() { + let decoder = decoder(); + let q = decoder.at_position(1).unwrap(); + assert_eq!(q.canonical("A"), Some("C")); + assert_eq!(q.canonical("D"), Some("B")); + assert_eq!(q.canonical("E"), None); + assert_eq!(q.printed_key, vec!["B".to_string()]); + } + + #[test] + fn positions_become_recorded_numbers() { + let mut set = ResponseSet::new(); + set.rows.push(row(2, "A")); + let warnings = apply(&mut set, &decoder(), Numbering::Printed); + assert!(warnings.is_empty(), "{warnings:?}"); + assert_eq!(set.rows[0].item_number, 9, "position 2 is recorded as 9"); + assert_eq!(set.rows[0].form_position, Some(2)); + assert_eq!(set.rows[0].selected, vec!["A".to_string()], "printed kept"); + assert_eq!(set.rows[0].selected_source, vec!["B".to_string()]); + } + + #[test] + fn rows_from_another_form_are_left_alone() { + let mut set = ResponseSet::new(); + let mut other = row(1, "A"); + other.form = Some("A".into()); + set.rows.push(other); + apply(&mut set, &decoder(), Numbering::Printed); + assert!(set.rows[0].selected_source.is_empty()); + assert_eq!(set.rows[0].item_number, 1); + } + + #[test] + fn an_unknown_position_is_reported_not_guessed() { + let mut set = ResponseSet::new(); + set.rows.push(row(7, "A")); + let warnings = apply(&mut set, &decoder(), Numbering::Printed); + assert!( + warnings.iter().any(|w| w.contains("not on this form")), + "{warnings:?}" + ); + assert_eq!(set.rows[0].item_number, 7, "left as found"); + } + + #[test] + fn a_perfect_fit_says_nothing() { + let fits = vec![FormFit { + form: "A".into(), + matched: 30, + compared: 30, + mismatches: Vec::new(), + }]; + assert!(form_warning("A", &fits).is_none()); + } + + #[test] + fn a_swapped_directory_names_the_form_it_really_is() { + let fits = vec![ + FormFit { + form: "B".into(), + matched: 30, + compared: 30, + mismatches: Vec::new(), + }, + FormFit { + form: "A".into(), + matched: 8, + compared: 30, + mismatches: (1..=22).collect(), + }, + ]; + let warning = form_warning("A", &fits).expect("a mismatch this large must warn"); + assert!(warning.contains("almost certainly form B"), "{warning}"); + } + + #[test] + fn a_single_regraded_key_warns_without_accusing_the_wrong_form() { + let fits = vec![FormFit { + form: "A".into(), + matched: 29, + compared: 30, + mismatches: vec![14], + }]; + let warning = form_warning("A", &fits).unwrap(); + assert!(warning.contains("q14"), "{warning}"); + assert!(warning.contains("regraded"), "{warning}"); + } +} diff --git a/src/data/gradescope.rs b/src/data/gradescope.rs index 050739c..6f18c7a 100644 --- a/src/data/gradescope.rs +++ b/src/data/gradescope.rs @@ -633,6 +633,10 @@ pub fn to_responses(questions: &[Question], ctx: &Context) -> Import { assessment_id: ctx.assessment_id.clone(), date: ctx.date, form: ctx.form.clone(), + // The printed position and the bank's lettering are written + // later, by `decode::apply`, which is the only place that knows + // which form this directory holds. + form_position: None, student_key, sid: row.sid.clone(), name: row.name.clone(), @@ -642,7 +646,9 @@ pub fn to_responses(questions: &[Question], ctx: &Context) -> Import { item_ref: None, item_version: None, selected, + selected_source: Vec::new(), eliminated, + eliminated_source: Vec::new(), correct, credit, points_possible: points, diff --git a/src/data/intake.rs b/src/data/intake.rs new file mode 100644 index 0000000..385b93b --- /dev/null +++ b/src/data/intake.rs @@ -0,0 +1,565 @@ +// SPDX-License-Identifier: Prosperity-3.0.0 +// Copyright Scientific Computing Studio +// Source: https://git.scient.ing/education/coursebank + +//! Reading several forms of one exam back in at once. +//! +//! A two-form exam is two Gradescope assignments, each exporting its own +//! directory of `1.csv` through `N.csv`, each numbered against its own paper. They +//! are one administration: one set of students, one item pool, one set of +//! statistics. Ingesting them one command at a time does not work, because the +//! store keys on the administration and the second write replaces the first. +//! +//! So this module takes the whole set: +//! +//! ```text +//! coursebank ingest gradescope A=exports/e1-a B=exports/e1-b --assessment e1 +//! ``` +//! +//! and does four things the single-directory path cannot: +//! +//! *Checks each directory is the form it claims to be.* Every export carries the +//! graded key in its point-value row; every form knows what its printed key should +//! be. Comparing them catches a swapped pair of directories immediately rather +//! than three weeks later, when the distractor table looks strange. See +//! [`crate::decode::identify_form`]. +//! +//! *Translates each form into the bank's vocabulary before merging.* Otherwise +//! form A's option C and form B's option C land in the same column of the same +//! table while meaning different things. +//! +//! *Merges into one response set*, written once, so `analyze` and `report` see the +//! whole class. +//! +//! *Reports what the merge revealed*: a student who appears on two forms, a form +//! that is missing a question the other has, a form that ran materially harder +//! than the other. + +use std::collections::{BTreeMap, BTreeSet}; +use std::path::{Path, PathBuf}; + +use crate::assessment::{AssessmentFile, Form}; +use crate::catalog::Catalog; +use crate::date::Date; +use crate::decode::{self, FormDecoder, FormFit, Numbering}; +use crate::error::{Error, Result}; +use crate::gradescope::{self, Context, Question}; +use crate::responses::ResponseSet; +use crate::seal::SealFile; + +/// One directory of graded questions, and the form it holds. +#[derive(Debug, Clone)] +pub struct Source { + /// The form id. + pub form: String, + /// The directory of per-question CSV exports. + pub dir: PathBuf, +} + +impl Source { + /// Parses a `FORM=DIR` argument. + /// + /// A bare path is accepted and takes the fallback form, so the single-form + /// case stays as short as it was. + /// + /// # Arguments + /// + /// * `text` - the argument, e.g. `A=exports/e1-a` or `exports/e1`. + /// * `fallback` - the form to use when the argument names none. + /// + /// # Returns + /// + /// The source. + /// + /// # Errors + /// + /// Returns [`Error::Usage`] when the argument names no form and no fallback + /// was given, or when the form label is empty. + pub fn parse(text: &str, fallback: Option<&str>) -> Result { + // Split on the first `=` only: a directory name may contain one, a form + // label may not. + if let Some((form, dir)) = text.split_once('=') { + let form = form.trim(); + if form.is_empty() { + return Err(Error::usage(format!( + "`{text}` has an empty form label; write it as FORM=DIR, e.g. A=exports/e1-a" + ))); + } + if !dir.trim().is_empty() { + return Ok(Source { + form: form.to_string(), + dir: PathBuf::from(dir.trim()), + }); + } + } + match fallback { + Some(form) => Ok(Source { + form: form.to_string(), + dir: PathBuf::from(text.trim()), + }), + None => Err(Error::usage(format!( + "`{text}` does not say which form it holds; write it as FORM=DIR (e.g. \ + A=exports/e1-a) or pass --form" + ))), + } + } +} + +/// How to run an intake. +#[derive(Debug, Clone)] +pub struct Options { + /// The administration date, overriding the record's. + pub date: Option, + /// How the exports number their questions. + pub numbering: Numbering, + /// Whether a form that fails its key check stops the ingest. + /// + /// On by default. A directory that does not match the form it was named as is + /// the one ingest error that produces confident, wrong analysis rather than an + /// obvious failure, so the default is to refuse and say so. + pub strict: bool, +} + +impl Default for Options { + fn default() -> Options { + Options { + date: None, + numbering: Numbering::Printed, + strict: true, + } + } +} + +/// What one form's directory contributed. +#[derive(Debug, Clone)] +pub struct FormIntake { + /// The form id. + pub form: String, + /// The directory it came from. + pub dir: PathBuf, + /// Where its mapping came from. + pub provenance: decode::Provenance, + /// How many students it held. + pub students: usize, + /// How many questions it held. + pub questions: usize, + /// How the graded keys compared to every declared form, best first. + pub fits: Vec, + /// The parsed question files, kept so grading-time decisions stay available. + pub parsed: Vec, +} + +impl FormIntake { + /// This form's own fit. + pub fn own_fit(&self) -> Option<&FormFit> { + self.fits + .iter() + .find(|f| f.form.eq_ignore_ascii_case(&self.form)) + } +} + +/// The result of reading every form of one administration. +#[derive(Debug, Clone)] +pub struct Intake { + /// The merged, translated responses. + pub responses: ResponseSet, + /// Per-form detail. + pub forms: Vec, + /// Problems that did not stop the ingest. + pub warnings: Vec, +} + +impl Intake { + /// Every parsed question file, across forms. + /// + /// # Returns + /// + /// Pairs of form id and question. + pub fn questions(&self) -> Vec<(&str, &Question)> { + self.forms + .iter() + .flat_map(|f| f.parsed.iter().map(move |q| (f.form.as_str(), q))) + .collect() + } +} + +/// Reads every source into one response set. +/// +/// # Arguments +/// +/// * `catalog` - the loaded course. +/// * `record` - the assessment record. +/// * `seal` - the seal, when one was written. Strongly preferred: it describes the +/// paper that was printed rather than the paper the bank would print today. +/// * `sources` - the directories and the forms they hold. +/// * `opts` - how to run. +/// +/// # Returns +/// +/// The merged intake. +/// +/// # Errors +/// +/// Returns [`Error::Usage`] when a source names a form the record does not +/// declare, when two sources name the same form, or when a key check fails under +/// `strict`. Propagates parse errors from the exports themselves. +pub fn run( + catalog: &Catalog, + record: &AssessmentFile, + seal: Option<&SealFile>, + sources: &[Source], + opts: &Options, +) -> Result { + if sources.is_empty() { + return Err(Error::usage( + "no directories to ingest; pass one per form, e.g. A=exports/e1-a B=exports/e1-b" + .to_string(), + )); + } + + let declared = declared_forms(record); + let mut seen: BTreeSet = BTreeSet::new(); + for source in sources { + if !seen.insert(source.form.to_ascii_uppercase()) { + return Err(Error::usage(format!( + "form {} was given twice; each form is one directory", + source.form + ))); + } + if !declared + .iter() + .any(|f| f.id.eq_ignore_ascii_case(&source.form)) + { + return Err(Error::usage(format!( + "the record for `{}` declares no form `{}`; it declares {}", + record.assessment.id, + source.form, + declared + .iter() + .map(|f| f.id.as_str()) + .collect::>() + .join(", ") + ))); + } + } + + // One decoder per declared form, not just per ingested form: identifying a + // swapped directory means testing it against the forms it might be. + let mut decoders: Vec = Vec::new(); + for form in &declared { + decoders.push(FormDecoder::resolve(seal, catalog, record, form)?); + } + + let mut merged = ResponseSet::new(); + let mut forms = Vec::new(); + let mut warnings = Vec::new(); + let mut blocking = Vec::new(); + + for source in sources { + let decoder = decoders + .iter() + .find(|d| d.form.eq_ignore_ascii_case(&source.form)) + .expect("every source's form was checked against the declared list"); + + let ctx = Context { + course: catalog.course.course.code.clone(), + term: record + .assessment + .term + .clone() + .unwrap_or_else(|| catalog.course.course.term.clone()), + assessment_id: record.assessment.id.clone(), + date: opts.date.or(record.assessment.date), + form: Some(decoder.form.clone()), + }; + + let import = gradescope::ingest_dir(&source.dir, &ctx)?; + let mut set = import.responses; + + let fits = decode::identify_form(&import.questions, &decoders, opts.numbering); + if let Some(problem) = decode::form_warning(&decoder.form, &fits) { + let message = format!("{} [{}]", problem, source.dir.display()); + let convincing_alternative = fits + .iter() + .any(|f| !f.form.eq_ignore_ascii_case(&decoder.form) && f.is_convincing()); + if opts.strict && convincing_alternative { + blocking.push(message); + } else { + warnings.push(message); + } + } + + warnings.extend(decode::apply(&mut set, decoder, opts.numbering)); + + forms.push(FormIntake { + form: decoder.form.clone(), + dir: source.dir.clone(), + provenance: decoder.provenance, + students: set.students().len(), + questions: set.all_items().len(), + fits, + parsed: import.questions, + }); + + merged.absorb(set); + } + + if !blocking.is_empty() { + blocking.push( + "Nothing was written. Fix the form labels, or pass --allow-mismatch if the keys really \ + did change after printing." + .to_string(), + ); + return Err(Error::Invalid(blocking)); + } + + warnings.extend(cross_form_checks(&merged, &forms, record)); + merged.warnings.extend(warnings.clone()); + + Ok(Intake { + responses: merged, + forms, + warnings, + }) +} + +/// The forms a record declares, with the implicit single form for a record that +/// declares none. +fn declared_forms(record: &AssessmentFile) -> Vec
{ + if record.forms.is_empty() { + vec![Form { + id: "A".to_string(), + seed: 0, + shuffle_items: false, + shuffle_options: false, + }] + } else { + record.forms.clone() + } +} + +/// Checks that only merging several forms can make. +fn cross_form_checks( + merged: &ResponseSet, + forms: &[FormIntake], + record: &AssessmentFile, +) -> Vec { + let mut out = Vec::new(); + if forms.len() < 2 { + return out; + } + + // A student on two forms sat one exam and was graded twice, or two people + // share an identifier. Either way the response set now double counts them. + let mut by_student: BTreeMap<&str, BTreeSet<&str>> = BTreeMap::new(); + for row in &merged.rows { + if let Some(form) = row.form.as_deref() { + by_student + .entry(row.student_key.as_str()) + .or_default() + .insert(form); + } + } + let doubled: Vec<&str> = by_student + .iter() + .filter(|(_, forms)| forms.len() > 1) + .map(|(student, _)| *student) + .collect(); + if !doubled.is_empty() { + out.push(format!( + "{} student(s) appear on more than one form ({}); each is counted twice in every \ + total until one submission is removed", + doubled.len(), + doubled + .iter() + .take(5) + .copied() + .collect::>() + .join(", ") + )); + } + + // Every form is the same items in a different order, so a question present on + // one and absent from another means a directory is short a file. + let expected: BTreeSet = record + .items + .iter() + .filter(|p| !p.dropped) + .map(|p| p.number) + .collect(); + for form in forms { + let present: BTreeSet = merged + .rows + .iter() + .filter(|r| { + r.form + .as_deref() + .map(|f| f.eq_ignore_ascii_case(&form.form)) + .unwrap_or(false) + }) + .map(|r| r.item_number) + .collect(); + let missing: Vec = expected + .difference(&present) + .map(|n| n.to_string()) + .collect(); + if !missing.is_empty() { + out.push(format!( + "form {}: no responses for question(s) {}; the export directory is missing those \ + files", + form.form, + missing.join(", ") + )); + } + } + + // Forms are meant to be the same test. A large gap between their means is + // either a permutation that made one form easier or an uneven split of the + // class, and both are worth knowing before any grade is released. + let means: Vec<(String, f64, usize)> = forms + .iter() + .map(|f| { + let students: Vec<&str> = merged + .rows + .iter() + .filter(|r| { + r.form + .as_deref() + .map(|x| x.eq_ignore_ascii_case(&f.form)) + .unwrap_or(false) + }) + .map(|r| r.student_key.as_str()) + .collect::>() + .into_iter() + .collect(); + let possible = merged.points_available(); + let percents: Vec = students + .iter() + .map(|s| { + if possible > 0.0 { + 100.0 * merged.scored_total(s) / possible + } else { + 0.0 + } + }) + .collect(); + let mean = if percents.is_empty() { + 0.0 + } else { + percents.iter().sum::() / percents.len() as f64 + }; + (f.form.clone(), mean, percents.len()) + }) + .collect(); + + if let (Some(low), Some(high)) = ( + means + .iter() + .min_by(|a, b| a.1.partial_cmp(&b.1).unwrap_or(std::cmp::Ordering::Equal)), + means + .iter() + .max_by(|a, b| a.1.partial_cmp(&b.1).unwrap_or(std::cmp::Ordering::Equal)), + ) { + let gap = high.1 - low.1; + if gap >= 5.0 && low.2 >= 5 && high.2 >= 5 { + out.push(format!( + "form {} averaged {:.0}% and form {} averaged {:.0}%, a {:.0}-point gap. With {} \ + and {} students that may be the split rather than the forms, but it is worth \ + looking at before releasing grades", + high.0, high.1, low.0, low.1, gap, high.2, low.2 + )); + } + } + + out +} + +/// A one-line summary of where a directory came from and what it held. +/// +/// # Arguments +/// +/// * `intake` - one form's intake. +/// +/// # Returns +/// +/// The line, without a trailing newline. +pub fn describe(intake: &FormIntake) -> String { + let fit = match intake.own_fit() { + Some(fit) if fit.compared > 0 => { + format!(", keys matched {}/{}", fit.matched, fit.compared) + } + _ => String::new(), + }; + format!( + "form {}: {} student(s) x {} question(s) from {} (mapping {}{fit})", + intake.form, + intake.students, + intake.questions, + intake.dir.display(), + intake.provenance.label(), + ) +} + +/// Whether a path looks like a Gradescope per-question export directory. +/// +/// Used to give a better error than "no numbered CSV files" when someone points +/// at the zip they downloaded, or at the directory above the one they meant. +/// +/// # Arguments +/// +/// * `dir` - the candidate directory. +/// +/// # Returns +/// +/// `true` when it holds at least one file named like `1.csv`. +pub fn looks_like_export(dir: &Path) -> bool { + let Ok(entries) = std::fs::read_dir(dir) else { + return false; + }; + entries.filter_map(|e| e.ok()).any(|entry| { + let path = entry.path(); + path.extension().and_then(|e| e.to_str()) == Some("csv") + && path + .file_stem() + .and_then(|s| s.to_str()) + .map(|s| s.chars().all(|c| c.is_ascii_digit())) + .unwrap_or(false) + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_labelled_source_parses() { + let source = Source::parse("B=exports/e1-b", None).unwrap(); + assert_eq!(source.form, "B"); + assert_eq!(source.dir, PathBuf::from("exports/e1-b")); + } + + #[test] + fn a_bare_path_needs_a_fallback_form() { + let source = Source::parse("exports/e1", Some("A")).unwrap(); + assert_eq!(source.form, "A"); + assert_eq!(source.dir, PathBuf::from("exports/e1")); + + let err = Source::parse("exports/e1", None).unwrap_err(); + assert!(err.to_string().contains("which form"), "{err}"); + } + + #[test] + fn a_windows_style_path_is_not_mistaken_for_a_label() { + // The split is on the first `=`, and a drive letter has none, so this is + // only a hazard for a path that genuinely contains one. + let source = Source::parse("A=C:/exports/e1-a", None).unwrap(); + assert_eq!(source.form, "A"); + assert_eq!(source.dir, PathBuf::from("C:/exports/e1-a")); + } + + #[test] + fn an_empty_label_is_refused() { + let err = Source::parse("=exports/e1", None).unwrap_err(); + assert!(err.to_string().contains("empty form label"), "{err}"); + } +} diff --git a/src/data/responses.rs b/src/data/responses.rs index 12f4a0e..d7dd79a 100644 --- a/src/data/responses.rs +++ b/src/data/responses.rs @@ -52,6 +52,9 @@ pub struct Response { pub date: Option, /// Which form the student took, when forms were used. pub form: Option, + /// Where this question sat on the student's own form, when that differs from + /// the recorded number. Provenance, not a join key. + pub form_position: Option, /// The identifier analysis groups by. A pseudonym when pseudonymizing. pub student_key: String, @@ -74,8 +77,14 @@ pub struct Response { /// Option letters the student chose. pub selected: Vec, + /// The selected options in the bank's own lettering, written at ingest by + /// [`crate::decode::apply`]. Empty when the form was never decoded, which is + /// the case for data ingested before seals existed. + pub selected_source: Vec, /// Option letters the student eliminated, for elimination-scored items. pub eliminated: Vec, + /// The eliminated options in the bank's lettering. + pub eliminated_source: Vec, /// Whether the response earned full credit. `None` when it cannot be /// determined, e.g. a blank response on an item with no recorded key. pub correct: Option, @@ -132,6 +141,19 @@ impl Response { pub fn selected_joined(&self) -> String { self.selected.join(",") } + + /// The choices, in the bank's lettering when it is known. + /// + /// `selected` is what the student marked on the page they held; this is what + /// they chose. On an unshuffled form the two agree, which is why reading the + /// wrong one is a bug that only appears once you add a second form. + pub fn chosen(&self) -> &[String] { + if self.selected_source.is_empty() { + &self.selected + } else { + &self.selected_source + } + } } /// A set of responses plus anything worth telling the user about the ingest. @@ -393,12 +415,23 @@ impl ResponseSet { // Apply the record's credit overrides, which is how a decision to // award partial credit after the fact becomes visible in analysis. - if !p.credit_overrides.is_empty() && r.selected.len() == 1 { - if let Some(over) = p.credit_overrides.get(&r.selected[0]) { - if (*over - r.credit).abs() > 1e-9 { - r.credit = *over; - r.score = *over * r.points_possible; - r.correct = Some(*over >= 0.999); + // + // The override map is keyed in the bank's letters, so the lookup has + // to use `chosen()` rather than `selected`. Cloning the letter first + // ends the borrow of `r` before the row is written to. + let chosen = if r.chosen().len() == 1 { + r.chosen().first().cloned() + } else { + None + }; + if !p.credit_overrides.is_empty() { + if let Some(letter) = chosen { + if let Some(over) = p.credit_overrides.get(&letter) { + if (*over - r.credit).abs() > 1e-9 { + r.credit = *over; + r.score = *over * r.points_possible; + r.correct = Some(*over >= 0.999); + } } } } @@ -590,6 +623,19 @@ pub struct FlatResponse { pub bonus: bool, /// Whether the item was dropped. pub dropped: bool, + + // Appended rather than interleaved: the columns above are the order every + // CSV written so far uses, and `#[serde(default)]` is what keeps one written + // last term readable now that three more exist. + /// The printed position, 0 when unknown. + #[serde(default)] + pub form_position: u32, + /// Comma-joined selected letters in the bank's lettering. + #[serde(default)] + pub selected_source: String, + /// Comma-joined eliminated letters in the bank's lettering. + #[serde(default)] + pub eliminated_source: String, } impl FlatResponse { @@ -610,6 +656,7 @@ impl FlatResponse { assessment_id: r.assessment_id.clone(), date: r.date.map(|d| d.to_string()).unwrap_or_default(), form: r.form.clone().unwrap_or_default(), + form_position: r.form_position.unwrap_or(0), student_key: r.student_key.clone(), sid: r.sid.clone().unwrap_or_default(), email: r.email.clone().unwrap_or_default(), @@ -618,7 +665,9 @@ impl FlatResponse { item_ref: r.item_ref.clone().unwrap_or_default(), item_version: r.item_version.unwrap_or(0), selected: r.selected.join(","), + selected_source: r.selected_source.join(","), eliminated: r.eliminated.join(","), + eliminated_source: r.eliminated_source.join(","), correct: match r.correct { Some(true) => "1".to_string(), Some(false) => "0".to_string(), @@ -660,6 +709,11 @@ impl FlatResponse { assessment_id: self.assessment_id.clone(), date: self.date.parse().ok(), form: none_if_empty(&self.form), + form_position: if self.form_position == 0 { + None + } else { + Some(self.form_position) + }, student_key: self.student_key.clone(), sid: none_if_empty(&self.sid), name: None, @@ -673,7 +727,9 @@ impl FlatResponse { Some(self.item_version) }, selected: split(&self.selected), + selected_source: split(&self.selected_source), eliminated: split(&self.eliminated), + eliminated_source: split(&self.eliminated_source), correct: match self.correct.as_str() { "1" | "true" => Some(true), "0" | "false" => Some(false), @@ -713,6 +769,7 @@ mod tests { assessment_id: "e1".into(), date: None, form: None, + form_position: None, student_key: student.into(), sid: Some(format!("sid-{student}")), name: None, @@ -722,7 +779,9 @@ mod tests { item_ref: None, item_version: None, selected: vec!["A".into()], + selected_source: vec![], eliminated: vec![], + eliminated_source: vec![], correct: Some(credit >= 0.999), credit, points_possible: 2.0, diff --git a/src/data/store.rs b/src/data/store.rs index 51cb41d..10d4028 100644 --- a/src/data/store.rs +++ b/src/data/store.rs @@ -541,6 +541,7 @@ mod tests { assessment_id: "exam-4".into(), date: None, form: None, + form_position: None, student_key: student.into(), sid: None, name: None, @@ -550,7 +551,9 @@ mod tests { item_ref: Some("bank::q-x-001".into()), item_version: Some(2), selected: vec!["C".into()], + selected_source: vec![], eliminated: vec![], + eliminated_source: vec![], correct: Some(true), credit: 1.0, points_possible: 1.5, diff --git a/src/data/store_parquet.rs b/src/data/store_parquet.rs index 32143fc..b521325 100644 --- a/src/data/store_parquet.rs +++ b/src/data/store_parquet.rs @@ -61,6 +61,9 @@ pub fn schema() -> Schema { Field::new("topics", DataType::Utf8, false), Field::new("bonus", DataType::Boolean, false), Field::new("dropped", DataType::Boolean, false), + Field::new("form_position", DataType::UInt32, false), + Field::new("selected_source", DataType::Utf8, false), + Field::new("eliminated_source", DataType::Utf8, false), ]) } @@ -120,6 +123,9 @@ fn to_batch(rows: &[FlatResponse]) -> Result { s(|r| &r.topics), boolc(|r| r.bonus), boolc(|r| r.dropped), + u32c(|r| r.form_position), + s(|r| &r.selected_source), + s(|r| &r.eliminated_source), ]; RecordBatch::try_new(Arc::new(schema()), columns).map_err(|e| { @@ -237,6 +243,20 @@ fn from_batch(batch: &RecordBatch, path: &Path) -> Result> { .ok_or_else(|| column_error(name, "boolean", path)) }; + // The three decode columns are read optionally rather than required, so a + // file written before seals existed still loads. A missing column is not a + // foreign file; it is last term's data. + let optional_strings = |name: &str| -> Option<&StringArray> { + batch + .column_by_name(name) + .and_then(|c| c.as_any().downcast_ref::()) + }; + let optional_uints = |name: &str| -> Option<&UInt32Array> { + batch + .column_by_name(name) + .and_then(|c| c.as_any().downcast_ref::()) + }; + let administration_id = strings("administration_id")?; let course = strings("course")?; let term = strings("term")?; @@ -263,6 +283,10 @@ fn from_batch(batch: &RecordBatch, path: &Path) -> Result> { let bonus = bools("bonus")?; let dropped = bools("dropped")?; + let form_position = optional_uints("form_position"); + let selected_source = optional_strings("selected_source"); + let eliminated_source = optional_strings("eliminated_source"); + let mut out = Vec::with_capacity(batch.num_rows()); for i in 0..batch.num_rows() { out.push(FlatResponse { @@ -272,6 +296,7 @@ fn from_batch(batch: &RecordBatch, path: &Path) -> Result> { assessment_id: assessment_id.value(i).to_string(), date: date.value(i).to_string(), form: form.value(i).to_string(), + form_position: form_position.map(|a| a.value(i)).unwrap_or(0), student_key: student_key.value(i).to_string(), sid: sid.value(i).to_string(), email: email.value(i).to_string(), @@ -280,7 +305,13 @@ fn from_batch(batch: &RecordBatch, path: &Path) -> Result> { item_ref: item_ref.value(i).to_string(), item_version: item_version.value(i), selected: selected.value(i).to_string(), + selected_source: selected_source + .map(|a| a.value(i).to_string()) + .unwrap_or_default(), eliminated: eliminated.value(i).to_string(), + eliminated_source: eliminated_source + .map(|a| a.value(i).to_string()) + .unwrap_or_default(), correct: correct.value(i).to_string(), credit: credit.value(i), points_possible: points_possible.value(i), @@ -336,6 +367,9 @@ mod tests { topics: "kinetics".into(), bonus: false, dropped: false, + form_position: 0, + selected_source: String::new(), + eliminated_source: String::new(), } } diff --git a/src/export/typst.rs b/src/export/typst.rs index bb7fd28..a31b87a 100644 --- a/src/export/typst.rs +++ b/src/export/typst.rs @@ -50,6 +50,7 @@ //! the template by someone who has not read this comment. pub mod config; +pub mod diagnostic; pub mod payload; pub mod template; pub mod value; diff --git a/src/export/typst/config.rs b/src/export/typst/config.rs index 59aa16c..c99e5f8 100644 --- a/src/export/typst/config.rs +++ b/src/export/typst/config.rs @@ -43,11 +43,25 @@ pub enum Variant { Key, /// A bubble sheet matching the form. AnswerSheet, + /// One student's diagnostic, which carries no questions. + StudentReport, + /// The instructor's class diagnostic. + CohortReport, } impl Variant { /// Every variant, in the order `export` writes them. - pub const ALL: [Variant; 3] = [Variant::Exam, Variant::Key, Variant::AnswerSheet]; + pub const ALL: [Variant; 5] = [ + Variant::Exam, + Variant::Key, + Variant::AnswerSheet, + Variant::StudentReport, + Variant::CohortReport, + ]; + + /// The variants `export typst` produces. A report is not built from an + /// assessment record alone, so `export typst` must not default to it. + pub const EXAM: [Variant; 3] = [Variant::Exam, Variant::Key, Variant::AnswerSheet]; /// The token used on the command line, in config keys, and in file names. pub fn as_str(self) -> &'static str { @@ -55,6 +69,8 @@ impl Variant { Variant::Exam => "exam", Variant::Key => "key", Variant::AnswerSheet => "answer-sheet", + Variant::StudentReport => "student-report", + Variant::CohortReport => "cohort-report", } } @@ -103,6 +119,8 @@ impl Variant { Variant::Exam => "", Variant::Key => "-key", Variant::AnswerSheet => "-answer-sheet", + Variant::StudentReport => "-student", + Variant::CohortReport => "-cohort", } } } @@ -422,6 +440,30 @@ impl RenderConfig { calibration: false, }; } + Variant::StudentReport => { + // Belt and braces. The student payload is built by + // `typst::diagnostic`, which has no question text to reveal in the + // first place; this says so in the one place someone would look. + config.reveal = Reveal::Nothing; + config.stimulus = StimulusMode::Omit; + config.fields = Fields { + uid: false, + title: false, + points: true, + level: true, + objectives: true, + topics: false, + assets: false, + source_letters: false, + design: false, + calibration: false, + }; + } + Variant::CohortReport => { + config.reveal = Reveal::Everything; + config.stimulus = StimulusMode::Omit; + config.fields.calibration = true; + } } config } diff --git a/src/export/typst/diagnostic.rs b/src/export/typst/diagnostic.rs new file mode 100644 index 0000000..6dc5793 --- /dev/null +++ b/src/export/typst/diagnostic.rs @@ -0,0 +1,904 @@ +// SPDX-License-Identifier: Prosperity-3.0.0 +// Copyright Scientific Computing Studio +// Source: https://git.scient.ing/education/coursebank + +//! Handing a diagnostic to a Typst template. +//! +//! The same arrangement the exam export uses: this module decides what the +//! template is told, the template decides what it looks like, and the two meet at +//! a marker comment. Nothing here knows about page size or colour. +//! +//! Two slots are filled, both already defined for the exam path: +//! +//! | Slot | Injected | +//! |:--|:--| +//! | `meta` | `#let cb-meta = (...)` — course, assessment, generator, class size | +//! | `data` | `#let cb-data = (...)` — one student's diagnostic, or the class's | +//! +//! The `questions` slot is deliberately unused. It exists to emit question stems, +//! and a diagnostic has none to emit: see +//! [`crate::diagnostic::StudentDiagnostic`], which has no field for one. A student +//! report template that wanted to print a stem would have nothing to print it +//! from, which is the property worth preserving. +//! +//! Prose fields — objective text, the feedback written for a chosen option, a +//! reading's focus sentence — go through the same markup path as an exam stem, so +//! `$\Delta G$` in `course.yaml` renders the same way in a report as it does on +//! the paper. + +use crate::assessment::AssessmentFile; +use crate::catalog::Catalog; +use crate::diagnostic::{ + Bin, CohortDiagnostic, CohortObjectiveRow, CohortQuestionRow, StudentDiagnostic, StudyGroup, +}; +use crate::error::{Error, Result}; +use crate::layout::Layout; +use crate::typst::config::{RenderConfig, Variant}; +use crate::typst::payload::markup_value; +use crate::typst::template::{self, Origin, Slot}; +use crate::typst::value::Value; + +/// A rendered diagnostic document. +#[derive(Debug, Clone)] +pub struct Document { + /// Which document this is. + pub variant: Variant, + /// Where the template came from. + pub origin: Origin, + /// Which slots it declared. + pub slots: Vec, + /// The Typst source. + pub text: String, + /// Advisory problems. + pub warnings: Vec, +} + +/// The identity block every diagnostic carries. +#[derive(Debug, Clone)] +pub struct Meta { + /// Course code. + pub course_code: String, + /// Course title. + pub course_title: String, + /// The term. + pub term: String, + /// Institution, when the course names one. + pub institution: Option, + /// Assessment id. + pub assessment_id: String, + /// Assessment title. + pub assessment_title: String, + /// The administration date, as `YYYY-MM-DD`. + pub date: Option, + /// The day the report was generated. + pub generated_on: String, + /// The tool version, so a report found later can be traced. + pub version: String, + /// How many students sat the assessment. + pub n_students: usize, + /// The course's mastery threshold, so a template can draw the line in the + /// same place the classification used. + pub mastery_threshold: f64, + /// How many items an objective needs before it is classified at all. + pub min_items_for_mastery: usize, +} + +impl Meta { + /// Builds the identity block. + /// + /// # Arguments + /// + /// * `catalog` - the loaded course. + /// * `record` - the assessment record. + /// * `n_students` - the cohort size. + /// + /// # Returns + /// + /// The block. + pub fn new(catalog: &Catalog, record: &AssessmentFile, n_students: usize) -> Meta { + Meta { + course_code: catalog.course.course.code.clone(), + course_title: catalog.course.course.title.clone(), + term: record + .assessment + .term + .clone() + .unwrap_or_else(|| catalog.course.course.term.clone()), + institution: catalog.course.course.institution.clone(), + assessment_id: record.assessment.id.clone(), + assessment_title: record.assessment.title.clone(), + date: record.assessment.date.map(|d| d.to_string()), + generated_on: crate::date::Date::today().to_string(), + version: crate::VERSION.to_string(), + n_students, + mastery_threshold: catalog.course.policy.mastery_threshold, + min_items_for_mastery: catalog.course.policy.min_items_for_mastery, + } + } + + /// The metadata as a Typst value. + fn value(&self, config: &RenderConfig) -> Value { + let mut course = Value::dict(); + course.insert("code", Value::str(&self.course_code)); + course.insert("title", Value::str(&self.course_title)); + course.insert("term", Value::str(&self.term)); + if let Some(institution) = &self.institution { + course.insert("institution", Value::str(institution)); + } + + let mut assessment = Value::dict(); + assessment.insert("id", Value::str(&self.assessment_id)); + assessment.insert("title", Value::str(&self.assessment_title)); + assessment.insert_some("date", self.date.as_ref().map(Value::str)); + + let mut generator = Value::dict(); + generator.insert("tool", Value::str("coursebank")); + generator.insert("version", Value::str(&self.version)); + generator.insert("on", Value::str(&self.generated_on)); + + let mut policy = Value::dict(); + policy.insert("mastery-threshold", Value::Float(self.mastery_threshold)); + policy.insert( + "min-items-for-mastery", + Value::Int(self.min_items_for_mastery as i64), + ); + + let mut out = Value::dict(); + out.insert("course", course); + out.insert("assessment", assessment); + out.insert("generator", generator); + out.insert("policy", policy); + out.insert("class-size", Value::Int(self.n_students as i64)); + out.insert("extra", extra_value(config)); + out + } +} + +/// The render config's `extra` block, carried through untouched. +fn extra_value(config: &RenderConfig) -> Value { + let mut out = Value::dict(); + for (key, value) in &config.extra { + out.insert(key.clone(), crate::typst::value::from_yaml(value, false)); + } + out +} + +/// One student's diagnostic as a Typst value. +/// +/// # Arguments +/// +/// * `diagnostic` - the assembled diagnostic. +/// * `config` - the render config, for markup handling. +/// +/// # Returns +/// +/// A dictionary the template binds as `cb-data`. +pub fn student_value(diagnostic: &StudentDiagnostic, config: &RenderConfig) -> Value { + let content = config.content.is_content(); + let mut out = Value::dict(); + + out.insert("student-key", Value::str(&diagnostic.student_key)); + out.insert_some("name", diagnostic.name.as_ref().map(Value::str)); + out.insert_some("sid", diagnostic.sid.as_ref().map(Value::str)); + out.insert_some("form", diagnostic.form.as_ref().map(Value::str)); + + let mut score = Value::dict(); + score.insert("points", Value::Float(diagnostic.score.points)); + score.insert("possible", Value::Float(diagnostic.score.points_possible)); + score.insert("percent", Value::Float(diagnostic.score.percent)); + score.insert("bonus", Value::Float(diagnostic.score.bonus_points)); + score.insert("correct", Value::Int(diagnostic.score.correct as i64)); + score.insert("items", Value::Int(diagnostic.score.n_items as i64)); + out.insert("score", score); + + if let Some(standing) = &diagnostic.standing { + let mut value = Value::dict(); + value.insert("class-mean", Value::Float(standing.class_mean)); + value.insert("class-sd", Value::Float(standing.class_sd)); + value.insert("band", Value::str(&standing.band)); + value.insert_some("theta", standing.theta.map(Value::Float)); + value.insert_some("theta-se", standing.theta_se.map(Value::Float)); + out.insert("standing", value); + } + + out.insert( + "levels", + Value::Array( + diagnostic + .levels + .iter() + .map(|level| { + let mut value = Value::dict(); + value.insert("level", Value::Int(level.level as i64)); + value.insert("name", Value::str(&level.name)); + value.insert("blurb", Value::str(&level.blurb)); + value.insert("items", Value::Int(level.n_items as i64)); + value.insert("rate", Value::Float(level.rate)); + value.insert_some("class-rate", level.class_rate.map(Value::Float)); + value.insert_some("comparison", level.comparison.as_ref().map(Value::str)); + value + }) + .collect(), + ), + ); + + out.insert( + "objectives", + Value::Array( + diagnostic + .objectives + .iter() + .map(|objective| { + let mut value = Value::dict(); + value.insert("id", Value::str(&objective.id)); + value.insert("text", markup_value(&objective.text, content)); + value.insert_some("unit", objective.unit.as_ref().map(Value::str)); + value.insert("items", Value::Int(objective.n_items as i64)); + value.insert("credit", Value::Float(objective.credit)); + value.insert("rate", Value::Float(objective.rate)); + value.insert("lower", Value::Float(objective.lower)); + value.insert("upper", Value::Float(objective.upper)); + value.insert_some("class-rate", objective.class_rate.map(Value::Float)); + value.insert("status", Value::str(&objective.status)); + value.insert("symbol", Value::str(&objective.symbol)); + value.insert("confident", Value::Bool(objective.confident)); + value.insert("thin-evidence", Value::Bool(objective.thin_evidence)); + value.insert( + "levels", + Value::Array( + objective + .levels + .iter() + .map(|l| Value::Int(*l as i64)) + .collect(), + ), + ); + value + }) + .collect(), + ), + ); + + let objective_refs = |rows: &[crate::diagnostic::ObjectiveRef]| -> Value { + Value::Array( + rows.iter() + .map(|row| { + let mut value = Value::dict(); + value.insert("id", Value::str(&row.id)); + value.insert("text", markup_value(&row.text, content)); + value.insert("rate", Value::Float(row.rate)); + value.insert("items", Value::Int(row.n_items as i64)); + value + }) + .collect(), + ) + }; + out.insert("strengths", objective_refs(&diagnostic.strengths)); + out.insert("focus", objective_refs(&diagnostic.focus)); + + out.insert( + "questions", + Value::Array( + diagnostic + .questions + .iter() + .map(|question| { + let mut value = Value::dict(); + value.insert("number", Value::Int(question.number as i64)); + value.insert_some("position", question.position.map(|p| Value::Int(p as i64))); + value.insert_some("level", question.level.map(|l| Value::Int(l as i64))); + value.insert( + "objectives", + Value::Array(question.objectives.iter().map(|o| Value::str(o)).collect()), + ); + value.insert_some("correct", question.correct.map(Value::Bool)); + value.insert("credit", Value::Float(question.credit)); + value.insert("bonus", Value::Bool(question.bonus)); + value.insert("blank", Value::Bool(question.blank)); + value.insert_some("class-rate", question.class_rate.map(Value::Float)); + value.insert_some( + "feedback", + question + .feedback + .as_ref() + .map(|text| markup_value(text, content)), + ); + value.insert( + "taught-in", + Value::Array(question.taught_in.iter().map(|s| Value::str(s)).collect()), + ); + value + }) + .collect(), + ), + ); + + out.insert( + "study", + Value::Array( + diagnostic + .study + .iter() + .map(|group| study_value(group, content)) + .collect(), + ), + ); + + out +} + +/// One study group as a Typst value. +fn study_value(group: &StudyGroup, content: bool) -> Value { + let mut out = Value::dict(); + out.insert("objective", Value::str(&group.objective)); + out.insert("text", markup_value(&group.text, content)); + out.insert("rate", Value::Float(group.rate)); + out.insert( + "readings", + Value::Array( + group + .readings + .iter() + .map(|reading| { + let mut value = Value::dict(); + value.insert("citation", Value::str(&reading.citation)); + value.insert("lecture", Value::str(&reading.lecture)); + value.insert("lecture-title", Value::str(&reading.lecture_title)); + value.insert_some("url", reading.url.as_ref().map(Value::str)); + value.insert_some( + "focus", + reading.focus.as_ref().map(|t| markup_value(t, content)), + ); + value.insert_some( + "summary", + reading.summary.as_ref().map(|t| markup_value(t, content)), + ); + value.insert("supplemental", Value::Bool(reading.supplemental)); + value + }) + .collect(), + ), + ); + out +} + +/// The class diagnostic as a Typst value. +/// +/// # Arguments +/// +/// * `diagnostic` - the assembled diagnostic. +/// * `config` - the render config, for markup handling. +/// +/// # Returns +/// +/// A dictionary the template binds as `cb-data`. +pub fn cohort_value(diagnostic: &CohortDiagnostic, config: &RenderConfig) -> Value { + let content = config.content.is_content(); + let mut out = Value::dict(); + + out.insert("students", Value::Int(diagnostic.n_students as i64)); + out.insert("items", Value::Int(diagnostic.n_items as i64)); + + let mut distribution = Value::dict(); + distribution.insert("mean", Value::Float(diagnostic.distribution.mean)); + distribution.insert("median", Value::Float(diagnostic.distribution.median)); + distribution.insert("sd", Value::Float(diagnostic.distribution.sd)); + distribution.insert("min", Value::Float(diagnostic.distribution.min)); + distribution.insert("max", Value::Float(diagnostic.distribution.max)); + distribution.insert( + "bins", + Value::Array(diagnostic.distribution.bins.iter().map(bin_value).collect()), + ); + out.insert("distribution", distribution); + + let mut reliability = Value::dict(); + reliability.insert_some("alpha", diagnostic.reliability.alpha.map(Value::Float)); + reliability.insert_some("sem", diagnostic.reliability.sem.map(Value::Float)); + reliability.insert("mean-p", Value::Float(diagnostic.reliability.mean_p)); + reliability.insert_some( + "mean-point-biserial", + diagnostic.reliability.mean_point_biserial.map(Value::Float), + ); + reliability.insert( + "interpretation", + Value::str(&diagnostic.reliability.interpretation), + ); + out.insert("reliability", reliability); + + out.insert( + "levels", + Value::Array( + diagnostic + .levels + .iter() + .map(|level| { + let mut value = Value::dict(); + value.insert("level", Value::Int(level.level as i64)); + value.insert("name", Value::str(&level.name)); + value.insert("items", Value::Int(level.n_items as i64)); + value.insert("rate", Value::Float(level.rate)); + value + }) + .collect(), + ), + ); + + out.insert( + "objectives", + Value::Array( + diagnostic + .objectives + .iter() + .map(|o| cohort_objective_value(o, content)) + .collect(), + ), + ); + out.insert( + "gaps", + Value::Array( + diagnostic + .gaps + .iter() + .map(|o| cohort_objective_value(o, content)) + .collect(), + ), + ); + + out.insert( + "questions", + Value::Array( + diagnostic + .questions + .iter() + .map(|q| cohort_question_value(q, content)) + .collect(), + ), + ); + out.insert( + "revise", + Value::Array( + diagnostic + .revise + .iter() + .map(|q| cohort_question_value(q, content)) + .collect(), + ), + ); + + out.insert( + "forms", + Value::Array( + diagnostic + .forms + .iter() + .map(|form| { + let mut value = Value::dict(); + value.insert("id", Value::str(&form.id)); + value.insert("students", Value::Int(form.n_students as i64)); + value.insert("mean", Value::Float(form.mean)); + value.insert("sd", Value::Float(form.sd)); + value + }) + .collect(), + ), + ); + + out.insert( + "blueprint", + Value::Array(diagnostic.blueprint.iter().map(|s| Value::str(s)).collect()), + ); + + out.insert( + "patterns", + Value::Array( + diagnostic + .patterns + .iter() + .map(|pattern| { + let mut value = Value::dict(); + value.insert("label", Value::str(&pattern.label)); + value.insert("students", Value::Int(pattern.n_students as i64)); + let mut means = Value::dict(); + for (level, mean) in &pattern.level_means { + means.insert(format!("l{level}"), Value::Float(*mean)); + } + value.insert("level-means", means); + value + }) + .collect(), + ), + ); + + out.insert( + "warnings", + Value::Array(diagnostic.warnings.iter().map(|s| Value::str(s)).collect()), + ); + + out +} + +/// One histogram bin as a Typst value. +fn bin_value(bin: &Bin) -> Value { + let mut value = Value::dict(); + value.insert("low", Value::Int(bin.low as i64)); + value.insert("high", Value::Int(bin.high as i64)); + value.insert("count", Value::Int(bin.count as i64)); + value +} + +/// One class objective row as a Typst value. +fn cohort_objective_value(objective: &CohortObjectiveRow, content: bool) -> Value { + let mut value = Value::dict(); + value.insert("id", Value::str(&objective.id)); + value.insert("text", markup_value(&objective.text, content)); + value.insert("items", Value::Int(objective.n_items as i64)); + value.insert("rate", Value::Float(objective.rate)); + value.insert("meeting", Value::Int(objective.meeting as i64)); + value.insert("developing", Value::Int(objective.developing as i64)); + value.insert("not-yet", Value::Int(objective.not_yet as i64)); + value.insert("thin", Value::Int(objective.thin as i64)); + value.insert("below-threshold", Value::Bool(objective.below_threshold)); + value +} + +/// One class question row as a Typst value. +fn cohort_question_value(question: &CohortQuestionRow, content: bool) -> Value { + let mut value = Value::dict(); + value.insert("number", Value::Int(question.number as i64)); + value.insert_some("item", question.item.as_ref().map(Value::str)); + value.insert_some("level", question.level.map(|l| Value::Int(l as i64))); + value.insert( + "objectives", + Value::Array(question.objectives.iter().map(|o| Value::str(o)).collect()), + ); + value.insert("p", Value::Float(question.p_value)); + value.insert_some("point-biserial", question.point_biserial.map(Value::Float)); + value.insert_some("discrimination", question.discrimination.map(Value::Float)); + value.insert("blank-rate", Value::Float(question.blank_rate)); + value.insert( + "key", + Value::Array(question.key.iter().map(|k| Value::str(k)).collect()), + ); + value.insert( + "options", + Value::Array( + question + .options + .iter() + .map(|option| { + let mut value = Value::dict(); + value.insert("letter", Value::str(&option.letter)); + value.insert("count", Value::Int(option.count as i64)); + value.insert("rate", Value::Float(option.rate)); + value.insert("is-key", Value::Bool(option.is_key)); + value.insert_some("point-biserial", option.point_biserial.map(Value::Float)); + value.insert("nonfunctioning", Value::Bool(option.nonfunctioning)); + value + }) + .collect(), + ), + ); + value.insert( + "flags", + Value::Array(question.flags.iter().map(|f| Value::str(f)).collect()), + ); + value.insert( + "notes", + Value::Array( + question + .notes + .iter() + .map(|n| markup_value(n, content)) + .collect(), + ), + ); + let mut by_form = Value::dict(); + for (form, p) in &question.by_form { + by_form.insert(form.clone(), Value::Float(*p)); + } + value.insert("by-form", by_form); + value +} + +/// Emits a `#let` binding for a slot. +fn binding(name: &str, value: &Value) -> String { + format!("#let {name} = {}\n", value.to_typst(0)) +} + +/// Renders one student's report. +/// +/// # Arguments +/// +/// * `layout` - the course layout, for the template lookup. +/// * `meta` - the identity block. +/// * `diagnostic` - the student's diagnostic. +/// * `config` - the render config for this variant. +/// * `explicit` - a template path overriding the lookup. +/// +/// # Returns +/// +/// The rendered document. +/// +/// # Errors +/// +/// Returns [`Error::Io`] when an explicit template cannot be read and +/// [`Error::Invalid`] when a template's markers are malformed. +pub fn render_student( + layout: &Layout, + meta: &Meta, + diagnostic: &StudentDiagnostic, + config: &RenderConfig, + explicit: Option<&std::path::Path>, +) -> Result { + render( + layout, + Variant::StudentReport, + &meta.assessment_id, + meta, + student_value(diagnostic, config), + config, + explicit, + ) +} + +/// Renders the class report. +/// +/// # Arguments +/// +/// * `layout` - the course layout, for the template lookup. +/// * `meta` - the identity block. +/// * `diagnostic` - the class diagnostic. +/// * `config` - the render config for this variant. +/// * `explicit` - a template path overriding the lookup. +/// +/// # Returns +/// +/// The rendered document. +/// +/// # Errors +/// +/// As [`render_student`]. +pub fn render_cohort( + layout: &Layout, + meta: &Meta, + diagnostic: &CohortDiagnostic, + config: &RenderConfig, + explicit: Option<&std::path::Path>, +) -> Result { + render( + layout, + Variant::CohortReport, + &meta.assessment_id, + meta, + cohort_value(diagnostic, config), + config, + explicit, + ) +} + +/// The shared rendering path. +#[allow(clippy::too_many_arguments)] +fn render( + layout: &Layout, + variant: Variant, + assessment_id: &str, + meta: &Meta, + data: Value, + config: &RenderConfig, + explicit: Option<&std::path::Path>, +) -> Result { + let template = template::load(layout, variant, Some(assessment_id), explicit)?; + + let mut bodies = Vec::new(); + if template.wants(Slot::Meta) { + bodies.push(( + Slot::Meta, + binding(&config.meta_binding, &meta.value(config)), + )); + } + if template.wants(Slot::Data) { + bodies.push((Slot::Data, binding(&config.data_binding, &data))); + } + + let mut warnings = Vec::new(); + if template.is_inert() { + warnings.push(format!( + "the template {} declares no coursebank markers, so the report is empty; add `// \ + coursebank:data` where the body belongs", + template.origin + )); + } else if !template.wants(Slot::Data) { + warnings.push(format!( + "the template {} declares no `data` slot, so it received the metadata but not the \ + report itself", + template.origin + )); + } + if template.wants(Slot::Questions) { + return Err(Error::Invalid(vec![format!( + "the template {} declares a `questions` slot, but a diagnostic report carries no \ + questions to fill it with. Remove the marker: a report that reproduces the exam \ + cannot be returned before a makeup is given", + template.origin + )])); + } + + Ok(Document { + variant, + origin: template.origin.clone(), + slots: template.slots(), + text: template.render(&bodies), + warnings, + }) +} + +/// The file stem a student's report is written under. +/// +/// Uses the student key rather than the name: a key is unique, filesystem-safe, +/// and already a pseudonym when the store is pseudonymized. +/// +/// # Arguments +/// +/// * `assessment_id` - the assessment id. +/// * `student_key` - the student key. +/// +/// # Returns +/// +/// The stem, with no extension. +pub fn student_stem(assessment_id: &str, student_key: &str) -> String { + let safe: String = student_key + .chars() + .map(|c| { + if c.is_ascii_alphanumeric() || c == '-' || c == '_' { + c + } else { + '-' + } + }) + .collect(); + format!("{assessment_id}-{safe}") +} + +/// Summarizes what a set of rendered reports covered, for the command line. +/// +/// # Arguments +/// +/// * `written` - how many files were written. +/// * `students` - how many students they cover. +/// +/// # Returns +/// +/// A sentence. +pub fn summary(written: usize, students: usize) -> String { + format!( + "{written} file(s) for {students} student(s); compile them with `typst compile` or the \ + loop in the guide" + ) +} + +/// Tallies what a class diagnostic would tell you to do next. +/// +/// Kept here rather than in the template so that the command line and the PDF +/// agree about what counts as a finding. +/// +/// # Arguments +/// +/// * `diagnostic` - the class diagnostic. +/// +/// # Returns +/// +/// Short lines, most important first. +pub fn headline(diagnostic: &CohortDiagnostic) -> Vec { + let mut out = Vec::new(); + out.push(format!( + "{} student(s), mean {:.0}% (SD {:.1}), median {:.0}%", + diagnostic.n_students, + diagnostic.distribution.mean, + diagnostic.distribution.sd, + diagnostic.distribution.median + )); + if !diagnostic.gaps.is_empty() { + out.push(format!( + "{} objective(s) the class did not meet; worst is {} at {:.0}%", + diagnostic.gaps.len(), + diagnostic.gaps[0].id, + diagnostic.gaps[0].rate * 100.0 + )); + } + if !diagnostic.revise.is_empty() { + let numbers: Vec = diagnostic + .revise + .iter() + .take(6) + .map(|q| format!("q{}", q.number)) + .collect(); + out.push(format!( + "{} question(s) to look at before reuse: {}", + diagnostic.revise.len(), + numbers.join(", ") + )); + } + if diagnostic.forms.len() > 1 { + let spread = diagnostic + .forms + .iter() + .map(|f| f.mean) + .fold(f64::NEG_INFINITY, f64::max) + - diagnostic + .forms + .iter() + .map(|f| f.mean) + .fold(f64::INFINITY, f64::min); + out.push(format!( + "{} forms, {:.0} points apart at the mean", + diagnostic.forms.len(), + spread + )); + } + out +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::diagnostic::{Distribution, ReliabilityRow}; + + fn empty_cohort() -> CohortDiagnostic { + CohortDiagnostic { + n_students: 24, + n_items: 36, + distribution: Distribution { + mean: 72.5, + median: 74.0, + sd: 12.0, + min: 41.0, + max: 97.0, + bins: vec![Bin { + low: 70, + high: 80, + count: 9, + }], + }, + reliability: ReliabilityRow { + alpha: Some(0.71), + sem: Some(2.1), + mean_p: 0.72, + mean_point_biserial: Some(0.24), + interpretation: "acceptable for a classroom exam".into(), + }, + levels: Vec::new(), + objectives: Vec::new(), + gaps: Vec::new(), + questions: Vec::new(), + revise: Vec::new(), + forms: Vec::new(), + blueprint: Vec::new(), + patterns: Vec::new(), + warnings: Vec::new(), + } + } + + #[test] + fn the_headline_leads_with_the_distribution() { + let lines = headline(&empty_cohort()); + assert!(lines[0].contains("24 student(s)"), "{lines:?}"); + assert!( + lines[0].contains("73%") || lines[0].contains("72%"), + "{lines:?}" + ); + } + + #[test] + fn a_student_stem_is_filesystem_safe() { + assert_eq!(student_stem("e1", "s-9f8e7d"), "e1-s-9f8e7d"); + assert_eq!(student_stem("e1", "ada@x.edu"), "e1-ada-x-edu"); + } + + #[test] + fn the_cohort_value_carries_no_question_text() { + let config = RenderConfig::for_variant(Variant::CohortReport); + let text = cohort_value(&empty_cohort(), &config).to_typst(0); + assert!(text.contains("reliability")); + assert!(!text.contains("stem")); + } +} diff --git a/src/export/typst/payload.rs b/src/export/typst/payload.rs index a8ba3fb..656fb5e 100644 --- a/src/export/typst/payload.rs +++ b/src/export/typst/payload.rs @@ -1123,7 +1123,7 @@ fn rewrite_latex_math(source: &str) -> String { out } -fn markup_value(source: &str, content: bool) -> Value { +pub(crate) fn markup_value(source: &str, content: bool) -> Value { let source = rewrite_latex_math(source); if content { Value::content(source) diff --git a/src/export/typst/template.rs b/src/export/typst/template.rs index a73942c..ddc7e4a 100644 --- a/src/export/typst/template.rs +++ b/src/export/typst/template.rs @@ -76,6 +76,12 @@ const EMBEDDED_KEY: &str = include_str!("templates/key.typ"); /// The bundled answer sheet template. const EMBEDDED_ANSWER_SHEET: &str = include_str!("templates/answer-sheet.typ"); +/// The bundled student diagnostic template. +const EMBEDDED_STUDENT_REPORT: &str = include_str!("templates/student-report.typ"); + +/// The bundled class diagnostic template. +const EMBEDDED_COHORT_REPORT: &str = include_str!("templates/cohort-report.typ"); + /// An injection point a template can declare. #[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] pub enum Slot { @@ -174,6 +180,8 @@ pub fn embedded(variant: Variant) -> &'static str { Variant::Exam => EMBEDDED_EXAM, Variant::Key => EMBEDDED_KEY, Variant::AnswerSheet => EMBEDDED_ANSWER_SHEET, + Variant::StudentReport => EMBEDDED_STUDENT_REPORT, + Variant::CohortReport => EMBEDDED_COHORT_REPORT, } } diff --git a/src/export/typst/templates/cohort-report.typ b/src/export/typst/templates/cohort-report.typ new file mode 100644 index 0000000..c7ce4ec --- /dev/null +++ b/src/export/typst/templates/cohort-report.typ @@ -0,0 +1,558 @@ +// coursebank — class diagnostic template +// +// This file is a template, not generated output. `coursebank report cohort` +// replaces only the marked regions below. +// +// coursebank template dump --variant cohort-report +// typst watch templates/cohort-report.typ +// +// Markers: +// +// // coursebank:begin meta course, assessment, class size, policy +// // coursebank:end meta +// // coursebank:begin data the class diagnostic +// // coursebank:end data +// +// This document is for you, not for the class. It carries item statistics, the +// revision queue, and the per-option breakdown — everything the student report +// withholds except the question text itself, which is in the bank where it +// belongs. Do not hand it out: an option table tells a reader which letter was +// keyed. + +#import "@preview/mitex:0.2.7": mi + +// ───────────────────────────────────────────────────────────────────────────── +// Data +// ───────────────────────────────────────────────────────────────────────────── + +// coursebank:begin meta +#let cb-meta = ( + course: (code: "COURSE 101", title: "Sample Course", term: "2026f"), + assessment: (id: "sample", title: "Sample assessment", date: "2026-01-01"), + generator: (tool: "coursebank", version: "0.0.0", on: "2026-01-02"), + policy: (mastery-threshold: 0.75, min-items-for-mastery: 2), + class-size: 24, + extra: (:), +) +// coursebank:end meta + +// coursebank:begin data +#let cb-data = ( + students: 24, + items: 36, + distribution: ( + mean: 71.2, median: 73.0, sd: 11.4, min: 44.0, max: 94.0, + bins: ( + (low: 40, high: 50, count: 1), + (low: 50, high: 60, count: 3), + (low: 60, high: 70, count: 6), + (low: 70, high: 80, count: 9), + (low: 80, high: 90, count: 4), + (low: 90, high: 100, count: 1), + ), + ), + reliability: ( + alpha: 0.71, sem: 2.1, mean-p: 0.72, mean-point-biserial: 0.24, + interpretation: "Reliability is 0.71, acceptable for a classroom exam.", + ), + levels: ( + (level: 1, name: "Remember", items: 6, rate: 0.91), + (level: 3, name: "Apply", items: 9, rate: 0.64), + ), + objectives: ( + (id: "lo-sample-gap", text: [A sample objective the class struggled with.], + items: 3, rate: 0.41, meeting: 4, developing: 7, not-yet: 13, thin: 0, + below-threshold: true), + ), + gaps: ( + (id: "lo-sample-gap", text: [A sample objective the class struggled with.], + items: 3, rate: 0.41, meeting: 4, developing: 7, not-yet: 13, thin: 0, + below-threshold: true), + ), + questions: ( + (number: 1, item: "bank::q-sample-001", level: 1, objectives: ("lo-sample-gap",), + p: 0.42, point-biserial: 0.05, discrimination: 0.10, blank-rate: 0.0, + key: ("B",), + options: ( + (letter: "A", count: 9, rate: 0.375, is-key: false, point-biserial: 0.11, nonfunctioning: false), + (letter: "B", count: 10, rate: 0.417, is-key: true, point-biserial: 0.05, nonfunctioning: false), + (letter: "C", count: 5, rate: 0.208, is-key: false, point-biserial: -0.2, nonfunctioning: false), + (letter: "D", count: 0, rate: 0.0, is-key: false, nonfunctioning: true), + ), + flags: ("ambiguous",), + notes: ([Distractor A drew as many strong students as the key.],), + by-form: (A: 0.55, B: 0.30)), + ), + revise: (), + forms: ( + (id: "A", students: 12, mean: 73.5, sd: 10.2), + (id: "B", students: 12, mean: 68.9, sd: 12.1), + ), + blueprint: (), + patterns: (), + warnings: (), +) +// coursebank:end data + +// ───────────────────────────────────────────────────────────────────────────── +// Settings +// ───────────────────────────────────────────────────────────────────────────── + +#let extra = cb-meta.at("extra", default: (:)) +#let accent = rgb(extra.at("accent", default: "#1f4e79")) +#let body-font = extra.at("font", default: "Roboto") +#let body-size = eval(extra.at("font-size", default: "9.5pt")) +#let paper = extra.at("paper", default: "us-letter") +#let show-options = extra.at("option-tables", default: true) + +#let threshold = cb-meta.policy.at("mastery-threshold", default: 0.75) + +#let ok-color = rgb("#2a9d8f") +#let mid-color = rgb("#D19F1F") +#let bad-color = rgb("#E24E29") + +#set page( + paper: paper, + margin: (x: 1.7cm, y: 2.0cm), + header: text(size: 0.8em, fill: luma(110))[ + #cb-meta.course.code · #cb-meta.assessment.title · class diagnostic + #h(1fr) + instructor copy + ], + footer: context text(size: 0.8em, fill: luma(110))[ + #h(1fr) + Page #counter(page).display("1 of 1", both: true) + ], +) +#set text(font: body-font, size: body-size, lang: "en") +#set par(justify: false, leading: 0.6em) +#show heading.where(level: 1): it => block(above: 1.4em, below: 0.7em)[ + #text(size: 1.15em, weight: "bold", fill: accent)[#it.body] + #v(-0.45em) + #line(length: 100%, stroke: 0.6pt + accent.lighten(55%)) +] + +// ───────────────────────────────────────────────────────────────────────────── +// Helpers +// ───────────────────────────────────────────────────────────────────────────── + +#let markup(v) = if type(v) == str { eval(v, mode: "markup") } else { v } +#let pct(rate) = str(calc.round(rate * 100)) + "%" +#let num(value, digits: 2) = str(calc.round(value, digits: digits)) +#let signed(value) = (if value >= 0 { "+" } else { "" }) + num(value) + +#let rate-color(rate) = { + if rate >= threshold { ok-color } + else if rate >= threshold * 0.6 { mid-color } + else { bad-color } +} + +#let badge(label, color) = box( + fill: color.lighten(82%), + radius: 3pt, + inset: (x: 4pt, y: 2pt), +)[#text(size: 0.7em, weight: "bold", fill: color.darken(12%))[#label]] + +#let bar(rate, color: accent, width: 2.4cm) = { + let r = calc.max(0.0, calc.min(1.0, rate)) + box(baseline: 0.15em)[ + #stack( + dir: ltr, + box(width: width * r, height: 0.58em, fill: color, radius: (left: 2pt)), + box(width: width * (1.0 - r), height: 0.58em, fill: luma(232), radius: (right: 2pt)), + ) + ] +} + +#let stat-card(label, value, note: none) = block( + width: 100%, + fill: luma(247), + radius: 4pt, + inset: (x: 9pt, y: 8pt), +)[ + #text(size: 0.74em, fill: luma(95))[#upper(label)] + #v(0.12em) + #text(size: 1.35em, weight: "bold", fill: accent)[#value] + #if note != none [ + #v(0.08em) + #text(size: 0.76em, fill: luma(95))[#note] + ] +] + +// ───────────────────────────────────────────────────────────────────────────── +// Heading and headline numbers +// ───────────────────────────────────────────────────────────────────────────── + +#block(below: 1em)[ + #text(size: 1.5em, weight: "bold")[#cb-meta.assessment.title — class diagnostic] + #linebreak() + #text(size: 0.95em, fill: luma(110))[ + #cb-meta.course.code · #cb-meta.course.term + #{ + let date = cb-meta.assessment.at("date", default: none) + if date != none [ · administered #date ] + } + ] +] + +#let dist = cb-data.distribution +#let rel = cb-data.reliability + +#grid( + columns: (1fr, 1fr, 1fr, 1fr), + gutter: 9pt, + stat-card("students", str(cb-data.students), note: str(cb-data.items) + " scored items"), + stat-card("mean", str(calc.round(dist.mean)) + "%", note: "median " + str(calc.round(dist.median)) + "%"), + stat-card("spread", "SD " + num(dist.sd, digits: 1), note: str(calc.round(dist.min)) + "–" + str(calc.round(dist.max)) + "%"), + stat-card( + "reliability", + { + let alpha = rel.at("alpha", default: none) + if alpha != none { num(alpha) } else { "n/a" } + }, + note: { + let sem = rel.at("sem", default: none) + if sem != none { "SEM " + num(sem, digits: 1) + " items" } else { "KR-20" } + }, + ), +) + +#v(0.6em) +#text(size: 0.88em, fill: luma(90))[#rel.interpretation] + +// ───────────────────────────────────────────────────────────────────────────── +// Distribution +// ───────────────────────────────────────────────────────────────────────────── + +#let bins = dist.at("bins", default: ()) +#let peak = if bins.len() > 0 { calc.max(..bins.map(b => b.count)) } else { 0 } + +#if peak > 0 [ + = Score distribution + + #let height = 3.2cm + #align(center)[ + #grid( + columns: (1fr,) * bins.len(), + gutter: 5pt, + ..bins.map(b => { + let share = if peak > 0 { b.count / peak } else { 0.0 } + stack( + dir: ttb, + spacing: 3pt, + align(center + bottom)[ + #box(height: height)[ + #align(bottom)[ + #box( + width: 100%, + height: height * share, + fill: accent.lighten(if b.low >= 70 { 55% } else { 25% }), + radius: (top: 2pt), + ) + ] + ] + ], + align(center)[#text(size: 0.72em, weight: "bold")[#b.count]], + align(center)[#text(size: 0.68em, fill: luma(110))[#b.low]], + ) + }), + ) + ] + #align(center)[#text(size: 0.72em, fill: luma(120))[percent scored, in ten-point bins]] +] + +// ───────────────────────────────────────────────────────────────────────────── +// Forms +// ───────────────────────────────────────────────────────────────────────────── + +#let forms = cb-data.at("forms", default: ()) + +#if forms.len() > 1 [ + = Forms + + #text(size: 0.9em, fill: luma(95))[ + Forms differ only in order, so their means should differ only by who sat + them. A persistent gap points at an item whose permutation made it easier or + harder, and the per-question columns further down are where to look. + ] + + #v(0.4em) + + #table( + columns: (auto, auto, auto, auto), + stroke: none, + align: (left, right, right, right), + inset: (x: 7pt, y: 4pt), + fill: (_, row) => if calc.odd(row) { luma(250) } else { white }, + table.header( + text(size: 0.78em, fill: luma(95))[FORM], + text(size: 0.78em, fill: luma(95))[STUDENTS], + text(size: 0.78em, fill: luma(95))[MEAN], + text(size: 0.78em, fill: luma(95))[SD], + ), + ..forms.map(form => ( + [*#form.id*], + [#form.students], + [#num(form.mean, digits: 1)%], + [#num(form.sd, digits: 1)], + )).flatten(), + ) +] + +// ───────────────────────────────────────────────────────────────────────────── +// Levels +// ───────────────────────────────────────────────────────────────────────────── + +#if cb-data.levels.len() > 0 [ + = By level + + #table( + columns: (auto, auto, auto, 1fr), + stroke: none, + align: (left + horizon, right + horizon, right + horizon, left + horizon), + inset: (x: 7pt, y: 4pt), + fill: (_, row) => if calc.odd(row) { luma(250) } else { white }, + table.header( + text(size: 0.78em, fill: luma(95))[LEVEL], + text(size: 0.78em, fill: luma(95))[ITEMS], + text(size: 0.78em, fill: luma(95))[CLASS], + text(size: 0.78em, fill: luma(95))[], + ), + ..cb-data.levels.map(level => ( + [#level.level · *#level.name*], + [#level.items], + [#pct(level.rate)], + bar(level.rate, color: rate-color(level.rate), width: 5cm), + )).flatten(), + ) +] + +// ───────────────────────────────────────────────────────────────────────────── +// Objectives +// ───────────────────────────────────────────────────────────────────────────── + +#let objectives = cb-data.at("objectives", default: ()) + +#if objectives.len() > 0 [ + = By objective, worst first + + #text(size: 0.9em, fill: luma(95))[ + The counts split the class into how many are meeting, developing, and not yet + on each objective. An objective where the class divides evenly is a different + teaching problem from one where nearly everyone is short. + ] + + #v(0.4em) + + #table( + columns: (1fr, auto, auto, auto, auto, auto, auto), + stroke: none, + align: (left + horizon, right + horizon, right + horizon, right + horizon, right + horizon, right + horizon, left + horizon), + inset: (x: 6pt, y: 4pt), + fill: (_, row) => if calc.odd(row) { luma(250) } else { white }, + table.header( + text(size: 0.78em, fill: luma(95))[OBJECTIVE], + text(size: 0.78em, fill: luma(95))[Q], + text(size: 0.78em, fill: luma(95))[CLASS], + text(size: 0.78em, fill: luma(95))[MET], + text(size: 0.78em, fill: luma(95))[DEV], + text(size: 0.78em, fill: luma(95))[NOT], + text(size: 0.78em, fill: luma(95))[], + ), + ..objectives.map(objective => ( + { + markup(objective.text) + if objective.at("below-threshold", default: false) { + [ #badge("below " + pct(threshold), bad-color)] + } + }, + [#objective.items], + [#pct(objective.rate)], + [#objective.meeting], + [#objective.developing], + [#objective.at("not-yet", default: 0)], + bar(objective.rate, color: rate-color(objective.rate), width: 2.2cm), + )).flatten(), + ) +] + +// ───────────────────────────────────────────────────────────────────────────── +// Items +// ───────────────────────────────────────────────────────────────────────────── + +#let questions = cb-data.at("questions", default: ()) + +#if questions.len() > 0 [ + #pagebreak(weak: true) + = Item analysis + + #text(size: 0.9em, fill: luma(95))[ + #emph[p] is the share answering correctly; #emph[r] is the corrected + item-total correlation, which should be positive and is the single most + useful column; #emph[D] is the upper group minus the lower group. Option + letters are the bank's, not any one form's. + ] + + #v(0.4em) + + #table( + columns: (auto, auto, auto, auto, auto, auto, 1fr), + stroke: none, + align: (right + horizon, right + horizon, right + horizon, right + horizon, right + horizon, right + horizon, left + horizon), + inset: (x: 5pt, y: 3.5pt), + fill: (_, row) => if calc.odd(row) { luma(250) } else { white }, + table.header( + text(size: 0.78em, fill: luma(95))[Q], + text(size: 0.78em, fill: luma(95))[LVL], + text(size: 0.78em, fill: luma(95))[KEY], + text(size: 0.78em, fill: luma(95))[p], + text(size: 0.78em, fill: luma(95))[r], + text(size: 0.78em, fill: luma(95))[D], + text(size: 0.78em, fill: luma(95))[FLAGS], + ), + ..questions.map(q => ( + [*#q.number*], + { + let level = q.at("level", default: none) + if level != none [#level] else [] + }, + [#q.at("key", default: ()).join("")], + text(fill: rate-color(q.p))[#num(q.p)], + { + let r = q.at("point-biserial", default: none) + if r == none { text(fill: luma(140))[n/a] } + else if r < 0.0 { text(fill: bad-color, weight: "bold")[#signed(r)] } + else if r < 0.15 { text(fill: mid-color)[#signed(r)] } + else { [#signed(r)] } + }, + { + let d = q.at("discrimination", default: none) + if d != none [#signed(d)] else [—] + }, + { + let flags = q.at("flags", default: ()) + let forms-gap = { + let by-form = q.at("by-form", default: (:)) + let values = by-form.values() + if values.len() > 1 and calc.max(..values) - calc.min(..values) >= 0.25 { + (badge("form gap " + pct(calc.max(..values) - calc.min(..values)), mid-color),) + } else { () } + } + stack( + dir: ltr, + spacing: 3pt, + ..flags.map(f => badge(f, bad-color)), + ..forms-gap, + ) + }, + )).flatten(), + ) +] + +// ───────────────────────────────────────────────────────────────────────────── +// The revise queue +// ───────────────────────────────────────────────────────────────────────────── + +#let revise = cb-data.at("revise", default: ()) + +#if revise.len() > 0 [ + = Before you use these again + + #for q in revise [ + #block(breakable: false, above: 0.9em, width: 100%)[ + #text(weight: "bold", fill: accent)[Question #q.number] + #{ + let item = q.at("item", default: none) + if item != none [ #text(size: 0.8em, fill: luma(120))[#item]] + } + #h(0.5em) + #stack(dir: ltr, spacing: 3pt, ..q.at("flags", default: ()).map(f => badge(f, bad-color))) + + #v(0.25em) + #for note in q.at("notes", default: ()) [ + #text(size: 0.9em)[— #markup(note)] + #linebreak() + ] + + #if show-options and q.at("options", default: ()).len() > 0 [ + #v(0.3em) + #table( + columns: (auto, auto, auto, auto, 1fr), + stroke: none, + align: (center + horizon, right + horizon, right + horizon, left + horizon, left + horizon), + inset: (x: 5pt, y: 3pt), + table.header( + text(size: 0.74em, fill: luma(95))[OPT], + text(size: 0.74em, fill: luma(95))[n], + text(size: 0.74em, fill: luma(95))[SHARE], + text(size: 0.74em, fill: luma(95))[], + text(size: 0.74em, fill: luma(95))[r], + ), + ..q.options.map(option => ( + { + if option.at("is-key", default: false) { + text(weight: "bold", fill: ok-color)[#option.letter] + } else { [#option.letter] } + }, + [#option.count], + [#pct(option.rate)], + bar( + option.rate, + color: if option.at("is-key", default: false) { ok-color } else { luma(160) }, + width: 2.6cm, + ), + { + let r = option.at("point-biserial", default: none) + if r == none { text(fill: luma(150))[—] } + else if not option.at("is-key", default: false) and r > 0.0 { + text(fill: mid-color)[#signed(r)] + } else { text(size: 0.95em)[#signed(r)] } + }, + )).flatten(), + ) + ] + ] + ] +] + +// ───────────────────────────────────────────────────────────────────────────── +// Blueprint and cautions +// ───────────────────────────────────────────────────────────────────────────── + +#let blueprint = cb-data.at("blueprint", default: ()) +#let warnings = cb-data.at("warnings", default: ()) +#let patterns = cb-data.at("patterns", default: ()) + +#if patterns.len() > 0 [ + = Response profiles + + #for pattern in patterns [ + - *#pattern.label* — #pattern.students student(s) + ] +] + +#if blueprint.len() > 0 [ + = Where the form differs from its blueprint + + #for line in blueprint [ + - #line + ] +] + +#if warnings.len() > 0 [ + = Cautions + + #for line in warnings [ + - #text(size: 0.9em)[#line] + ] +] + +#v(1.2em) +#line(length: 100%, stroke: 0.5pt + luma(210)) +#v(0.4em) +#text(size: 0.76em, fill: luma(120))[ + Generated #cb-meta.generator.on by #cb-meta.generator.tool #cb-meta.generator.version. + With #cb-data.students students, an item statistic has a standard error of + roughly 0.2; treat single-item results as provisional and the pattern across + items as real. Instructor copy — the option tables identify the key. +] diff --git a/src/export/typst/templates/student-report.typ b/src/export/typst/templates/student-report.typ new file mode 100644 index 0000000..0d4d259 --- /dev/null +++ b/src/export/typst/templates/student-report.typ @@ -0,0 +1,552 @@ +// coursebank — individual diagnostic template +// +// This file is a template, not generated output. `coursebank report students` +// replaces only the marked regions below and leaves every other line exactly as +// you wrote it, so this is where layout decisions belong. +// +// coursebank template dump --variant student-report +// typst watch templates/student-report.typ +// +// Two markers are in play: +// +// // coursebank:begin meta course, assessment, class size, policy +// // coursebank:end meta +// // coursebank:begin data one student's diagnostic +// // coursebank:end data +// +// Both regions ship with sample values, so `typst watch` works before any report +// has been generated. +// +// What is not in the payload: the questions. There is no stem field and no option +// text field, on any question, in any configuration. A student report is handed +// back before the makeup exam is given, and a report that reproduces the paper +// cannot be. If you find yourself wanting to print the question, print its number +// and let the student read it off their own copy. + +#import "@preview/mitex:0.2.7": mi + +// ───────────────────────────────────────────────────────────────────────────── +// Data +// ───────────────────────────────────────────────────────────────────────────── + +// coursebank:begin meta +#let cb-meta = ( + course: (code: "COURSE 101", title: "Sample Course", term: "2026f"), + assessment: (id: "sample", title: "Sample assessment", date: "2026-01-01"), + generator: (tool: "coursebank", version: "0.0.0", on: "2026-01-02"), + policy: (mastery-threshold: 0.75, min-items-for-mastery: 2), + class-size: 24, + extra: (:), +) +// coursebank:end meta + +// coursebank:begin data +#let cb-data = ( + student-key: "s-000000000000", + name: "Sample Student", + form: "A", + score: (points: 27.0, possible: 36.0, percent: 75.0, bonus: 1.0, correct: 27, items: 36), + standing: (class-mean: 71.2, class-sd: 11.4, band: "upper half"), + levels: ( + (level: 1, name: "Remember", blurb: "recalling terms, facts, and definitions", + items: 6, rate: 1.0, class-rate: 0.91, comparison: "above the class"), + (level: 3, name: "Apply", blurb: "using a procedure in a new situation", + items: 9, rate: 0.56, class-rate: 0.64, comparison: "below the class"), + ), + objectives: ( + (id: "lo-sample-met", text: [A sample objective this student met.], items: 3, + credit: 3.0, rate: 1.0, lower: 0.44, upper: 1.0, class-rate: 0.81, + status: "meeting", symbol: "✓", confident: false, thin-evidence: false, levels: (1, 2)), + (id: "lo-sample-gap", text: [A sample objective to work on.], items: 3, + credit: 1.0, rate: 0.33, lower: 0.06, upper: 0.79, class-rate: 0.58, + status: "not yet", symbol: "✗", confident: false, thin-evidence: false, levels: (3,)), + ), + strengths: ((id: "lo-sample-met", text: [A sample objective this student met.], rate: 1.0, items: 3),), + focus: ((id: "lo-sample-gap", text: [A sample objective to work on.], rate: 0.33, items: 3),), + questions: ( + (number: 1, level: 1, objectives: ("lo-sample-met",), correct: true, credit: 1.0, + bonus: false, blank: false, class-rate: 0.91, taught-in: ()), + (number: 2, level: 3, objectives: ("lo-sample-gap",), correct: false, credit: 0.0, + bonus: false, blank: false, class-rate: 0.58, + feedback: [This is the note written for the option that was chosen.], + taught-in: ("Enthalpy (L1.1), slides 12, 13",)), + ), + study: ( + (objective: "lo-sample-gap", text: [A sample objective to work on.], rate: 0.33, + readings: ( + (citation: "KKW §6.2", lecture: "L1.1", lecture-title: "Enthalpy", + focus: [What to take from this section.], supplemental: false), + )), + ), +) +// coursebank:end data + +// ───────────────────────────────────────────────────────────────────────────── +// Settings +// ───────────────────────────────────────────────────────────────────────────── + +#let extra = cb-meta.at("extra", default: (:)) +#let accent = rgb(extra.at("accent", default: "#1f4e79")) +#let body-font = extra.at("font", default: "Roboto") +#let body-size = eval(extra.at("font-size", default: "10pt")) +#let paper = extra.at("paper", default: "us-letter") + +// Turn sections off from templates/typst.yaml rather than by deleting code, so +// a course that does not want the question map keeps the rest of this file. +#let show-question-map = extra.at("question-map", default: true) +#let show-feedback = extra.at("feedback", default: true) +#let show-study = extra.at("study-plan", default: true) +#let show-comparison = extra.at("comparison", default: true) + +#let threshold = cb-meta.policy.at("mastery-threshold", default: 0.75) + +#let ok-color = rgb("#2a9d8f") +#let mid-color = rgb("#D19F1F") +#let bad-color = rgb("#E24E29") +#let thin-color = luma(150) + +#set page( + paper: paper, + margin: (x: 1.9cm, y: 2.1cm), + header: text(size: 0.8em, fill: luma(110))[ + #cb-meta.course.code · #cb-meta.assessment.title · individual diagnostic + ], + footer: context text(size: 0.8em, fill: luma(110))[ + #h(1fr) + Page #counter(page).display("1 of 1", both: true) + ], +) +#set text(font: body-font, size: body-size, lang: "en") +#set par(justify: false, leading: 0.62em) +#show heading.where(level: 1): it => block(above: 1.4em, below: 0.7em)[ + #text(size: 1.15em, weight: "bold", fill: accent)[#it.body] + #v(-0.45em) + #line(length: 100%, stroke: 0.6pt + accent.lighten(55%)) +] +#show heading.where(level: 2): it => block(above: 1em, below: 0.45em)[ + #text(size: 1.02em, weight: "bold")[#it.body] +] + +// ───────────────────────────────────────────────────────────────────────────── +// Helpers +// ───────────────────────────────────────────────────────────────────────────── + +#let markup(v) = if type(v) == str { eval(v, mode: "markup") } else { v } + +// A rate is a fraction in 0..1; a percent is already out of 100. Keeping the two +// straight is the only arithmetic this template does. +#let pct(rate) = str(calc.round(rate * 100)) + "%" +#let pct1(value) = str(calc.round(value, digits: 0)) + "%" + +#let status-color(status) = { + if status == "meeting" { ok-color } + else if status == "developing" { mid-color } + else if status == "not yet" { bad-color } + else { thin-color } +} + +#let rate-color(rate) = { + if rate >= threshold { ok-color } + else if rate >= threshold * 0.6 { mid-color } + else { bad-color } +} + +#let badge(label, color) = box( + fill: color.lighten(82%), + radius: 3pt, + inset: (x: 5pt, y: 2.5pt), +)[#text(size: 0.72em, weight: "bold", fill: color.darken(12%))[#label]] + +// Two boxes side by side rather than an overlay: no `place`, no coordinate +// arithmetic, and it degrades to something sensible at any width. +#let bar(rate, color: accent, width: 3.6cm) = { + let r = calc.max(0.0, calc.min(1.0, rate)) + box(baseline: 0.15em)[ + #stack( + dir: ltr, + box(width: width * r, height: 0.62em, fill: color, radius: (left: 2pt)), + box(width: width * (1.0 - r), height: 0.62em, fill: luma(232), radius: (right: 2pt)), + ) + ] +} + +#let stat-card(label, value, note: none) = block( + width: 100%, + fill: luma(247), + radius: 4pt, + inset: (x: 10pt, y: 9pt), +)[ + #text(size: 0.75em, fill: luma(95))[#upper(label)] + #v(0.15em) + #text(size: 1.45em, weight: "bold", fill: accent)[#value] + #if note != none [ + #v(0.1em) + #text(size: 0.78em, fill: luma(95))[#note] + ] +] + +// ───────────────────────────────────────────────────────────────────────────── +// Heading +// ───────────────────────────────────────────────────────────────────────────── + +#block(below: 0.6em)[ + #text(size: 1.5em, weight: "bold")[#cb-meta.assessment.title] + #h(0.6em) + #text(size: 1em, fill: luma(110))[ + #cb-meta.course.code — #cb-meta.course.title + ] +] + +#block(below: 1.2em)[ + #text(size: 0.95em)[ + *#cb-data.at("name", default: cb-data.student-key)* + #{ + let sid = cb-data.at("sid", default: none) + if sid != none [ · #sid ] + } + #{ + let form = cb-data.at("form", default: none) + if form != none [ · Form #form ] + } + #{ + let date = cb-meta.assessment.at("date", default: none) + if date != none [ · #date ] + } + ] +] + +// ───────────────────────────────────────────────────────────────────────────── +// Score +// ───────────────────────────────────────────────────────────────────────────── + +#let score = cb-data.score +#let standing = cb-data.at("standing", default: none) + +#grid( + columns: (1fr, 1fr, 1fr), + gutter: 10pt, + stat-card( + "score", + pct1(score.percent), + note: str(score.points) + " of " + str(score.possible) + " points" + + (if score.at("bonus", default: 0.0) > 0.0 { " (+" + str(score.bonus) + " bonus)" } else { "" }), + ), + stat-card( + "questions", + str(score.correct) + " / " + str(score.items), + note: "answered correctly", + ), + if standing != none and show-comparison { + stat-card( + "class", + pct1(standing.class-mean), + note: "class average · you are in the " + standing.band, + ) + } else { + stat-card("class size", str(cb-meta.at("class-size", default: 0)), note: "students") + }, +) + +// ───────────────────────────────────────────────────────────────────────────── +// Levels +// ───────────────────────────────────────────────────────────────────────────── + +#if cb-data.levels.len() > 0 [ + = How you did, by kind of thinking + + #text(size: 0.9em, fill: luma(95))[ + Questions are written at levels. Scoring well on recall and less well on + application is a different problem from scoring evenly and low, and it has a + different fix. + ] + + #v(0.5em) + + #table( + columns: (auto, 1fr, auto, auto, auto), + stroke: none, + align: (left + horizon, left + horizon, left + horizon, right + horizon, left + horizon), + inset: (x: 5pt, y: 5pt), + fill: (_, row) => if calc.odd(row) { luma(250) } else { white }, + table.header( + text(size: 0.78em, fill: luma(95))[LEVEL], + text(size: 0.78em, fill: luma(95))[WHAT IT ASKS], + text(size: 0.78em, fill: luma(95))[YOU], + text(size: 0.78em, fill: luma(95))[], + text(size: 0.78em, fill: luma(95))[CLASS], + ), + ..cb-data.levels.map(level => ( + [*#level.name* #text(size: 0.8em, fill: luma(120))[(#level.items)]], + text(size: 0.85em, fill: luma(80))[#level.blurb], + bar(level.rate, color: rate-color(level.rate), width: 2.6cm), + [#pct(level.rate)], + { + let class-rate = level.at("class-rate", default: none) + if class-rate != none and show-comparison { + text(size: 0.85em, fill: luma(100))[#pct(class-rate)] + } else { [] } + }, + )).flatten(), + ) +] + +// ───────────────────────────────────────────────────────────────────────────── +// Objectives +// ───────────────────────────────────────────────────────────────────────────── + +#if cb-data.objectives.len() > 0 [ + = What you have learned, objective by objective + + #text(size: 0.9em, fill: luma(95))[ + Each line is one thing the course asked you to be able to do, and how many + questions measured it. Two or three questions is thin evidence, so treat a + single line as a hint rather than a verdict; the pattern across lines is what + to trust. + ] + + #v(0.5em) + + #table( + columns: (auto, 1fr, auto, auto, auto), + stroke: none, + align: (center + horizon, left + horizon, center + horizon, left + horizon, right + horizon), + inset: (x: 5pt, y: 5pt), + fill: (_, row) => if calc.odd(row) { luma(250) } else { white }, + table.header( + text(size: 0.78em, fill: luma(95))[], + text(size: 0.78em, fill: luma(95))[OBJECTIVE], + text(size: 0.78em, fill: luma(95))[Q], + text(size: 0.78em, fill: luma(95))[YOU], + text(size: 0.78em, fill: luma(95))[CLASS], + ), + ..cb-data.objectives.map(objective => ( + text(fill: status-color(objective.status), weight: "bold")[#objective.symbol], + { + markup(objective.text) + if objective.at("thin-evidence", default: false) { + [ #badge("too few questions to say", thin-color)] + } + }, + text(size: 0.85em, fill: luma(110))[#objective.items], + { + stack( + dir: ltr, + spacing: 5pt, + bar(objective.rate, color: status-color(objective.status), width: 2.1cm), + text(size: 0.9em)[#pct(objective.rate)], + ) + }, + { + let class-rate = objective.at("class-rate", default: none) + if class-rate != none and show-comparison { + text(size: 0.85em, fill: luma(100))[#pct(class-rate)] + } else { [] } + }, + )).flatten(), + ) + + #v(0.4em) + #text(size: 0.78em, fill: luma(110))[ + #text(fill: ok-color, weight: "bold")[✓] meeting · #text(fill: mid-color, weight: "bold")[~] + developing · #text(fill: bad-color, weight: "bold")[✗] not yet · + #text(fill: thin-color, weight: "bold")[?] not enough evidence. The line is + drawn at #pct(threshold). + ] +] + +// ───────────────────────────────────────────────────────────────────────────── +// Strengths and focus +// ───────────────────────────────────────────────────────────────────────────── + +#let strengths = cb-data.at("strengths", default: ()) +#let focus = cb-data.at("focus", default: ()) + +#if strengths.len() > 0 or focus.len() > 0 [ + = Where to put your time + + #grid( + columns: (1fr, 1fr), + gutter: 12pt, + if focus.len() > 0 { + block( + width: 100%, + fill: bad-color.lighten(93%), + radius: 4pt, + inset: 10pt, + )[ + #text(weight: "bold", fill: bad-color.darken(15%))[Work on these first] + #v(0.35em) + #for objective in focus [ + - #markup(objective.text) + #text(size: 0.8em, fill: luma(110))[ (#pct(objective.rate) of #objective.items)] + ] + ] + } else { [] }, + if strengths.len() > 0 { + block( + width: 100%, + fill: ok-color.lighten(93%), + radius: 4pt, + inset: 10pt, + )[ + #text(weight: "bold", fill: ok-color.darken(18%))[Solid, keep it] + #v(0.35em) + #for objective in strengths [ + - #markup(objective.text) + ] + ] + } else { [] }, + ) +] + +// ───────────────────────────────────────────────────────────────────────────── +// Study plan +// ───────────────────────────────────────────────────────────────────────────── + +#let study = cb-data.at("study", default: ()) + +#if show-study and study.len() > 0 [ + = What to read + + #text(size: 0.9em, fill: luma(95))[ + These are the sections behind the objectives above, taken from the course + reading list rather than chosen generically. + ] + + #v(0.4em) + + #for group in study [ + #block(breakable: false, above: 0.8em)[ + == #markup(group.text) + + #for reading in group.readings [ + #block(inset: (left: 0.8em), above: 0.35em)[ + #{ + let url = reading.at("url", default: none) + let cite = text(weight: "bold")[#reading.citation] + if url != none { link(url)[#cite] } else { cite } + } + #text(size: 0.85em, fill: luma(110))[ + — #reading.lecture-title (#reading.lecture)#{ + if reading.at("supplemental", default: false) { ", supplemental" } + } + ] + #{ + let focus-note = reading.at("focus", default: none) + if focus-note != none [ + \ #text(size: 0.9em)[#markup(focus-note)] + ] + } + ] + ] + ] + ] +] + +// ───────────────────────────────────────────────────────────────────────────── +// Question map +// ───────────────────────────────────────────────────────────────────────────── + +#let questions = cb-data.at("questions", default: ()) + +#let question-box(q) = { + let color = if q.at("blank", default: false) { luma(160) } + else if q.at("correct", default: false) == true { ok-color } + else if q.at("credit", default: 0.0) > 0.0 { mid-color } + else { bad-color } + box( + width: 100%, + fill: color.lighten(85%), + stroke: 0.5pt + color.lighten(45%), + radius: 3pt, + inset: (x: 2pt, y: 4pt), + )[ + #align(center)[ + #text(size: 0.85em, weight: "bold", fill: color.darken(18%))[#q.number] + #{ + let level = q.at("level", default: none) + if level != none { + linebreak() + text(size: 0.62em, fill: luma(110))[L#level] + } + } + ] + ] +} + +#if show-question-map and questions.len() > 0 [ + = Question by question + + #text(size: 0.9em, fill: luma(95))[ + Numbers refer to your own copy of the exam. The questions themselves are not + reproduced here. + ] + + #v(0.5em) + + #grid( + columns: (1fr,) * 10, + gutter: 4pt, + ..questions.map(question-box), + ) + + #v(0.5em) + #text(size: 0.78em, fill: luma(110))[ + #box(width: 0.7em, height: 0.7em, fill: ok-color.lighten(70%), radius: 2pt) correct · + #box(width: 0.7em, height: 0.7em, fill: mid-color.lighten(70%), radius: 2pt) partial · + #box(width: 0.7em, height: 0.7em, fill: bad-color.lighten(70%), radius: 2pt) incorrect · + #box(width: 0.7em, height: 0.7em, fill: luma(210), radius: 2pt) blank + ] +] + +// ───────────────────────────────────────────────────────────────────────────── +// Notes on what was missed +// ───────────────────────────────────────────────────────────────────────────── + +#let missed = questions.filter(q => + q.at("credit", default: 0.0) < 0.999 and q.at("feedback", default: none) != none +) + +#if show-feedback and missed.len() > 0 [ + = Notes on the ones you missed + + #text(size: 0.9em, fill: luma(95))[ + Each note is written for the specific option you chose, so it names the idea + behind that answer rather than restating the right one. + ] + + #v(0.5em) + + #for q in missed [ + #block(breakable: false, above: 0.7em, width: 100%)[ + #grid( + columns: (2.6em, 1fr), + gutter: 6pt, + align(top)[#text(weight: "bold", fill: accent)[#q.number.]], + [ + #markup(q.feedback) + #{ + let taught = q.at("taught-in", default: ()) + if taught.len() > 0 [ + \ #text(size: 0.82em, fill: luma(110))[Covered in: #taught.join("; ")] + ] + } + ], + ) + ] + ] +] + +// ───────────────────────────────────────────────────────────────────────────── +// Footer note +// ───────────────────────────────────────────────────────────────────────────── + +#v(1.2em) +#line(length: 100%, stroke: 0.5pt + luma(210)) +#v(0.4em) +#text(size: 0.76em, fill: luma(120))[ + Generated #cb-meta.generator.on by #cb-meta.generator.tool #cb-meta.generator.version + from #cb-meta.at("class-size", default: 0) student(s). Percentages on a handful of + questions carry wide uncertainty; bring this to office hours rather than reading + it as a verdict. +] diff --git a/src/lib.rs b/src/lib.rs index 4782d67..d5559ea 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -100,14 +100,14 @@ pub mod util; pub use util::{date, hash, markup, rng, yaml, zipfile}; -pub use model::{assessment, bank, catalog, course, history, item, layout, taxonomy}; +pub use model::{assessment, bank, catalog, course, history, item, layout, seal, taxonomy}; pub use authoring::{jsonschema, lint, select}; pub use data::store_parquet; -pub use data::{canvas, gradescope, responses, store}; +pub use data::{canvas, decode, gradescope, intake, responses, store}; -pub use analysis::{calibrate, classical, irt, students}; +pub use analysis::{calibrate, classical, diagnostic, irt, students}; pub use export::site; pub use export::{lecture, practice, qti, report, typst}; diff --git a/src/model.rs b/src/model.rs index acded0c..68a0a71 100644 --- a/src/model.rs +++ b/src/model.rs @@ -31,4 +31,5 @@ pub mod course; pub mod history; pub mod item; pub mod layout; +pub mod seal; pub mod taxonomy; diff --git a/src/model/layout.rs b/src/model/layout.rs index de454cf..ddc0c24 100644 --- a/src/model/layout.rs +++ b/src/model/layout.rs @@ -76,6 +76,15 @@ impl Layout { self.root.join("templates") } + /// Directory holding sealed administrations. + /// + /// Deliberately not `assessments/`: + /// [`crate::assessment::AssessmentFile::load_all`] parses every `.yaml` in + /// that directory, and a seal is not an assessment record. + pub fn seals(&self) -> PathBuf { + self.root.join(crate::seal::SEAL_DIR) + } + /// Creates every directory in the layout. /// /// # Errors diff --git a/src/model/seal.rs b/src/model/seal.rs new file mode 100644 index 0000000..dbf6170 --- /dev/null +++ b/src/model/seal.rs @@ -0,0 +1,1310 @@ +// SPDX-License-Identifier: Prosperity-3.0.0 +// Copyright Scientific Computing Studio +// Source: https://git.scient.ing/education/coursebank + +//! The sealed administration: what the students actually held in their hands. +//! +//! An assessment record says which item sat at question 14. It does not say what +//! that item *said*, and it does not say that on form B question 14's option C was +//! the bank's option A. Both facts are derivable — the first from the bank, the +//! second from the form seed — right up until the bank changes, and then they are +//! silently derivable to the wrong answer. +//! +//! A seal freezes them. It is written once, before the exam is administered, and +//! it holds three things the record does not: +//! +//! * **The content as administered.** Stem and option text, verbatim, with a +//! fingerprint on each. Reword a distractor next term and the seal still shows +//! what this cohort was asked. +//! * **The permutation, expanded.** For every form and every printed question, the +//! map from printed letter to bank letter, plus the key as printed. This is what +//! turns "the student chose C" into "the student chose the bank's option A", +//! which is the only form of that sentence worth storing. +//! * **Digests over both.** One per form, one per item, and one over the whole +//! file. Re-derive any of them and compare; a mismatch names what moved. +//! +//! The seal is not a cache. Nothing reads it for speed. It exists so that the +//! question "is what I am about to analyze the same thing the students saw?" has +//! an answer that does not depend on the bank having stayed still. +//! +//! # Where it lives +//! +//! `seals/.yaml`, in its own directory rather than beside the +//! assessment record, because [`crate::assessment::AssessmentFile::load_all`] +//! parses every `.yaml` in `assessments/` and a seal is not an assessment. +//! +//! # The order of operations +//! +//! ```text +//! assemble ──▶ export typst ──▶ seal ──▶ print ──▶ administer ──▶ ingest +//! │ │ +//! └────────── verified against ──────┘ +//! ``` +//! +//! Seal after exporting and before printing. Sealing earlier is harmless; +//! sealing after ingest is not, because a seal written from a drifted bank +//! records the drift as truth. + +use std::collections::{BTreeMap, BTreeSet}; +use std::path::{Path, PathBuf}; + +use serde::{Deserialize, Serialize}; + +use crate::assessment::{AssessmentFile, Form, Placement}; +use crate::catalog::{Catalog, Severity}; +use crate::course::SCHEMA_VERSION; +use crate::date::Date; +use crate::error::{Error, Result}; +use crate::hash::{fingerprint, hex, sha256}; +use crate::item::Item; +use crate::layout::Layout; +use crate::select; +use crate::taxonomy::Level; +use crate::yaml; + +/// The directory holding seals, relative to the course root. +pub const SEAL_DIR: &str = "seals"; + +/// The prefix on a digest string, naming the algorithm so a future change is +/// visible in the file rather than inferred from the length. +const DIGEST_PREFIX: &str = "sha256:"; + +/// A whole seal file. +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct SealFile { + /// Schema version this file targets. + #[serde( + default = "default_version", + deserialize_with = "yaml::flexible_string" + )] + pub schema_version: String, + + /// Identity, provenance, and the digest over everything else. + pub seal: SealMeta, + + /// The items as administered, by recorded question number. + #[serde(default)] + pub items: Vec, + + /// One entry per form, holding that form's expanded permutation. + #[serde(default)] + pub forms: Vec, +} + +/// Identity and provenance. +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct SealMeta { + /// The assessment id this seals. + pub assessment: String, + /// The assessment title, denormalized so the file reads standalone. + pub title: String, + /// The course code. + pub course: String, + /// The term. + pub term: String, + /// The administration date, when the record carried one. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub date: Option, + /// The day the seal was written. + pub sealed_on: Date, + /// The tool and version that wrote it. + pub generator: Generator, + /// Whether option and stem text were stored, or only their fingerprints. + #[serde(default = "yes")] + pub content: bool, + /// Recorded question numbers that were already dropped when the seal was + /// written, and therefore were not printed. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub dropped: Vec, + /// The digest over every item and form below. + /// + /// Written last and checked first. [`SealFile::recompute_digest`] rebuilds it + /// from the file's own contents; [`SealFile::verify`] rebuilds a seal from the + /// live course and compares this against it. + pub digest: String, +} + +/// What wrote a seal. +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct Generator { + /// The tool name. + pub tool: String, + /// The tool version. + pub version: String, +} + +/// One item, frozen as administered. +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct SealedItem { + /// The recorded question number, which is the join key to grading exports. + pub number: u32, + /// The item's global id. + pub item: String, + /// The item version as administered. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub version: Option, + /// The item's content fingerprint, the same one + /// [`crate::item::Item::fingerprint`] computes, so a seal and a bank can be + /// compared without re-hashing either by hand. + pub fingerprint: String, + /// Points as administered. + pub points: f64, + /// Whether it was scored as bonus. + #[serde(default, skip_serializing_if = "is_false")] + pub bonus: bool, + /// The cognitive level as administered. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub level: Option, + /// Objectives as administered. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub learning_objectives: Vec, + /// The keyed letters in the bank's own lettering, before any shuffle. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub key: Vec, + /// The stem as administered. Absent when the seal was written without content. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub stem: Option, + /// The options in bank order. + #[serde(default)] + pub options: Vec, +} + +/// One option, frozen as administered. +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct SealedOption { + /// The letter this option carries in the bank. + pub letter: String, + /// Whether it was keyed correct. + #[serde(default, skip_serializing_if = "is_false")] + pub correct: bool, + /// Credit it earned, when the bank set one explicitly. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub credit: Option, + /// A fingerprint of the option text, so drift is detectable even when the + /// seal was written without content. + pub digest: String, + /// The option text as administered. Absent when the seal was written without + /// content. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub text: Option, +} + +/// One form, with its permutation expanded rather than implied by a seed. +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct SealedForm { + /// The form id, e.g. `A`. + pub id: String, + /// The seed the permutation came from, kept so it can be re-derived. + pub seed: u64, + /// Whether item order was permuted. + #[serde(default)] + pub shuffle_items: bool, + /// Whether option order was permuted. + #[serde(default)] + pub shuffle_options: bool, + /// A digest over this form's question list alone, so a single form can be + /// checked without reading the rest of the file. + pub digest: String, + /// The printed questions, in printed order. + #[serde(default)] + pub questions: Vec, +} + +/// One question as printed on one form. +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct SealedQuestion { + /// Where it sat on the page, counting from 1. This is what a grading export + /// names its file after. + pub position: u32, + /// The recorded question number, which is what the assessment record and the + /// response store use. + pub number: u32, + /// The item's global id, repeated here so a form block reads on its own. + pub item: String, + /// The keyed letters *as printed on this form*. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub printed_key: Vec, + /// The letter map, in printed order: `printed` is what the student saw, + /// `canonical` is the bank's letter for the same option. + #[serde(default)] + pub options: Vec, +} + +/// One printed letter and the bank letter it stands for. +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct LetterMap { + /// The letter as printed on this form. + pub printed: String, + /// The letter the same option carries in the bank. + pub canonical: String, +} + +/// What to put in a seal. +#[derive(Debug, Clone)] +pub struct Options { + /// Whether to store stem and option text, not just fingerprints. + /// + /// On by default. The whole point of a seal is to answer "what did they + /// actually see" after the bank has moved on, and a fingerprint answers only + /// "not this". + pub content: bool, + /// Which forms to seal. Empty means every form the record declares, or a + /// single unshuffled form `A` when it declares none. + pub forms: Vec, +} + +impl Default for Options { + fn default() -> Options { + Options { + content: true, + forms: Vec::new(), + } + } +} + +/// One way a live course has drifted from a seal. +#[derive(Debug, Clone)] +pub struct Drift { + /// How much it matters. + pub severity: Severity, + /// A machine-readable code, so a check can be silenced or grepped. + pub code: &'static str, + /// What moved, in a sentence. + pub message: String, +} + +impl Drift { + /// Builds a drift finding. + /// + /// # Arguments + /// + /// * `severity` - how much it matters. + /// * `code` - the stable code. + /// * `message` - the explanation. + /// + /// # Returns + /// + /// The finding. + fn new(severity: Severity, code: &'static str, message: impl Into) -> Drift { + Drift { + severity, + code, + message: message.into(), + } + } + + /// Whether this finding should stop an analysis rather than annotate it. + pub fn is_blocking(&self) -> bool { + self.severity == Severity::High + } +} + +impl std::fmt::Display for Drift { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "[{}] {}", self.code, self.message) + } +} + +/// The printed label for a zero-based option position. +/// +/// This mirrors [`crate::typst::config::LetterStyle::Upper`], which is the +/// lettering every bundled exam template uses and the only lettering a Gradescope +/// rubric column can carry. A course that prints numeric or Roman option labels +/// must map them back to letters before ingest; the seal always speaks letters. +/// +/// # Arguments +/// +/// * `position` - the printed position, counting from zero. +/// +/// # Returns +/// +/// `A`, `B`, ... `Z`, `AA`. +pub fn printed_letter(position: usize) -> String { + let mut n = position; + let mut letters = Vec::new(); + loop { + letters.push((b'A' + (n % 26) as u8) as char); + if n < 26 { + break; + } + n = n / 26 - 1; + } + letters.iter().rev().collect() +} + +/// The forms to seal, resolving an empty request and a record with no forms. +/// +/// # Arguments +/// +/// * `record` - the assessment record. +/// * `wanted` - form ids, or empty for all of them. +/// +/// # Returns +/// +/// The forms, in the record's order. +/// +/// # Errors +/// +/// Returns [`Error::Usage`] when a named form is not declared. +fn forms_to_seal(record: &AssessmentFile, wanted: &[String]) -> Result> { + if record.forms.is_empty() { + // A record with no declared forms was still printed once, and that single + // printing is a form: unshuffled, seed zero. Sealing it costs nothing and + // means the ingest path has one shape rather than two. + return Ok(vec![Form { + id: "A".to_string(), + seed: 0, + shuffle_items: false, + shuffle_options: false, + }]); + } + if wanted.is_empty() { + return Ok(record.forms.clone()); + } + let mut out = Vec::new(); + for id in wanted { + let form = record + .forms + .iter() + .find(|f| f.id.eq_ignore_ascii_case(id)) + .ok_or_else(|| { + Error::usage(format!( + "no form `{id}` on this assessment; it declares {}", + record + .forms + .iter() + .map(|f| f.id.as_str()) + .collect::>() + .join(", ") + )) + })?; + out.push(form.clone()); + } + Ok(out) +} + +/// Builds a seal from the live course. +/// +/// # Arguments +/// +/// * `catalog` - the loaded course. +/// * `record` - the assessment record. +/// * `opts` - what to include. +/// +/// # Returns +/// +/// The seal, with every digest computed. +/// +/// # Errors +/// +/// Returns [`Error::Unresolved`] when a placement references an item that is not +/// in the bank, and [`Error::Usage`] when a requested form is not declared. +pub fn build(catalog: &Catalog, record: &AssessmentFile, opts: &Options) -> Result { + let default_points = catalog.course.policy.points_per_item; + let forms = forms_to_seal(record, &opts.forms)?; + + let mut items = Vec::new(); + for placement in &record.items { + let entry = catalog.require(&placement.item)?; + items.push(sealed_item(placement, &entry.item, default_points, opts)); + } + items.sort_by_key(|i| i.number); + + let mut sealed_forms = Vec::new(); + for form in &forms { + sealed_forms.push(sealed_form(catalog, record, form)?); + } + + let dropped: Vec = record + .items + .iter() + .filter(|p| p.dropped) + .map(|p| p.number) + .collect(); + + let mut file = SealFile { + schema_version: SCHEMA_VERSION.to_string(), + seal: SealMeta { + assessment: record.assessment.id.clone(), + title: record.assessment.title.clone(), + course: catalog.course.course.code.clone(), + term: record + .assessment + .term + .clone() + .unwrap_or_else(|| catalog.course.course.term.clone()), + date: record.assessment.date, + sealed_on: Date::today(), + generator: Generator { + tool: "coursebank".to_string(), + version: crate::VERSION.to_string(), + }, + content: opts.content, + dropped, + digest: String::new(), + }, + items, + forms: sealed_forms, + }; + file.seal.digest = file.recompute_digest(); + Ok(file) +} + +/// Freezes one item. +fn sealed_item( + placement: &Placement, + item: &Item, + default_points: f64, + opts: &Options, +) -> SealedItem { + let options: Vec = item + .options + .iter() + .map(|choice| SealedOption { + letter: choice.id.clone(), + correct: choice.correct, + credit: choice.credit, + digest: fingerprint([choice.id.as_str(), choice.text.trim()]), + text: if opts.content { + Some(choice.text.clone()) + } else { + None + }, + }) + .collect(); + + SealedItem { + number: placement.number, + item: placement.item.clone(), + version: placement.version.or(Some(item.version)), + fingerprint: item.fingerprint(), + points: placement + .points + .unwrap_or_else(|| item.points(default_points)), + bonus: placement.bonus, + level: placement.level.or(Some(item.level)), + learning_objectives: if placement.learning_objectives.is_empty() { + item.learning_objectives.clone() + } else { + placement.learning_objectives.clone() + }, + key: if placement.key.is_empty() { + item.key_letters() + } else { + placement.key.clone() + }, + stem: if opts.content { + Some(item.stem.clone()) + } else { + None + }, + options, + } +} + +/// Expands one form's permutation. +/// +/// The expansion comes from [`select::layout`] and [`select::option_order`], the +/// same two functions every export calls, so a seal cannot describe a paper the +/// exporter would not have produced. +fn sealed_form(catalog: &Catalog, record: &AssessmentFile, form: &Form) -> Result { + let mut questions = Vec::new(); + + // Dropped placements are not printed, so they take no printed position. A + // drop recorded before sealing therefore shifts every later position, exactly + // as it shifts them on the page. + let printed: Vec = select::layout(record, form) + .into_iter() + .filter(|p| !p.dropped) + .collect(); + + for (index, placement) in printed.iter().enumerate() { + let entry = catalog.require(&placement.item)?; + let item = &entry.item; + let n = item.options.len(); + let order = select::option_order(form, &placement.item, n); + + let canonical_key: BTreeSet = if placement.key.is_empty() { + item.key_letters().into_iter().collect() + } else { + placement.key.iter().cloned().collect() + }; + + let mut options = Vec::with_capacity(n); + let mut printed_key = Vec::new(); + for (position, source_index) in order.iter().enumerate() { + let canonical = item + .options + .get(*source_index) + .map(|c| c.id.clone()) + .unwrap_or_else(|| printed_letter(*source_index)); + let printed = printed_letter(position); + if canonical_key.contains(&canonical) { + printed_key.push(printed.clone()); + } + options.push(LetterMap { printed, canonical }); + } + + questions.push(SealedQuestion { + position: index as u32 + 1, + number: placement.number, + item: placement.item.clone(), + printed_key, + options, + }); + } + + let mut sealed = SealedForm { + id: form.id.clone(), + seed: form.seed, + shuffle_items: form.shuffle_items, + shuffle_options: form.shuffle_options, + digest: String::new(), + questions, + }; + sealed.digest = form_digest(&sealed); + Ok(sealed) +} + +/// The digest over one form's question list. +fn form_digest(form: &SealedForm) -> String { + let mut buf = String::new(); + buf.push_str(&format!( + "form\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\n", + form.id, form.seed, form.shuffle_items, form.shuffle_options + )); + for q in &form.questions { + buf.push_str(&format!( + "q\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}", + q.position, + q.number, + q.item, + q.printed_key.join(",") + )); + for map in &q.options { + buf.push_str(&format!("{}>{};", map.printed, map.canonical)); + } + buf.push('\n'); + } + digest_of(&buf) +} + +/// A digest string over some canonical text. +fn digest_of(text: &str) -> String { + format!("{DIGEST_PREFIX}{}", hex(&sha256(text.as_bytes()))) +} + +impl SealFile { + /// The path a seal lives at. + /// + /// # Arguments + /// + /// * `layout` - the course layout. + /// * `assessment_id` - the assessment id. + /// + /// # Returns + /// + /// `seals/.yaml` under the course root. + pub fn path(layout: &Layout, assessment_id: &str) -> PathBuf { + layout + .root + .join(SEAL_DIR) + .join(format!("{assessment_id}.yaml")) + } + + /// Loads a seal. + /// + /// # Arguments + /// + /// * `path` - the YAML file. + /// + /// # Returns + /// + /// The parsed seal. + /// + /// # Errors + /// + /// Returns a load error. + pub fn load(path: &Path) -> Result { + yaml::read(path) + } + + /// Loads the seal for an assessment, if one has been written. + /// + /// Absence is not an error. A course that has never sealed anything should + /// still ingest, with the permutation derived from the record instead. + /// + /// # Arguments + /// + /// * `layout` - the course layout. + /// * `assessment_id` - the assessment id. + /// + /// # Returns + /// + /// The seal, or `None`. + /// + /// # Errors + /// + /// Returns a load error when the file exists but does not parse. + pub fn find(layout: &Layout, assessment_id: &str) -> Result> { + let path = SealFile::path(layout, assessment_id); + if path.is_file() { + Ok(Some(SealFile::load(&path)?)) + } else { + Ok(None) + } + } + + /// Writes the seal out. + /// + /// # Arguments + /// + /// * `path` - the destination. + /// + /// # Errors + /// + /// Returns [`Error::Io`] on a write failure. + pub fn save(&self, path: &Path) -> Result<()> { + yaml::write(path, self) + } + + /// One form's block. + /// + /// # Arguments + /// + /// * `id` - the form id, matched case-insensitively. + /// + /// # Returns + /// + /// The form, or `None`. + pub fn form(&self, id: &str) -> Option<&SealedForm> { + self.forms.iter().find(|f| f.id.eq_ignore_ascii_case(id)) + } + + /// One item's block, by recorded question number. + /// + /// # Arguments + /// + /// * `number` - the recorded number. + /// + /// # Returns + /// + /// The item, or `None`. + pub fn item(&self, number: u32) -> Option<&SealedItem> { + self.items.iter().find(|i| i.number == number) + } + + /// The canonical text this file's digest is computed over. + /// + /// Built field by field rather than by serializing the struct, so a change to + /// serde attributes, key order, or YAML quoting cannot invalidate every seal + /// ever written. Unit separators keep concatenation unambiguous. + /// + /// # Returns + /// + /// The digest input. + pub fn digest_input(&self) -> String { + let mut buf = String::new(); + buf.push_str(&format!( + "seal\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\n", + self.schema_version, self.seal.assessment, self.seal.course, self.seal.term + )); + for item in &self.items { + buf.push_str(&format!( + "item\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}", + item.number, + item.item, + item.version.unwrap_or(0), + item.fingerprint, + item.points, + item.bonus, + item.level.map(|l| l.code()).unwrap_or(0), + item.key.join(",") + )); + for option in &item.options { + buf.push_str(&format!( + "{}:{}:{};", + option.letter, option.correct, option.digest + )); + } + buf.push('\n'); + } + for form in &self.forms { + buf.push_str(&format!("formref\u{1f}{}\u{1f}{}\n", form.id, form.digest)); + } + buf + } + + /// Recomputes this file's digest from its own contents. + /// + /// # Returns + /// + /// The digest string. + pub fn recompute_digest(&self) -> String { + digest_of(&self.digest_input()) + } + + /// Checks that the file has not been edited since it was written. + /// + /// This catches a hand-edit of the seal itself, which is a different failure + /// from the bank drifting and deserves a different message. + /// + /// # Returns + /// + /// Findings, empty when the file is internally consistent. + pub fn check_self(&self) -> Vec { + let mut out = Vec::new(); + let expected = self.recompute_digest(); + if expected != self.seal.digest { + out.push(Drift::new( + Severity::High, + "seal-digest", + format!( + "the seal's own digest does not match its contents ({} recorded, {} \ + computed); the file was edited after it was written", + short(&self.seal.digest), + short(&expected) + ), + )); + } + for form in &self.forms { + let expected = form_digest(form); + if expected != form.digest { + out.push(Drift::new( + Severity::High, + "seal-form-digest", + format!( + "form {}'s permutation block was edited after sealing", + form.id + ), + )); + } + } + out + } + + /// Compares the seal against the live course. + /// + /// Everything this reports is a reason to distrust an analysis that pools the + /// sealed administration with anything else, and most of it is a reason to + /// distrust per-option feedback sent to a student. + /// + /// # Arguments + /// + /// * `catalog` - the loaded course, as it stands now. + /// * `record` - the assessment record, as it stands now. + /// + /// # Returns + /// + /// Findings, worst first, empty when nothing has moved. + pub fn verify(&self, catalog: &Catalog, record: &AssessmentFile) -> Vec { + let mut out = self.check_self(); + + let rebuilt = build( + catalog, + record, + &Options { + content: self.seal.content, + forms: self.forms.iter().map(|f| f.id.clone()).collect(), + }, + ); + let rebuilt = match rebuilt { + Ok(r) => r, + Err(e) => { + out.push(Drift::new( + Severity::High, + "seal-unbuildable", + format!("the course no longer produces this assessment at all: {e}"), + )); + return out; + } + }; + + if rebuilt.seal.digest == self.seal.digest { + return out; + } + + // The digests differ, so say what differs rather than that they do. + let sealed: BTreeMap = self.items.iter().map(|i| (i.number, i)).collect(); + let live: BTreeMap = + rebuilt.items.iter().map(|i| (i.number, i)).collect(); + + for (number, was) in &sealed { + let Some(now) = live.get(number) else { + out.push(Drift::new( + Severity::High, + "item-removed", + format!( + "question {number} ({}) is no longer on the assessment record", + was.item + ), + )); + continue; + }; + if was.item != now.item { + out.push(Drift::new( + Severity::High, + "item-replaced", + format!("question {number} was {} and is now {}", was.item, now.item), + )); + continue; + } + if was.fingerprint != now.fingerprint { + out.push(Drift::new( + Severity::High, + "item-edited", + format!( + "question {number} ({}) was edited after administration; the students saw \ + a different version of the stem or options", + was.item + ), + )); + } + if was.key != now.key { + out.push(Drift::new( + Severity::High, + "key-changed", + format!( + "question {number} ({}) was keyed {} and is now keyed {}", + was.item, + was.key.join(""), + now.key.join("") + ), + )); + } + if was.options.len() != now.options.len() { + out.push(Drift::new( + Severity::High, + "option-count-changed", + format!( + "question {number} ({}) had {} options and now has {}, so every printed \ + letter on every form means something else", + was.item, + was.options.len(), + now.options.len() + ), + )); + } + if was.points != now.points { + out.push(Drift::new( + Severity::Medium, + "points-changed", + format!( + "question {number} was worth {} point(s) and is now worth {}", + was.points, now.points + ), + )); + } + if was.learning_objectives != now.learning_objectives { + out.push(Drift::new( + Severity::Low, + "objectives-retagged", + format!( + "question {number} ({}) was retagged; per-objective results for this \ + administration were computed against the old tags", + was.item + ), + )); + } + if was.level != now.level { + out.push(Drift::new( + Severity::Low, + "level-changed", + format!("question {number} ({}) changed level", was.item), + )); + } + } + + for number in live.keys() { + if !sealed.contains_key(number) { + out.push(Drift::new( + Severity::Medium, + "item-added", + format!("question {number} was added to the record after sealing"), + )); + } + } + + for form in &self.forms { + let Some(now) = rebuilt.form(&form.id) else { + out.push(Drift::new( + Severity::High, + "form-removed", + format!("form {} is no longer declared on the record", form.id), + )); + continue; + }; + if now.digest != form.digest { + let reason = if now.seed != form.seed { + format!( + "its seed changed from {} to {}, so every option moved", + form.seed, now.seed + ) + } else if now.shuffle_options != form.shuffle_options + || now.shuffle_items != form.shuffle_items + { + "its shuffle settings changed".to_string() + } else { + "the items behind it changed".to_string() + }; + out.push(Drift::new( + Severity::High, + "form-permutation-changed", + format!( + "form {} no longer permutes the way it did when it was printed: {reason}", + form.id + ), + )); + } + } + + out.sort_by_key(|d| std::cmp::Reverse(d.severity)); + out + } +} + +/// How a form's printed answer key is distributed. +/// +/// A shuffle is uniform per item and says nothing about the sequence it produces +/// across a form. Over thirty-odd questions it will occasionally produce a run +/// students notice: six consecutive questions keying to `A` is unremarkable +/// probabilistically and very remarkable at a desk, where it reads as a mistake +/// and makes a student who is sure of all six doubt the last three. +/// +/// Sealing is the moment to look, because it is the last moment before printing. +/// Changing a form's seed reshuffles it; nothing else has to change. +#[derive(Debug, Clone)] +pub struct Balance { + /// The form id. + pub form: String, + /// How many questions key to each printed letter. + pub counts: BTreeMap, + /// The longest run of consecutive questions sharing a printed key, as + /// `(letter, first number, last number)`. + pub longest_run: Option<(String, u32, u32)>, + /// How long that run was. Stored rather than derived from the two numbers, + /// which are recorded numbers and need not be contiguous. + pub run: usize, + /// How many questions were counted. + pub total: usize, +} + +impl Balance { + /// The length of the longest run. + pub fn run_length(&self) -> usize { + self.run + } + + /// Advisories worth printing before the form is printed. + /// + /// Deliberately quiet. Both thresholds are set where the pattern stops being + /// something only a statistician would see: a run of five, and a letter + /// carrying more than twice its share. + /// + /// # Returns + /// + /// One message per observation, empty when the form looks unremarkable. + pub fn notes(&self) -> Vec { + let mut out = Vec::new(); + if self.total == 0 { + return out; + } + + if self.run_length() >= 5 { + if let Some((letter, first, last)) = &self.longest_run { + out.push(format!( + "form {}: questions {first} through {last} all key to {letter}. That is {} in \ + a row, which students read as an error in the key. Re-seed the form if you \ + have not printed it yet", + self.form, + self.run_length() + )); + } + } + + let share = self.total as f64 / self.counts.len().max(1) as f64; + for (letter, count) in &self.counts { + if *count as f64 > share * 2.0 && *count >= 4 { + out.push(format!( + "form {}: {count} of {} questions key to {letter}", + self.form, self.total + )); + } + } + out + } +} + +/// Measures one form's printed key. +/// +/// # Arguments +/// +/// * `form` - the sealed form. +/// +/// # Returns +/// +/// The balance. Questions with no single keyed letter, such as a +/// multiple-response item, are skipped rather than guessed at. +pub fn balance(form: &SealedForm) -> Balance { + let mut counts: BTreeMap = BTreeMap::new(); + let mut sequence: Vec<(u32, String)> = Vec::new(); + + for question in &form.questions { + if question.printed_key.len() != 1 { + continue; + } + let letter = question.printed_key[0].clone(); + *counts.entry(letter.clone()).or_insert(0) += 1; + sequence.push((question.number, letter)); + } + + let mut longest: Option<(String, u32, u32)> = None; + let mut best = 0usize; + let mut run = 1usize; + for i in 1..sequence.len() { + if sequence[i].1 == sequence[i - 1].1 { + run += 1; + if run > best { + best = run; + longest = Some(( + sequence[i].1.clone(), + sequence[i + 1 - run].0, + sequence[i].0, + )); + } + } else { + run = 1; + } + } + + Balance { + form: form.id.clone(), + total: sequence.len(), + counts, + longest_run: longest, + run: best.max(if sequence.is_empty() { 0 } else { 1 }), + } +} + +/// The first eight hex characters of a digest, for messages. +/// +/// # Arguments +/// +/// * `digest` - the full digest string. +/// +/// # Returns +/// +/// A short form such as `sha256:1a2b3c4d`. +pub fn short(digest: &str) -> String { + match digest.split_once(':') { + Some((algorithm, hex)) => { + format!("{algorithm}:{}", &hex[..hex.len().min(8)]) + } + None => digest.chars().take(8).collect(), + } +} + +fn default_version() -> String { + SCHEMA_VERSION.to_string() +} +fn yes() -> bool { + true +} +fn is_false(b: &bool) -> bool { + !*b +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn printed_letters_run_past_z() { + assert_eq!(printed_letter(0), "A"); + assert_eq!(printed_letter(3), "D"); + assert_eq!(printed_letter(25), "Z"); + assert_eq!(printed_letter(26), "AA"); + } + + fn sample() -> SealFile { + let mut file = SealFile { + schema_version: "1.0".into(), + seal: SealMeta { + assessment: "e1".into(), + title: "Exam 1".into(), + course: "BIOSC 1000".into(), + term: "2026f".into(), + date: None, + sealed_on: Date::today(), + generator: Generator { + tool: "coursebank".into(), + version: "test".into(), + }, + content: true, + dropped: Vec::new(), + digest: String::new(), + }, + items: vec![SealedItem { + number: 1, + item: "b::q-1".into(), + version: Some(1), + fingerprint: "abc".into(), + points: 1.0, + bonus: false, + level: Some(Level::Remember), + learning_objectives: vec!["lo-a".into()], + key: vec!["B".into()], + stem: Some("Stem.".into()), + options: vec![ + SealedOption { + letter: "A".into(), + correct: false, + credit: None, + digest: "d1".into(), + text: Some("First".into()), + }, + SealedOption { + letter: "B".into(), + correct: true, + credit: None, + digest: "d2".into(), + text: Some("Second".into()), + }, + ], + }], + forms: vec![SealedForm { + id: "A".into(), + seed: 7, + shuffle_items: false, + shuffle_options: true, + digest: String::new(), + questions: vec![SealedQuestion { + position: 1, + number: 1, + item: "b::q-1".into(), + printed_key: vec!["A".into()], + options: vec![ + LetterMap { + printed: "A".into(), + canonical: "B".into(), + }, + LetterMap { + printed: "B".into(), + canonical: "A".into(), + }, + ], + }], + }], + }; + file.forms[0].digest = form_digest(&file.forms[0]); + file.seal.digest = file.recompute_digest(); + file + } + + #[test] + fn a_fresh_seal_is_internally_consistent() { + assert!(sample().check_self().is_empty()); + } + + #[test] + fn editing_the_file_breaks_its_digest() { + let mut file = sample(); + file.items[0].key = vec!["A".into()]; + let findings = file.check_self(); + assert!( + findings.iter().any(|d| d.code == "seal-digest"), + "{findings:?}" + ); + } + + #[test] + fn editing_a_permutation_breaks_that_forms_digest() { + let mut file = sample(); + file.forms[0].questions[0].options[0].canonical = "A".into(); + let findings = file.check_self(); + assert!( + findings.iter().any(|d| d.code == "seal-form-digest"), + "{findings:?}" + ); + } + + #[test] + fn the_digest_is_order_sensitive() { + let file = sample(); + let mut swapped = file.clone(); + swapped.items[0].options.swap(0, 1); + assert_ne!(file.recompute_digest(), swapped.recompute_digest()); + } + + fn form_keyed(letters: &[&str]) -> SealedForm { + SealedForm { + id: "B".into(), + seed: 1, + shuffle_items: false, + shuffle_options: true, + digest: String::new(), + questions: letters + .iter() + .enumerate() + .map(|(i, letter)| SealedQuestion { + position: i as u32 + 1, + number: i as u32 + 1, + item: format!("b::q-{i}"), + printed_key: vec![letter.to_string()], + options: Vec::new(), + }) + .collect(), + } + } + + #[test] + fn a_long_run_of_one_letter_is_named_with_its_question_numbers() { + // Form B of a real exam did this: questions 2 through 7 all keyed to A. + let form = form_keyed(&["C", "A", "A", "A", "A", "A", "A", "C", "B"]); + let balance = balance(&form); + assert_eq!(balance.run_length(), 6); + assert_eq!( + balance.longest_run, + Some(("A".to_string(), 2, 7)), + "the run is reported by question number, not by index" + ); + let notes = balance.notes(); + assert!(notes.iter().any(|n| n.contains("2 through 7")), "{notes:?}"); + } + + #[test] + fn an_ordinary_form_says_nothing() { + let form = form_keyed(&["A", "B", "C", "D", "A", "C", "B", "D", "C", "A", "D", "B"]); + assert!(balance(&form).notes().is_empty()); + } + + #[test] + fn a_form_with_no_questions_does_not_panic() { + let balance = balance(&form_keyed(&[])); + assert_eq!(balance.run_length(), 0); + assert!(balance.notes().is_empty()); + } + + #[test] + fn short_digests_keep_the_algorithm() { + assert_eq!(short("sha256:0123456789abcdef"), "sha256:01234567"); + } + + #[test] + fn round_trips_through_yaml() { + let file = sample(); + let text = serde_yaml_ng::to_string(&file).unwrap(); + let back: SealFile = serde_yaml_ng::from_str(&text).unwrap(); + assert_eq!(back.seal.digest, file.seal.digest); + assert!(back.check_self().is_empty()); + assert_eq!(back.form("a").map(|f| f.seed), Some(7)); + } +}