From 5ac1e317c0ad6297b32bc53f38dbbea3656b2d3d Mon Sep 17 00:00:00 2001 From: Alex Maldonado Date: Sat, 26 Sep 2026 18:22:25 -0400 Subject: [PATCH] feat: cooked up something fierce --- src/analysis/calibrate.rs | 280 ++++++++++++++++++++++++++++++- src/analysis/classical.rs | 29 ++++ src/analysis/diagnostic.rs | 2 +- src/analysis/students.rs | 1 + src/authoring/jsonschema.rs | 91 +++++++++- src/authoring/lint.rs | 162 +++++++++++++++++- src/authoring/select.rs | 154 ++++++++++++++++- src/cli.rs | 11 ++ src/commands/project.rs | 19 +++ src/data/canvas.rs | 1 + src/data/decode.rs | 7 +- src/data/gradescope.rs | 1 + src/data/responses.rs | 89 ++++++++++ src/data/store.rs | 1 + src/data/store_parquet.rs | 7 + src/export/practice.rs | 15 +- src/export/qti.rs | 20 ++- src/export/site.rs | 14 +- src/export/typst.rs | 4 + src/export/typst/payload.rs | 5 +- src/migrate.rs | 65 ++++++++ src/model/assessment.rs | 41 +++++ src/model/bank.rs | 65 ++++++-- src/model/catalog.rs | 119 +++++++++++++- src/model/item.rs | 319 ++++++++++++++++++++++++++++++++++++ src/model/seal.rs | 7 +- 26 files changed, 1485 insertions(+), 44 deletions(-) diff --git a/src/analysis/calibrate.rs b/src/analysis/calibrate.rs index 880a87b..e171085 100644 --- a/src/analysis/calibrate.rs +++ b/src/analysis/calibrate.rs @@ -35,7 +35,7 @@ use crate::classical::{self, Analysis, ItemAnalysis, Thresholds}; use crate::date::Date; use crate::error::{Error, Result}; use crate::irt::{self, Fit}; -use crate::item::{Calibration, IrtParams, OptionStat}; +use crate::item::{Calibration, IrtParams, Item, OptionStat, VariantCalibration}; use crate::responses::ResponseSet; use crate::store::Store; use crate::taxonomy::Flag; @@ -267,6 +267,35 @@ pub fn plan(catalog: &Catalog, store: &Store, opts: &Options) -> Result { )); } + // Splitting by option set means an item administered three times with three + // different sets has three cells of 24 rather than one of 72. That is the + // honest picture, and it is worth saying out loud rather than leaving + // someone to read an IRT fit that was never possible. + for (uid, appearances) in &by_item { + let mut sizes: Vec = Vec::new(); + for (_, analysis) in appearances { + if analysis.variant.is_some() { + sizes.push(analysis.n); + } + } + if sizes.len() > 1 { + let distinct: BTreeSet<&str> = appearances + .iter() + .filter_map(|(_, a)| a.variant.as_deref()) + .collect(); + if distinct.len() > 1 { + warnings.push(format!( + "{uid}: {} option sets across {} administrations, largest n = {}. \ + Statistics are kept per set, because a stem shown with different \ + distractors is a different item.", + distinct.len(), + appearances.len(), + sizes.iter().copied().max().unwrap_or(0) + )); + } + } + } + let mut changes = Vec::new(); for (uid, appearances) in &by_item { @@ -304,6 +333,12 @@ pub fn plan(catalog: &Catalog, store: &Store, opts: &Options) -> Result { option_stats: pooled.option_stats.clone(), irt: irt_params, flags: pooled.flags.clone(), + variants: variant_records(&entry.item, appearances, previous_variants(entry)), + options: BTreeMap::new(), + }; + let calibration = Calibration { + options: option_histories(&calibration.variants), + ..calibration }; let previous = entry.item.calibration.as_ref(); @@ -703,6 +738,16 @@ pub fn plan_from_analysis( .collect(), irt: irt_params, flags: item.flags.clone(), + variants: variant_records( + &entry.item, + &[(admin.clone(), item.clone())], + previous_variants(entry), + ), + options: BTreeMap::new(), + }; + let calibration = Calibration { + options: option_histories(&calibration.variants), + ..calibration }; let previous = entry.item.calibration.as_ref(); @@ -730,6 +775,171 @@ pub fn plan_from_analysis( } } +/// The variant records an item already has, to be merged with the new ones. +fn previous_variants(entry: &crate::catalog::Entry) -> Vec { + entry + .item + .calibration + .as_ref() + .map(|c| c.variants.clone()) + .unwrap_or_default() +} + +/// Builds one calibration record per option set the item was administered in. +/// +/// The records this pass computes replace the stored ones for the same variant +/// and leave the rest alone. That is what makes a partial recalibration safe: +/// a pass given only this term's data must not silently discard the numbers for +/// an option set that was retired two terms ago. +/// +/// Statistics come from the same [`pool`] used for the flat summary, so the two +/// agree for an item whose pool is its form — the case every pre-2.0 item is in. +/// +/// # Arguments +/// +/// * `item` - the bank item. +/// * `appearances` - the administration id and analysis of each appearance. +/// * `previous` - the records already stored. +/// +/// # Returns +/// +/// The merged records, in variant order. +fn variant_records( + item: &Item, + appearances: &[(String, ItemAnalysis)], + previous: Vec, +) -> Vec { + // Appearances whose administration mixed two option sets under one question + // number carry no variant, and there is no set for them to describe. + let mut by_variant: BTreeMap> = BTreeMap::new(); + for (admin, analysis) in appearances { + if let Some(variant) = &analysis.variant { + by_variant + .entry(variant.clone()) + .or_default() + .push((admin.clone(), analysis.clone())); + } + } + + let mut merged: BTreeMap = previous + .into_iter() + .map(|v| (v.variant.clone(), v)) + .collect(); + + for (variant, group) in by_variant { + let pooled = pool(&group); + // The option set is recovered from the digest's own record when the + // stored one has it, and from the options that were actually chosen + // otherwise, so a record written from data alone still says what it + // describes. + let (key, distractors) = describe(item, &variant, &group, merged.get(&variant)); + merged.insert( + variant.clone(), + VariantCalibration { + variant, + key, + distractors, + administrations: group.iter().map(|(a, _)| a.clone()).collect(), + n_examinees: Some(pooled.n), + p_value: Some(round4(pooled.p_value)), + point_biserial: pooled.point_biserial.map(round4), + discrimination_index: pooled.discrimination_index.map(round4), + option_stats: pooled.option_stats.clone(), + irt: None, + flags: pooled.flags.clone(), + }, + ); + } + + merged.into_values().collect() +} + +/// Which options a variant administered. +/// +/// Prefers what a stored record already says. Failing that, the options that +/// appear in the statistics are the ones students saw, and the item says which +/// of those are keyed. +fn describe( + item: &Item, + variant: &str, + group: &[(String, ItemAnalysis)], + stored: Option<&VariantCalibration>, +) -> (Vec, Vec) { + if let Some(stored) = stored { + if !stored.key.is_empty() + && item.variant_digest(&stored.key, &stored.distractors) == variant + { + return (stored.key.clone(), stored.distractors.clone()); + } + } + + let mut seen: BTreeSet = BTreeSet::new(); + for (_, analysis) in group { + seen.extend(analysis.options.keys().cloned()); + } + let keyed: BTreeSet = group + .iter() + .flat_map(|(_, a)| a.key.iter().cloned()) + .collect(); + + let key: Vec = seen + .iter() + .filter(|o| keyed.contains(*o)) + .cloned() + .collect(); + let distractors: Vec = seen + .iter() + .filter(|o| !keyed.contains(*o)) + .cloned() + .collect(); + (key, distractors) +} + +/// Summarizes what each option has done across every set it appeared in. +/// +/// Deliberately coarse, because selection rates are shares of a fixed set and +/// averaging them across different sets is not a statistic. The one claim this +/// view supports is the one worth having: an option that draws nobody in any +/// set it has appeared in is not doing anything, and that is the evidence for +/// retiring it — evidence a single administration cannot provide. +/// +/// # Arguments +/// +/// * `variants` - the per-variant records. +/// +/// # Returns +/// +/// One history per option id. +fn option_histories( + variants: &[VariantCalibration], +) -> BTreeMap { + let mut out: BTreeMap = BTreeMap::new(); + let mut rates: BTreeMap> = BTreeMap::new(); + + for variant in variants { + let n = variant.n_examinees.unwrap_or(0); + for (option, stat) in &variant.option_stats { + let entry = out.entry(option.clone()).or_default(); + entry.appearances += 1; + entry.n_examinees += n; + if let Some(rate) = stat.selection_rate { + rates.entry(option.clone()).or_default().push(rate); + } + } + } + + for (option, entry) in &mut out { + if let Some(seen) = rates.get(option) { + if !seen.is_empty() { + entry.mean_selection_rate = + Some(round4(seen.iter().sum::() / seen.len() as f64)); + entry.never_chosen = seen.iter().all(|r| *r <= f64::EPSILON); + } + } + } + out +} + /// Rounds to four decimals. fn round4(x: f64) -> f64 { (x * 1e4).round() / 1e4 @@ -754,9 +964,77 @@ mod tests { option_stats: BTreeMap::new(), irt: None, flags: Vec::new(), + variants: Vec::new(), + options: BTreeMap::new(), } } + fn stat(rate: f64) -> OptionStat { + OptionStat { + selection_rate: Some(rate), + point_biserial: None, + upper_group_rate: None, + lower_group_rate: None, + } + } + + fn variant(id: &str, n: usize, dead_rate: f64) -> VariantCalibration { + VariantCalibration { + variant: id.into(), + n_examinees: Some(n), + option_stats: [ + ("o-key".to_string(), stat(0.8)), + ("o-dead".to_string(), stat(dead_rate)), + ] + .into_iter() + .collect(), + ..VariantCalibration::default() + } + } + + #[test] + fn an_option_that_draws_nobody_is_only_visible_across_sets() { + let histories = option_histories(&[variant("v1", 50, 0.0), variant("v2", 46, 0.0)]); + + let dead = &histories["o-dead"]; + assert_eq!(dead.appearances, 2); + assert_eq!(dead.n_examinees, 96); + // The claim the cross-variant view exists to support, and the one a + // single administration cannot make. + assert!(dead.never_chosen); + + assert!(!histories["o-key"].never_chosen); + assert_eq!(histories["o-key"].mean_selection_rate, Some(0.8)); + + // One set where it drew is enough to stop the claim. + let mixed = option_histories(&[variant("v1", 50, 0.0), variant("v2", 46, 0.04)]); + assert!(!mixed["o-dead"].never_chosen); + } + + #[test] + fn a_pass_with_no_data_for_an_item_keeps_the_records_it_has() { + let item: Item = serde_yaml_ng::from_str( + r#"id: q-x +status: approved +level: 1 +stem: s +options: + - { id: o-key, text: right, correct: true } + - { id: o-one, text: wrong } +"#, + ) + .expect("item parses"); + + let kept = variant_records(&item, &[], vec![variant("v-old", 96, 0.0)]); + assert_eq!( + kept.len(), + 1, + "a pass given nothing must not delete history" + ); + assert_eq!(kept[0].variant, "v-old"); + assert_eq!(kept[0].n_examinees, Some(96)); + } + #[test] fn a_first_calibration_is_all_new() { let next = calibration(0.7, Some(0.3), 24, "abc"); diff --git a/src/analysis/classical.rs b/src/analysis/classical.rs index 468ba36..233307b 100644 --- a/src/analysis/classical.rs +++ b/src/analysis/classical.rs @@ -134,6 +134,14 @@ pub struct ItemAnalysis { pub number: u32, /// The item's global id, when known. pub item_ref: Option, + /// The option set administered, when the rows agree on one. + /// + /// Within one administration an item has one variant, because a placement's + /// distractors are drawn once and shared by every form — only the printed + /// order differs. `None` means the rows disagreed, which happens when a + /// course opts into drawing distractors per form; the statistics below then + /// describe a mixture and cannot be pooled by option set. + pub variant: Option, /// How many students the item was administered to. pub n: usize, /// How many gave a non-blank response. @@ -553,6 +561,7 @@ pub fn analyze( item_ref: record .and_then(|r| r.placement(*number)) .map(|p| p.item.clone()), + variant: one_variant(set, *number), n, n_answered, blank_rate: blank as f64 / responded as f64, @@ -906,6 +915,25 @@ fn reliability(coded: &[Vec], totals: &[f64], p_values: &[f64], rpbs: &[f64 /// # Returns /// /// The letters that appear on full-credit responses. +/// The single variant every row for one question names, if they agree. +/// +/// Disagreement is not an error, it is a fact about the administration: a course +/// that draws distractors per form has two option sets under one question +/// number, and no pooled statistic describes both. Returning `None` is what +/// keeps the per-variant records from claiming otherwise. +fn one_variant(set: &ResponseSet, number: u32) -> Option { + let mut seen: Option<&str> = None; + for row in set.rows.iter().filter(|r| r.item_number == number) { + let variant = row.variant.as_deref()?; + match seen { + None => seen = Some(variant), + Some(first) if first == variant => {} + Some(_) => return None, + } + } + seen.map(str::to_string) +} + fn infer_key(rows: &[&crate::responses::Response]) -> Vec { let mut out: BTreeSet = BTreeSet::new(); for r in rows { @@ -1001,6 +1029,7 @@ mod tests { item_number: number, item_ref: None, item_version: None, + variant: None, selected: if letter.is_empty() { vec![] } else { diff --git a/src/analysis/diagnostic.rs b/src/analysis/diagnostic.rs index f8cfc10..75810ea 100644 --- a/src/analysis/diagnostic.rs +++ b/src/analysis/diagnostic.rs @@ -1680,7 +1680,7 @@ pub fn cohort( .map(|option| { let count = responses .iter() - .filter(|r| r.chosen().iter().any(|l| *l == option.id)) + .filter(|r| r.chosen().contains(&option.id)) .count(); OptionRow { text: Some(option.text.clone()), diff --git a/src/analysis/students.rs b/src/analysis/students.rs index 45b25d5..1b448ff 100644 --- a/src/analysis/students.rs +++ b/src/analysis/students.rs @@ -1363,6 +1363,7 @@ learning_objectives: item_number: number, item_ref: None, item_version: None, + variant: None, selected: vec!["A".into()], selected_source: vec![], eliminated: vec![], diff --git a/src/authoring/jsonschema.rs b/src/authoring/jsonschema.rs index 0796b59..fd3c4fb 100644 --- a/src/authoring/jsonschema.rs +++ b/src/authoring/jsonschema.rs @@ -749,7 +749,8 @@ fn option_schema() -> Value { }, "selection_rate_expected": proportion( "How often you expect this to be chosen. Compared against reality." - ) + ), + "retired": retirement_schema() } }) } @@ -944,6 +945,21 @@ fn calibration_schema() -> Value { "additionalProperties": option_stat_schema() }, "irt": irt_schema(), + "variants": { + "type": "array", + "description": "One record per option set ever administered. A stem shown with \ + different distractors is a different item, so a p-value pooled \ + across both would average two questions.", + "items": variant_calibration_schema() + }, + "options": { + "type": "object", + "description": "One record per option, pooled across every set it appeared in. \ + Supports one claim — this option draws nobody, anywhere — which \ + is what retires a distractor and what one administration cannot \ + show.", + "additionalProperties": option_history_schema() + }, "flags": { "type": "array", "items": { "type": "string", "enum": strings(&flags) } @@ -985,6 +1001,63 @@ fn history_schema() -> Value { }) } +/// The schema for one option set's statistics. +fn variant_calibration_schema() -> Value { + json!({ + "type": "object", + "required": ["variant"], + "additionalProperties": false, + "properties": { + "variant": { + "type": "string", + "description": "Digest of the stem, the administered options, and which was keyed." + }, + "key": string_array("The option ids keyed correct in this set."), + "distractors": string_array("The option ids offered alongside them."), + "administrations": string_array("The administrations pooled into these numbers."), + "n_examinees": { "type": "integer", "minimum": 0 }, + "p_value": proportion("Proportion correct, for this option set only."), + "point_biserial": { "type": "number", "minimum": -1.0, "maximum": 1.0 }, + "discrimination_index": { "type": "number", "minimum": -1.0, "maximum": 1.0 }, + "option_stats": { + "type": "object", + "description": "Per-option behaviour within this set, by option id.", + "additionalProperties": option_stat_schema() + }, + "irt": irt_schema(), + "flags": { + "type": "array", + "items": { + "type": "string", + "enum": strings( + &Flag::ALL.iter().map(|f| f.as_str()).collect::>() + ) + } + } + } + }) +} + +/// The schema for one option's cross-variant history. +fn option_history_schema() -> Value { + json!({ + "type": "object", + "additionalProperties": false, + "properties": { + "appearances": { "type": "integer", "minimum": 0 }, + "n_examinees": { "type": "integer", "minimum": 0 }, + "mean_selection_rate": proportion( + "Mean of the within-variant rates. For reading, not for acting on: each rate is \ + a share of a different set." + ), + "never_chosen": { + "type": "boolean", + "description": "Never chosen, anywhere. The claim that justifies retiring it." + } + } + }) +} + /// The schema for a retirement record. fn retirement_schema() -> Value { json!({ @@ -1289,7 +1362,21 @@ fn placement_schema() -> Value { }, "points": { "type": "number", "minimum": 0.0 }, "bonus": { "type": "boolean" }, - "key": string_array("Keyed option letters as administered."), + "key": string_array( + "The option ids keyed correct for this administration. One for a \ + single_best_answer, chosen from the item's pool of defensible keys." + ), + "distractors": string_array( + "The option ids offered alongside the key. Resolved when the assessment is \ + assembled and written out explicitly, so a later bank edit cannot change the \ + paper. Empty means the whole pool." + ), + "variant": { + "type": "string", + "description": "Digest of the item as this administration showed it: stem, \ + administered options, and which was keyed. The key statistics \ + pool on." + }, "level": level(), "learning_targets": string_array("Targets as administered."), "credit_overrides": { diff --git a/src/authoring/lint.rs b/src/authoring/lint.rs index 6760b5f..70accb4 100644 --- a/src/authoring/lint.rs +++ b/src/authoring/lint.rs @@ -88,6 +88,16 @@ pub enum Rule { /// An expectation of low discrimination on a higher-level item. ContradictoryDesign, + // --- the option pool --- + /// Fewer usable distractors than a form shows. + ThinOptionPool, + /// An option that has never been administered and has not been retired. + UnusedOption, + /// An option retired without saying what it did. + UnjustifiedRetirement, + /// A distractor that has never been chosen, in any set it appeared in. + NonfunctioningDistractor, + // --- evidence --- /// Statistics describe an older version of the item. StaleCalibration, @@ -104,7 +114,7 @@ impl Rule { /// /// Used by `--list-rules`, and by the test that keeps this list in step with /// the enum. - pub const ALL: [Rule; 26] = [ + pub const ALL: [Rule; 30] = [ Rule::KeyIsLongest, Rule::UnevenOptionLength, Rule::WordRepeatCue, @@ -127,6 +137,10 @@ impl Rule { Rule::WeakFormatForLevel, Rule::ScoredBonusLevel, Rule::ContradictoryDesign, + Rule::ThinOptionPool, + Rule::UnusedOption, + Rule::UnjustifiedRetirement, + Rule::NonfunctioningDistractor, Rule::StaleCalibration, Rule::DifficultyMissed, Rule::DiscriminationMissed, @@ -152,6 +166,9 @@ impl Rule { // Statistics attached to text that has since changed are actively // misleading, which is worse than absent. R::StaleCalibration => Severity::High, + // An item that cannot fill a form is an item `assemble` will put on + // a paper short an option. + R::ThinOptionPool => Severity::High, // An unanswerable question for a screen-reader user. R::AssetWithoutAltText => Severity::High, @@ -178,6 +195,13 @@ impl Rule { | R::NoStudentFeedback | R::DifficultyMissed | R::DiscriminationMissed => Severity::Low, + + // A distractor that draws nobody across several administrations is + // evidence to act on, not a style note. + R::NonfunctioningDistractor => Severity::Medium, + // Both are tidiness: the item still works, but its pool is + // carrying something nobody has accounted for. + R::UnusedOption | R::UnjustifiedRetirement => Severity::Low, } } @@ -206,6 +230,10 @@ impl Rule { Rule::WeakFormatForLevel => "complete-format-level", Rule::ScoredBonusLevel => "complete-bonus-policy", Rule::ContradictoryDesign => "complete-design-conflict", + Rule::ThinOptionPool => "pool-thin", + Rule::UnusedOption => "pool-unused", + Rule::UnjustifiedRetirement => "pool-unjustified-retirement", + Rule::NonfunctioningDistractor => "evidence-nonfunctioning", Rule::StaleCalibration => "evidence-stale", Rule::DifficultyMissed => "evidence-difficulty", Rule::DiscriminationMissed => "evidence-discrimination", @@ -238,9 +266,11 @@ impl Rule { | Rule::WeakFormatForLevel | Rule::ScoredBonusLevel | Rule::ContradictoryDesign => "completeness", + Rule::ThinOptionPool | Rule::UnusedOption | Rule::UnjustifiedRetirement => "pool", Rule::StaleCalibration | Rule::DifficultyMissed | Rule::DiscriminationMissed + | Rule::NonfunctioningDistractor | Rule::DuplicateStem => "evidence", } } @@ -270,6 +300,10 @@ impl Rule { Rule::WeakFormatForLevel, Rule::ScoredBonusLevel, Rule::ContradictoryDesign, + Rule::ThinOptionPool, + Rule::UnusedOption, + Rule::UnjustifiedRetirement, + Rule::NonfunctioningDistractor, Rule::StaleCalibration, Rule::DifficultyMissed, Rule::DiscriminationMissed, @@ -302,6 +336,12 @@ impl Rule { Rule::WeakFormatForLevel => "true/false at an analytic level", Rule::ScoredBonusLevel => "a level the policy reserves for bonus is scored", Rule::ContradictoryDesign => "low expected discrimination on a higher-level item", + Rule::ThinOptionPool => "fewer usable distractors than a form shows", + Rule::UnusedOption => "an option has never been administered and is not retired", + Rule::UnjustifiedRetirement => "an option was retired without saying what it did", + Rule::NonfunctioningDistractor => { + "a distractor has never been chosen in any set it appeared in" + } Rule::StaleCalibration => "statistics describe an older version of the item", Rule::DifficultyMissed => "observed difficulty was far from predicted", Rule::DiscriminationMissed => "observed discrimination contradicted the prediction", @@ -749,6 +789,87 @@ pub fn lint_item(entry: &Entry, course: &CourseFile, t: &Thresholds) -> Vec { + if retirement.reason.trim().len() < 12 { + push( + Rule::UnjustifiedRetirement, + Severity::Low, + format!( + "option `{}` is retired with no real reason. The reason is the \ + finding — what it drew, or failed to draw — and it is the only \ + part of a retirement worth anything in two years", + option.id + ), + ); + } + } + None => { + // An option nobody has been shown is a draft, and a draft + // sitting in an approved item's pool will eventually be + // drawn onto a paper without ever having been reviewed + // against data. + if let Some(history) = it + .calibration + .as_ref() + .filter(|c| !c.options.is_empty()) + .and_then(|c| c.options.get(&option.id)) + { + if history.never_chosen && history.appearances > 1 { + push( + Rule::NonfunctioningDistractor, + Severity::Medium, + format!( + "option `{}` has appeared in {} option set(s) across {} \ + examinees and has never been chosen. One administration \ + would not show this; several do.", + option.id, history.appearances, history.n_examinees + ), + ); + } + } else if it + .calibration + .as_ref() + .is_some_and(|c| !c.options.is_empty()) + { + push( + Rule::UnusedOption, + Severity::Low, + format!( + "option `{}` has never been administered. Either it is waiting \ + its turn, or it was drafted and forgotten — retire it and say \ + which.", + option.id + ), + ); + } + } + } + } + let _ = keys; + } + // --- evidence if !it.calibration_is_current() { push( @@ -1215,6 +1336,45 @@ mod tests { c } + #[test] + fn a_pool_too_thin_to_fill_a_form_is_flagged() { + let c = codes( + r#" +id: q-a-001 +status: draft +level: 2 +stem: Which mechanism best explains the sigmoidal binding curve? +options: + - { id: o-shift, text: Ligand binding shifts the tetramer to a higher-affinity state, correct: true } + - { id: o-fixed, text: Each subunit binds with the same fixed affinity throughout } +"#, + ); + // Policy shows four options and the pool can supply two, so `assemble` + // would put this on a paper two short. + assert!(c.contains(&"pool-thin"), "{c:?}"); + } + + #[test] + fn a_retirement_with_no_finding_is_flagged() { + let c = codes( + r#" +id: q-a-001 +status: draft +level: 2 +stem: Which mechanism best explains the sigmoidal binding curve? +options: + - { id: o-shift, text: Ligand binding shifts the tetramer to a higher-affinity state, correct: true } + - { id: o-fixed, text: Each subunit binds with the same fixed affinity throughout } + - { id: o-consumed, text: "Ligand is consumed as it binds, depleting the available pool" } + - { id: o-oxidation, text: The heme iron changes oxidation state upon binding } + - { id: o-cooperative, text: Subunits bind independently of one another, retired: { 'on': 2026-09-20, reason: bad } } +"#, + ); + assert!(c.contains(&"pool-unjustified-retirement"), "{c:?}"); + // Four live distractors is enough for a four-option form. + assert!(!c.contains(&"pool-thin"), "{c:?}"); + } + #[test] fn clean_item_passes() { let c = codes( diff --git a/src/authoring/select.rs b/src/authoring/select.rs index 286b607..398a0b8 100644 --- a/src/authoring/select.rs +++ b/src/authoring/select.rs @@ -33,8 +33,9 @@ use crate::course::{CourseFile, SCHEMA_VERSION}; use crate::date::Date; use crate::error::{Error, Result}; use crate::history::History; +use crate::item::{Choice, Item}; use crate::rng::Rng; -use crate::taxonomy::Level; +use crate::taxonomy::{Format, Level}; /// The result of a draw. #[derive(Debug, Clone)] @@ -466,15 +467,23 @@ pub fn to_record( .chain(selection.bonus.iter().map(|u| (u, true))), ) { let e = catalog.require(uid)?; + let (key, distractors) = draw_options( + &e.item, + catalog.course.policy.options_per_item, + blueprint.seed.unwrap_or(0), + uid, + ); items.push(Placement { number, item: uid.clone(), version: None, stem_digest: Some(e.item.stem_digest()), + variant: Some(e.item.variant_digest(&key, &distractors)), fingerprint: Some(e.item.fingerprint()), points: Some(e.item.points(default_points)), bonus: is_bonus || e.item.bonus, - key: e.item.key_letters(), + distractors, + key, level: Some(e.item.level), learning_targets: e.item.learning_targets.clone(), credit_overrides: BTreeMap::new(), @@ -571,6 +580,85 @@ pub fn layout(record: &AssessmentFile, form: &Form) -> Vec { scored.into_iter().chain(bonus).collect() } +/// Draws the key and the distractors one placement administers. +/// +/// Resolved here, at assembly, and written into the record as explicit lists. +/// Nothing downstream samples: an export that drew its own options would print +/// a different paper every time the bank was touched. +/// +/// The draw is seeded on the blueprint and the item, so re-running `assemble` +/// with the same seed produces the same paper, and two items in one assessment +/// draw independently. +/// +/// # Arguments +/// +/// * `item` - the item, whose options are a pool. +/// * `per_item` - how many options a form shows, from course policy. +/// * `seed` - the blueprint seed. +/// * `uid` - the item id, salting the draw. +/// +/// # Returns +/// +/// The keyed ids and the distractor ids, each sorted, naming options of `item`. +/// Both empty for an item with no options, which is an open response. +pub fn draw_options( + item: &Item, + per_item: usize, + seed: u64, + uid: &str, +) -> (Vec, Vec) { + let (keys, distractors) = item.pool(); + if keys.is_empty() && distractors.is_empty() { + return (Vec::new(), Vec::new()); + } + + // Multiple response keys every correct option; anything else keys one, and + // when the pool offers several defensible keys the draw picks one so that + // the record says which. + let wanted_keys = match item.format { + Format::MultipleResponse => keys.len(), + _ => 1.min(keys.len()), + }; + let mut rng = Rng::from_label(&format!("{seed}/{uid}/options")); + + let mut key_ids = pick(&keys, wanted_keys, &mut rng); + key_ids.sort(); + + // A pool with fewer usable distractors than the policy asks for is a + // finding, not a failure: the form comes out short and `lint` says so, + // rather than `assemble` refusing to build the assessment at all. + let wanted = per_item.saturating_sub(key_ids.len()); + let mut distractor_ids = pick(&distractors, wanted.min(distractors.len()), &mut rng); + distractor_ids.sort(); + + (key_ids, distractor_ids) +} + +/// Takes `n` options, preferring the ones that were designed rather than merely +/// written. +/// +/// A distractor carrying a misconception and an error type is one you thought +/// about; one carrying neither is filler. When the pool is larger than the form, +/// the thought-about ones go on the paper. The shuffle comes first so that +/// options of equal standing are drawn by seed rather than by declaration +/// order. +fn pick(options: &[&Choice], n: usize, rng: &mut Rng) -> Vec { + if n >= options.len() { + return options.iter().map(|o| o.id.clone()).collect(); + } + let mut order: Vec = (0..options.len()).collect(); + rng.shuffle(&mut order); + order.sort_by_key(|&i| { + let o = options[i]; + u8::from(o.misconception.is_none()) + u8::from(o.error_type.is_none()) + }); + order + .into_iter() + .take(n) + .map(|i| options[i].id.clone()) + .collect() +} + /// The option order for one item on one form. /// /// # Arguments @@ -693,6 +781,68 @@ mod tests { assert_eq!(form_label(27), "AB"); } + /// An item whose options are given as YAML, so the test needs no literal. + fn pool_item(options: &str) -> Item { + let src = format!( + r#"id: q-x +status: approved +level: 1 +cognitive_process: recall +stem: Which line holds the quality scores? +learning_targets: [t-x] +sources: [{{ lecture: L1 }}] +options: +{options}"# + ); + serde_yaml_ng::from_str(&src).expect("item parses") + } + + const DESIGNED: &str = r#" - { id: o-key, text: right, correct: true } + - { id: o-designed-a, text: a, misconception: mistakes the separator, error_type: recall_confusion } + - { id: o-designed-b, text: b, misconception: confuses the two, error_type: recall_confusion } + - { id: o-filler-a, text: c } + - { id: o-filler-b, text: d } +"#; + + #[test] + fn a_draw_prefers_designed_distractors_and_is_reproducible() { + let item = pool_item(DESIGNED); + + let (key, distractors) = draw_options(&item, 3, 1103, "q-x"); + assert_eq!(key, vec!["o-key".to_string()]); + assert_eq!(distractors.len(), 2); + // Thought-about distractors go on the paper before filler does. + assert!( + distractors.iter().all(|d| d.starts_with("o-designed")), + "{distractors:?}" + ); + + // Same seed, same paper. + assert_eq!(draw_options(&item, 3, 1103, "q-x"), (key, distractors)); + + // A retired option is not drawn, and the form comes out of the rest. + let retired = pool_item(&DESIGNED.replace( + "{ id: o-designed-a, text: a,", + "{ id: o-designed-a, text: a, retired: { 'on': 2026-09-20, reason: nonfunctioning },", + )); + let (_, after) = draw_options(&retired, 3, 1103, "q-x"); + assert!(!after.iter().any(|d| d == "o-designed-a"), "{after:?}"); + } + + #[test] + fn a_thin_pool_comes_out_short_rather_than_refusing_to_build() { + let item = pool_item( + " - { id: o-key, text: right, correct: true }\n - { id: o-one, text: wrong }\n", + ); + let (key, distractors) = draw_options(&item, 4, 7, "q-y"); + assert_eq!(key.len(), 1); + assert_eq!( + distractors.len(), + 1, + "one usable distractor, so one is drawn" + ); + } + #[test] fn option_order_is_a_reproducible_permutation() { let form = Form { diff --git a/src/cli.rs b/src/cli.rs index 5f11f19..ea3e3f7 100644 --- a/src/cli.rs +++ b/src/cli.rs @@ -137,6 +137,17 @@ pub(crate) enum MigrateCommand { #[arg(long)] dry_run: bool, }, + /// Fill in the stored `variant` column from the assessment records. + /// + /// Nothing in the tool needs it — a variant is derived from the placement + /// when a row has none. It is for pandas, DuckDB, and R, which see only + /// what is in the column and will otherwise average two option sets of one + /// stem into an item that never existed. + Variants { + /// Show what would change, and write nothing. + #[arg(long)] + dry_run: bool, + }, /// Drop `version:` and `history:`, which 2.0 ignores. /// /// A stem's text is its identity: reword it and it is a new item with a new diff --git a/src/commands/project.rs b/src/commands/project.rs index 5e912e8..4235799 100644 --- a/src/commands/project.rs +++ b/src/commands/project.rs @@ -108,9 +108,28 @@ pub(crate) fn migrate(cli: &Cli, sub: &MigrateCommand) -> Result { MigrateCommand::Ids { dry_run } => migrate_ids(cli, *dry_run), MigrateCommand::Options { dry_run } => migrate_options(cli, *dry_run), MigrateCommand::Stems { dry_run } => migrate_stems(cli, *dry_run), + MigrateCommand::Variants { dry_run } => migrate_variants(cli, *dry_run), } } +/// Fills in the stored variant column. +fn migrate_variants(cli: &Cli, dry_run: bool) -> Result { + let touched = migrate::store_variants(&cli.course, !dry_run)?; + if touched.is_empty() { + println!("nothing to fill in: every stored row already names its variant"); + return Ok(Outcome::Ok); + } + for (path, n) in &touched { + println!(" {:<44} {n:>5} row(s)", path.display()); + } + if dry_run { + println!("\nnothing written"); + return Ok(Outcome::Ok); + } + println!("\nrewrote {} data file(s)", touched.len()); + Ok(Outcome::Ok) +} + /// Drops the version fields 2.0 ignores. fn migrate_stems(cli: &Cli, dry_run: bool) -> Result { let touched = migrate::stems(&cli.course, !dry_run)?; diff --git a/src/data/canvas.rs b/src/data/canvas.rs index 5e94a62..a6ce18a 100644 --- a/src/data/canvas.rs +++ b/src/data/canvas.rs @@ -337,6 +337,7 @@ pub fn ingest( item_number: number, item_ref, item_version: None, + variant: None, selected, selected_source: Vec::new(), eliminated: Vec::new(), diff --git a/src/data/decode.rs b/src/data/decode.rs index e5d75af..447ae8f 100644 --- a/src/data/decode.rs +++ b/src/data/decode.rs @@ -248,7 +248,8 @@ impl FormDecoder { for (index, placement) in printed.iter().enumerate() { let entry = catalog.require(&placement.item)?; let item = &entry.item; - let order = select::option_order(form, &placement.item, item.options.len()); + let shown = item.administered(&placement.key, &placement.distractors); + let order = select::option_order(form, &placement.item, shown.len()); let canonical_key: BTreeSet = if placement.key.is_empty() { item.key_letters().into_iter().collect() @@ -260,8 +261,7 @@ impl FormDecoder { let mut to_printed = BTreeMap::new(); let mut printed_key = Vec::new(); for (position, source_index) in order.iter().enumerate() { - let canonical = item - .options + let canonical = shown .get(*source_index) .map(|c| c.id.clone()) .unwrap_or_else(|| printed_letter(*source_index)); @@ -738,6 +738,7 @@ mod tests { form_position: None, item_ref: None, item_version: None, + variant: None, selected: vec![selected.into()], eliminated: Vec::new(), selected_source: Vec::new(), diff --git a/src/data/gradescope.rs b/src/data/gradescope.rs index b4b497d..57f5914 100644 --- a/src/data/gradescope.rs +++ b/src/data/gradescope.rs @@ -645,6 +645,7 @@ pub fn to_responses(questions: &[Question], ctx: &Context) -> Import { item_number: q.number, item_ref: None, item_version: None, + variant: None, selected, selected_source: Vec::new(), eliminated, diff --git a/src/data/responses.rs b/src/data/responses.rs index e639460..9a92440 100644 --- a/src/data/responses.rs +++ b/src/data/responses.rs @@ -75,6 +75,16 @@ pub struct Response { /// The item version as administered. pub item_version: Option, + /// The variant administered: which option set this row's student saw. + /// + /// The grouping key for pooled statistics. Set at ingest from the + /// assessment record, and stored so that anything reading the Parquet + /// without this tool can group the same way — a store that only has + /// `item_ref` cannot tell two option sets of one stem apart, and will + /// average them. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub variant: Option, + /// Option letters the student chose. pub selected: Vec, /// The selected options in the bank's own lettering, written at ingest by @@ -314,6 +324,49 @@ impl ResponseSet { per_item.values().sum() } + /// How many rows each option set of each item has. + /// + /// What a calibration report needs to be honest about sample size: an item + /// administered three times with three different option sets has three + /// cells, not one, and reporting "n = 72" of it would be wrong three ways. + /// + /// # Returns + /// + /// Row counts keyed by item id and variant, with an empty variant for rows + /// that carry none. + pub fn variants(&self) -> BTreeMap<(String, String), usize> { + let mut out: BTreeMap<(String, String), usize> = BTreeMap::new(); + for row in &self.rows { + let Some(item) = &row.item_ref else { continue }; + let key = (item.clone(), row.variant.clone().unwrap_or_default()); + *out.entry(key).or_insert(0) += 1; + } + out + } + + /// The rows for one option set of one item. + /// + /// # Arguments + /// + /// * `item_ref` - the item id. + /// * `variant` - the variant digest, or `None` for rows carrying none. + /// + /// # Returns + /// + /// A set holding only those rows, keeping the warnings of the original. + pub fn for_variant(&self, item_ref: &str, variant: Option<&str>) -> ResponseSet { + ResponseSet { + rows: self + .rows + .iter() + .filter(|r| r.item_ref.as_deref() == Some(item_ref)) + .filter(|r| r.variant.as_deref() == variant) + .cloned() + .collect(), + warnings: self.warnings.clone(), + } + } + /// Builds the response matrix for psychometrics. /// /// # Arguments @@ -416,6 +469,9 @@ impl ResponseSet { }; r.item_ref = Some(p.item.clone()); r.item_version = p.version; + // Recorded when the record says so; derived from the option set + // otherwise, which is the case for every administration before 2.0. + r.variant = p.variant.clone(); r.bonus = r.bonus || p.bonus; r.dropped = r.dropped || p.dropped; r.dropped_full_credit = r.dropped_full_credit || p.dropped_with_credit(); @@ -434,6 +490,9 @@ impl ResponseSet { if let Some(cat) = catalog { if let Some(entry) = cat.get(&p.item) { + if r.variant.is_none() { + r.variant = Some(p.variant_of(&entry.item)); + } r.level = Some(entry.item.level); r.learning_targets = if p.learning_targets.is_empty() { entry.item.learning_targets.clone() @@ -636,6 +695,8 @@ pub struct FlatResponse { pub item_ref: String, /// The item version, 0 when unknown. pub item_version: u32, + /// The administered variant, empty when unknown. + pub variant: String, /// Comma-joined selected letters. pub selected: String, /// Comma-joined eliminated letters. @@ -710,6 +771,7 @@ impl FlatResponse { item_number: r.item_number, item_ref: r.item_ref.clone().unwrap_or_default(), item_version: r.item_version.unwrap_or(0), + variant: r.variant.clone().unwrap_or_default(), selected: r.selected.join(","), selected_source: r.selected_source.join(","), eliminated: r.eliminated.join(","), @@ -771,6 +833,7 @@ impl FlatResponse { // one ingested after it instead of splitting into two items. item_ref: none_if_empty(&self.item_ref) .map(|id| crate::item::canonical_id(&id).to_string()), + variant: none_if_empty(&self.variant), item_version: if self.item_version == 0 { None } else { @@ -828,6 +891,7 @@ mod tests { item_number: number, item_ref: None, item_version: None, + variant: None, selected: vec!["A".into()], selected_source: vec![], eliminated: vec![], @@ -846,6 +910,31 @@ mod tests { } } + #[test] + fn rows_group_by_the_option_set_they_administered() { + let mut set = ResponseSet::new(); + for (student, variant) in [("s1", "v1"), ("s2", "v1"), ("s3", "v2")] { + let mut r = row(student, 1, 1.0); + r.item_ref = Some("q-x".into()); + r.variant = Some(variant.into()); + set.rows.push(r); + } + // A row from before the column existed. + let mut old = row("s4", 1, 1.0); + old.item_ref = Some("q-x".into()); + set.rows.push(old); + + let counts = set.variants(); + assert_eq!(counts[&("q-x".to_string(), "v1".to_string())], 2); + assert_eq!(counts[&("q-x".to_string(), "v2".to_string())], 1); + // Unknown groups on its own rather than joining either set. + assert_eq!(counts[&("q-x".to_string(), String::new())], 1); + + assert_eq!(set.for_variant("q-x", Some("v1")).rows.len(), 2); + assert_eq!(set.for_variant("q-x", None).rows.len(), 1); + assert_eq!(set.for_variant("q-other", Some("v1")).rows.len(), 0); + } + #[test] fn matrix_is_students_by_items() { let mut set = ResponseSet::new(); diff --git a/src/data/store.rs b/src/data/store.rs index 063d513..55e8cae 100644 --- a/src/data/store.rs +++ b/src/data/store.rs @@ -614,6 +614,7 @@ mod tests { item_number: number, item_ref: Some("bank::q-x-001".into()), item_version: Some(2), + variant: None, selected: vec!["C".into()], selected_source: vec![], eliminated: vec![], diff --git a/src/data/store_parquet.rs b/src/data/store_parquet.rs index 51ff891..d5a7abf 100644 --- a/src/data/store_parquet.rs +++ b/src/data/store_parquet.rs @@ -49,6 +49,7 @@ pub fn schema() -> Schema { Field::new("item_number", DataType::UInt32, false), Field::new("item_ref", DataType::Utf8, false), Field::new("item_version", DataType::UInt32, false), + Field::new("variant", DataType::Utf8, false), Field::new("selected", DataType::Utf8, false), Field::new("eliminated", DataType::Utf8, false), Field::new("correct", DataType::Utf8, false), @@ -114,6 +115,7 @@ fn to_batch(rows: &[FlatResponse]) -> Result { u32c(|r| r.item_number), s(|r| &r.item_ref), u32c(|r| r.item_version), + s(|r| &r.variant), s(|r| &r.selected), s(|r| &r.eliminated), s(|r| &r.correct), @@ -279,6 +281,9 @@ fn from_batch(batch: &RecordBatch, path: &Path) -> Result> { let item_number = uints("item_number")?; let item_ref = strings("item_ref")?; let item_version = uints("item_version")?; + // Added after the first stores were written, so absent rather than fatal in + // a file from before 2.0; `coursebank migrate variants` fills it in. + let variant = optional_strings("variant"); let selected = strings("selected")?; let eliminated = strings("eliminated")?; let correct = strings("correct")?; @@ -314,6 +319,7 @@ fn from_batch(batch: &RecordBatch, path: &Path) -> Result> { item_number: item_number.value(i), item_ref: item_ref.value(i).to_string(), item_version: item_version.value(i), + variant: variant.map(|c| c.value(i).to_string()).unwrap_or_default(), selected: selected.value(i).to_string(), selected_source: selected_source .map(|a| a.value(i).to_string()) @@ -366,6 +372,7 @@ mod tests { item_number: number, item_ref: "bank::q-a-001".into(), item_version: 3, + variant: "4c81fa".into(), selected: "C".into(), eliminated: String::new(), correct: "1".into(), diff --git a/src/export/practice.rs b/src/export/practice.rs index 841198b..d9a5b13 100644 --- a/src/export/practice.rs +++ b/src/export/practice.rs @@ -289,7 +289,7 @@ fn worksheet_question( out.push_str("\n\n"); if item.has_options() { - let ordered = ordered_options(item, form, &placement.item); + let ordered = ordered_options(item, placement, form); for (position, source) in ordered.iter().enumerate() { out.push_str(&format!( "{}. {}\n", @@ -320,7 +320,7 @@ fn solution_question( out.push_str("\n\n"); if item.has_options() { - let ordered = ordered_options(item, form, &placement.item); + let ordered = ordered_options(item, placement, form); for (position, source) in ordered.iter().enumerate() { let mark = if source.correct { " ✓" } else { "" }; let note = source @@ -458,10 +458,11 @@ fn meta_line(placement: &Placement, item: &Item) -> String { /// Salted with the item's global id, the same value the Typst and QTI exports use, /// so a worksheet built for form B lists options in the order that form's paper and /// its Canvas quiz do. -fn ordered_options<'a>(item: &'a Item, form: &Form, uid: &str) -> Vec<&'a Choice> { - select::option_order(form, uid, item.options.len()) +fn ordered_options<'a>(item: &'a Item, placement: &Placement, form: &Form) -> Vec<&'a Choice> { + let shown = item.administered(&placement.key, &placement.distractors); + select::option_order(form, &placement.item, shown.len()) .into_iter() - .map(|i| &item.options[i]) + .map(|i| shown[i]) .collect() } @@ -579,6 +580,8 @@ items: item: "l11::q-enthalpy-001".into(), version: None, stem_digest: None, + distractors: Vec::new(), + variant: None, fingerprint: None, points: Some(1.0), bonus: false, @@ -595,6 +598,8 @@ items: item: "l11::q-enthalpy-op-001".into(), version: None, stem_digest: None, + distractors: Vec::new(), + variant: None, fingerprint: None, points: Some(2.0), bonus: false, diff --git a/src/export/qti.rs b/src/export/qti.rs index c2646b7..11030cc 100644 --- a/src/export/qti.rs +++ b/src/export/qti.rs @@ -476,10 +476,12 @@ pub fn build(catalog: &Catalog, record: &AssessmentFile, opts: &QtiOptions) -> R if !placement.bonus { total_points += points; } + let shown = item.administered(&placement.key, &placement.distractors); items.push(build_item( &record.assessment.id, &placement.item, item, + &shown, points, opts, )); @@ -564,14 +566,21 @@ pub fn build(catalog: &Catalog, record: &AssessmentFile, opts: &QtiOptions) -> R /// # Returns /// /// The element. -fn build_item(assessment_id: &str, uid: &str, item: &Item, points: f64, opts: &QtiOptions) -> Node { +fn build_item( + assessment_id: &str, + uid: &str, + item: &Item, + shown: &[&crate::item::Choice], + points: f64, + opts: &QtiOptions, +) -> Node { // An open-response item is an essay in Canvas: no choices, graded by hand. if !item.format.has_options() { return build_essay_item(assessment_id, uid, item, points, opts); } - let order = select::option_order(&opts.form, uid, item.options.len()); - let ordered: Vec<&crate::item::Choice> = order.iter().map(|i| &item.options[*i]).collect(); + let order = select::option_order(&opts.form, uid, shown.len()); + let ordered: Vec<&crate::item::Choice> = order.iter().map(|i| shown[*i]).collect(); // Option identifiers are numeric, mirroring Canvas's own exports, and are // derived from the item id so they survive regeneration. @@ -1259,6 +1268,7 @@ mod tests { defense: None, feedback_student: None, selection_rate_expected: None, + retired: None, } } @@ -1390,6 +1400,8 @@ items: item: "b::q-mcq".into(), version: None, stem_digest: None, + distractors: Vec::new(), + variant: None, fingerprint: None, points: Some(1.0), bonus: false, @@ -1406,6 +1418,8 @@ items: item: "b::q-open".into(), version: None, stem_digest: None, + distractors: Vec::new(), + variant: None, fingerprint: None, points: Some(2.0), bonus: false, diff --git a/src/export/site.rs b/src/export/site.rs index d1e23fb..d7be2e1 100644 --- a/src/export/site.rs +++ b/src/export/site.rs @@ -226,12 +226,13 @@ fn question_block( if item.has_options() { b.push_str(":::: {.q-choices}\n"); - let order = select::option_order(form, &placement.item, item.options.len()); + let shown = item.administered(&placement.key, &placement.distractors); + let order = select::option_order(form, &placement.item, shown.len()); for (position, &source) in order.iter().enumerate() { b.push_str(&format!( "{}. {}\n", position + 1, - markup::to_markdown(&item.options[source].text) + markup::to_markdown(&shown[source].text) )); } b.push_str("::::\n\n"); @@ -295,11 +296,12 @@ fn fragment( /// A single-best-answer or multiple-response fragment: the key, the model answer, /// the explanation, then per-distractor feedback. fn choice_fragment(course: &CourseFile, placement: &Placement, item: &Item, form: &Form) -> String { - let order = select::option_order(form, &placement.item, item.options.len()); + let shown = item.administered(&placement.key, &placement.distractors); + let order = select::option_order(form, &placement.item, shown.len()); let printed: Vec<(usize, &Choice)> = order .iter() .enumerate() - .map(|(position, &source)| (position, &item.options[source])) + .map(|(position, &source)| (position, shown[source])) .collect(); let mut out = String::new(); @@ -844,6 +846,8 @@ items: item: "b::q-mcq".into(), version: None, stem_digest: None, + distractors: Vec::new(), + variant: None, fingerprint: None, points: Some(1.0), bonus: false, @@ -860,6 +864,8 @@ items: item: "b::q-open".into(), version: None, stem_digest: None, + distractors: Vec::new(), + variant: None, fingerprint: None, points: Some(2.0), bonus: false, diff --git a/src/export/typst.rs b/src/export/typst.rs index 641e70c..88e61c5 100644 --- a/src/export/typst.rs +++ b/src/export/typst.rs @@ -432,6 +432,8 @@ mod tests { item: "b::q-1".into(), version: None, stem_digest: None, + distractors: Vec::new(), + variant: None, fingerprint: None, points: None, bonus: false, @@ -448,6 +450,8 @@ mod tests { item: "b::q-2".into(), version: None, stem_digest: None, + distractors: Vec::new(), + variant: None, fingerprint: None, points: None, bonus: false, diff --git a/src/export/typst/payload.rs b/src/export/typst/payload.rs index da51acd..12d572f 100644 --- a/src/export/typst/payload.rs +++ b/src/export/typst/payload.rs @@ -411,11 +411,12 @@ pub fn build( _ => (None, None), }; - let order = select::option_order(form, &placement.item, item.options.len()); + let shown = item.administered(&placement.key, &placement.distractors); + let order = select::option_order(form, &placement.item, shown.len()); let options: Vec = order .iter() .enumerate() - .map(|(position, source_index)| option(&item.options[*source_index], position, config)) + .map(|(position, source_index)| option(shown[*source_index], position, config)) .collect(); let key = if config.reveal.shows_key() { diff --git a/src/migrate.rs b/src/migrate.rs index bfe7d0c..bf8b3b4 100644 --- a/src/migrate.rs +++ b/src/migrate.rs @@ -18,6 +18,7 @@ //! | [`store_ids`] | `item_ref` in the response store | so other readers see one id per item | //! | [`options_plan`] and [`apply_options`] | option letters into names | a letter is a position, not an identity | //! | [`stems`] | drops `version:` and `history:` | a stem's text is its identity | +//! | [`store_variants`] | fills the stored `variant` column | so other readers can group by option set | //! //! # What the option migration does not touch //! @@ -1200,6 +1201,70 @@ fn is_key(trimmed: &str, key: &str) -> bool { } } +/// Fills in the stored `variant` column from the assessment records. +/// +/// Nothing in the tool needs this: the reader derives a variant from the +/// placement when the row has none. It is for the other readers. The store is +/// Parquet precisely so pandas, DuckDB, and R can use it without this tool, and +/// a store that only carries `item_ref` cannot tell two option sets of one stem +/// apart — it will average them into one item that never existed. +/// +/// # Arguments +/// +/// * `root` - the course directory. +/// * `write` - whether to write, or only to count. +/// +/// # Returns +/// +/// One entry per file that gains variants, with how many rows it gains. +/// +/// # Errors +/// +/// Propagates catalog, record, and store failures. +pub fn store_variants(root: &Path, write: bool) -> Result> { + let layout = Layout::new(root); + let catalog = crate::catalog::Catalog::load(root)?; + + // (assessment, item number) -> variant. Keyed on the number rather than the + // item id: a record may place one item twice, and the row knows which. + let mut by_question: BTreeMap<(String, u32), String> = BTreeMap::new(); + for record in crate::assessment::AssessmentFile::load_all(&layout.assessments())? { + for placement in &record.items { + let Some(entry) = catalog.get(&placement.item) else { + continue; + }; + by_question.insert( + (record.assessment.id.clone(), placement.number), + placement.variant_of(&entry.item), + ); + } + } + + let store = Store::open(layout.data())?; + let mut out = Vec::new(); + for path in store.files()? { + let mut rows = store::read_flat(&path)?; + let mut touched = 0; + for row in &mut rows { + if !row.variant.is_empty() { + continue; + } + let key = (row.assessment_id.clone(), row.item_number); + if let Some(variant) = by_question.get(&key) { + row.variant = variant.clone(); + touched += 1; + } + } + if touched > 0 { + if write { + store::write_flat(&path, &rows)?; + } + out.push((relative(root, &path), touched)); + } + } + Ok(out) +} + /// One block of YAML: a key, the lines under it, and the comments above it. #[derive(Debug, Clone, Default)] struct Block { diff --git a/src/model/assessment.rs b/src/model/assessment.rs index cf46986..2c5336e 100644 --- a/src/model/assessment.rs +++ b/src/model/assessment.rs @@ -282,6 +282,26 @@ pub struct Placement { /// The keyed letters as administered. #[serde(default, skip_serializing_if = "Vec::is_empty")] pub key: Vec, + + /// The option ids offered alongside the key. + /// + /// Resolved when the assessment is assembled and written out explicitly, + /// never sampled at export time. A blueprint may ask for a draw; the record + /// holds what was drawn. Otherwise a bank edit between assembling and + /// printing silently changes the paper, and the key printed on Tuesday + /// disagrees with the one printed on Wednesday. + /// + /// Empty means the whole pool, which is what every pre-2.0 record meant. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub distractors: Vec, + + /// The digest of the item as this administration showed it. + /// + /// See [`crate::item::Item::variant_digest`]. The key statistics pool on: + /// two administrations of one stem with different distractors are two + /// items, and averaging them is averaging different questions. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub variant: Option, /// The level as administered, denormalized so a record reads standalone. #[serde(default, skip_serializing_if = "Option::is_none")] pub level: Option, @@ -357,6 +377,27 @@ pub enum DropStyle { } impl Placement { + /// The variant this placement administered. + /// + /// Recorded when the assessment was assembled; derived from the option set + /// otherwise, which is what makes every record written before 2.0 poolable + /// without being rewritten. A pre-2.0 placement names its key and no + /// distractors, which means the whole pool — a well-defined option set, and + /// so a well-defined variant. + /// + /// # Arguments + /// + /// * `item` - the item this placement names. + /// + /// # Returns + /// + /// The digest. + pub fn variant_of(&self, item: &crate::item::Item) -> String { + self.variant + .clone() + .unwrap_or_else(|| item.variant_digest(&self.key, &self.distractors)) + } + /// Whether this placement was dropped by crediting every option. /// /// # Returns diff --git a/src/model/bank.rs b/src/model/bank.rs index 6694305..b9456c3 100644 --- a/src/model/bank.rs +++ b/src/model/bank.rs @@ -403,6 +403,15 @@ fn validate_item( if seen.contains(&o.id.as_str()) { issues.push(format!("option {pos}: duplicate option id `{}`", o.id)); } + if let Some(retirement) = &o.retired { + if retirement.reason.trim().is_empty() { + issues.push(format!( + "option {pos}: retired without a reason. The reason is the finding — what \ + the option did or failed to do — and it is the only part of a retirement \ + that is worth anything later." + )); + } + } seen.push(&o.id); let credit = o.credit(); @@ -442,14 +451,27 @@ fn validate_item( } // --- key --- + // Counted over the pool that can still be drawn: a retired option is a + // record, not an offer. + let (live_keys, live_distractors) = it.pool(); let keys = it.key_indices(); match it.format { Format::SingleBestAnswer => { - if keys.len() != 1 { - issues.push(format!( - "single_best_answer needs exactly one keyed option, has {}", - keys.len() - )); + // Several defensible keys is a pool, not a bug — it is what lets you + // test whether "fourth" or "last of the four" is doing the work. + // Exactly one of them reaches a student, and that is the + // placement's business: see + // [`crate::catalog::Catalog::validate_record`]. + if live_keys.is_empty() { + issues.push("single_best_answer needs at least one keyed option".into()); + } + if live_distractors.is_empty() { + let retired = it.options.iter().any(|o| o.retired.is_some()); + issues.push(if retired { + "every distractor is retired, so nothing can be drawn against the key".into() + } else { + "has no option that is not keyed correct, so it asks nothing".to_string() + }); } } Format::MultipleResponse => { @@ -780,20 +802,43 @@ mod tests { options: - { id: A, text: a, correct: true } - { id: B, text: b, correct: true } + - id: q-a-003 + status: draft + level: 1 + format: single_best_answer + stem: s + options: + - { id: o-key-one, text: a, correct: true } + - { id: o-key-two, text: b, correct: true } + - { id: o-wrong-one, text: c } + - { id: o-wrong-two, text: d } "#, ); let issues = b.validate(None); + // No key at all is still a bank problem: nothing can be drawn from it. assert!( issues .iter() - .any(|i| i.contains("exactly one keyed option")) + .any(|i| i.starts_with("q-a-001") && i.contains("at least one keyed option")), + "{issues:?}" ); - assert_eq!( + // Keying every option is still a bank problem, for the older reason: a + // question with nothing to choose against asks nothing. + assert!( issues .iter() - .filter(|i| i.contains("exactly one keyed option")) - .count(), - 2 + .any(|i| i.starts_with("q-a-002") && i.contains("asks nothing")), + "{issues:?}" + ); + // Two defensible keys alongside real distractors is not a problem. It + // is the pool doing its job — it is what lets you test whether the + // wording of the key is what students are answering. Exactly one of + // them reaches a student, and that is checked against the placement + // that administers it, in `Catalog::validate_placement`. + assert!( + !issues.iter().any(|i| i.starts_with("q-a-003") + && (i.contains("keyed option") || i.contains("asks nothing"))), + "{issues:?}" ); } diff --git a/src/model/catalog.rs b/src/model/catalog.rs index 3abe0c3..f4fb175 100644 --- a/src/model/catalog.rs +++ b/src/model/catalog.rs @@ -20,13 +20,13 @@ use std::collections::{BTreeMap, BTreeSet}; use std::path::{Path, PathBuf}; -use crate::assessment::AssessmentFile; +use crate::assessment::{AssessmentFile, Placement}; use crate::bank::BankFile; use crate::course::CourseFile; use crate::error::{Error, Result}; use crate::item::Item; use crate::layout::Layout; -use crate::taxonomy::{Level, Status, Tier}; +use crate::taxonomy::{Format, Level, Status, Tier}; use crate::yaml; /// One item plus everything needed to locate it again. @@ -261,6 +261,115 @@ impl Catalog { Ok(issues) } + /// Checks one placement's option set against the item's pool. + /// + /// The checks that moved here from the bank when options became a pool. A + /// bank holding two defensible keys and six distractors is sound; what has + /// to hold for a *form* is that exactly one key reached the student, that + /// none of the distractors was true, and that the count matches policy. + /// None of that can be decided by looking at the item alone. + /// + /// # Arguments + /// + /// * `p` - the placement. + /// * `item` - the item it names. + /// + /// # Returns + /// + /// One message per problem. + fn validate_placement(&self, p: &Placement, item: &Item) -> Vec { + let mut issues = Vec::new(); + if !item.format.has_options() { + return issues; + } + let at = |number: u32| format!("question {number} ({})", p.item); + + for id in p.key.iter().chain(p.distractors.iter()) { + if item.option(id).is_none() { + issues.push(format!( + "{}: `{id}` is not an option of this item", + at(p.number) + )); + } + } + for id in &p.distractors { + if p.key.iter().any(|k| k == id) { + issues.push(format!( + "{}: `{id}` is listed as both the key and a distractor", + at(p.number) + )); + } + if item.option(id).is_some_and(|o| o.correct) { + issues.push(format!( + "{}: `{id}` is offered as a distractor but the bank keys it correct", + at(p.number) + )); + } + } + for id in &p.key { + if item.option(id).is_some_and(|o| !o.correct) { + issues.push(format!( + "{}: `{id}` is keyed correct here but the bank does not key it. An option is \ + true or it is not; which true option a form uses is this record's choice, \ + but not whether it is true.", + at(p.number) + )); + } + } + + let single = item.format == Format::SingleBestAnswer; + if single && p.key.len() > 1 { + issues.push(format!( + "{}: single_best_answer administers exactly one key, this names {}", + at(p.number), + p.key.len() + )); + } + + // No distractor list means the whole pool, which is what a pre-2.0 + // record means and what an item whose pool is its form still means. + // The record still has to say which key, when the pool offers a choice. + if p.distractors.is_empty() { + let (keys, _) = item.pool(); + if p.key.is_empty() && keys.len() > 1 && single { + issues.push(format!( + "{}: the item offers {} defensible keys, so the record has to say which one \ + this assessment used", + at(p.number), + keys.len() + )); + } + } else { + let shown = item.administered(&p.key, &p.distractors); + let expected = self.course.policy.options_per_item; + if shown.len() != expected { + issues.push(format!( + "{}: administers {} option(s), but course policy is {expected} per item", + at(p.number), + shown.len() + )); + } + if single && p.key.is_empty() { + issues.push(format!( + "{}: names its distractors but not its key, so what was marked correct is \ + left to whatever the bank says today", + at(p.number) + )); + } + } + + if let Some(recorded) = &p.variant { + if *recorded != item.variant_digest(&p.key, &p.distractors) { + issues.push(format!( + "{}: an administered option has been reworded since this assessment. \ + Statistics pooled under this variant describe the older wording.", + at(p.number) + )); + } + } + issues + } + /// Checks every sealed administration's stems against the bank. /// /// The seal is the authority on what was administered, so this is the @@ -334,11 +443,7 @@ impl Catalog { )); } } - if !p.key.is_empty() && p.key != entry.item.key_letters() { - issues.push(format!( - "question {} ({}): the recorded key {:?} differs from the item's current key {:?}", - p.number, p.item, p.key, entry.item.key_letters())); - } + issues.extend(self.validate_placement(p, &entry.item)); } } } diff --git a/src/model/item.rs b/src/model/item.rs index cf7ca4a..34d2859 100644 --- a/src/model/item.rs +++ b/src/model/item.rs @@ -257,6 +257,17 @@ pub struct Choice { /// Your a priori guess at how often this option is chosen. #[serde(default, skip_serializing_if = "Option::is_none")] pub selection_rate_expected: Option, + + /// Why this option is no longer drawn, when it is not. + /// + /// A retired option stays in the file forever. It has to: a seal and four + /// terms of response rows refer to it by id, and deleting it would turn + /// every one of those references into a dangling one. What retirement does + /// is take it out of the pool an assessment draws from, with the reason + /// attached — "selected by 1 of 96 across two administrations" is a finding + /// about the option, and the place for it is next to the option. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub retired: Option, } impl Choice { @@ -605,6 +616,89 @@ pub struct Calibration { /// Machine-detected problems. #[serde(default, skip_serializing_if = "Vec::is_empty")] pub flags: Vec, + + /// One record per option set ever administered. + /// + /// What the flat fields above cannot express once options are a pool. A + /// stem shown with distractors `{third, second, first}` is a measurably + /// easier item than the same stem with `{third, plus-line, line-two}`, so a + /// p-value pooled across both is the average of two different questions. + /// Statistics are computed and compared per variant; the flat fields remain + /// as the pre-2.0 summary, and for an item whose pool is its form the two + /// agree. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub variants: Vec, + + /// One record per option, pooled across every set it appeared in. + /// + /// The capability the pool is worth the trouble for. Selection rates are + /// shares of a fixed set, so they are only comparable *within* a variant — + /// which means this view supports exactly one kind of claim, and it is the + /// useful one: this option draws nobody, anywhere. That is the evidence + /// that retires a distractor, and one administration cannot supply it. + #[serde(default, skip_serializing_if = "std::collections::BTreeMap::is_empty")] + pub options: std::collections::BTreeMap, +} + +/// Statistics for one option set, as administered. +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct VariantCalibration { + /// The digest this record describes. See [`Item::variant_digest`]. + pub variant: String, + /// The option ids keyed correct. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub key: Vec, + /// The option ids offered alongside them. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub distractors: Vec, + /// The administrations pooled into these numbers. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub administrations: Vec, + /// Examinees pooled. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub n_examinees: Option, + /// Proportion correct. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub p_value: Option, + /// Corrected item-total point-biserial correlation. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub point_biserial: Option, + /// Upper-minus-lower-group discrimination index. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub discrimination_index: Option, + /// Per-option behaviour within this set, keyed by option id. + #[serde(default, skip_serializing_if = "std::collections::BTreeMap::is_empty")] + pub option_stats: std::collections::BTreeMap, + /// Fitted item response theory parameters for this set. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub irt: Option, + /// Machine-detected problems with this set. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub flags: Vec, +} + +/// What one option has done across every set it has appeared in. +/// +/// Deliberately coarse. Averaging selection rates across variants is not +/// meaningful — each is a share of a different set — so `mean_selection_rate` +/// is a summary for reading, not a statistic to act on. `never_chosen` is the +/// one field that carries weight, and it needs several administrations to earn. +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct OptionHistory { + /// How many distinct variants this option has appeared in. + #[serde(default)] + pub appearances: usize, + /// Examinees who saw it, summed across those variants. + #[serde(default)] + pub n_examinees: usize, + /// Mean of its within-variant selection rates. For reading only. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub mean_selection_rate: Option, + /// Whether it has never been chosen, anywhere. + #[serde(default, skip_serializing_if = "is_false")] + pub never_chosen: bool, } /// How one option behaved. @@ -932,6 +1026,98 @@ impl Item { fingerprint(parts.iter().map(|s| s.as_str())) } + /// The options an assessment administers, in the order the bank declares + /// them. + /// + /// Since 2.0 `options` is a *pool*: it may hold several defensible keys and + /// more distractors than any one form shows, and which of them a student + /// saw is a property of the placement rather than of the item. Everything + /// that renders, seals, decodes, or scores an administration has to work + /// from this rather than from `options`, or the paper and the key disagree. + /// + /// Bank order, not administered order: the per-form permutation is + /// [`crate::select::option_order`]'s business, and keeping the two separate + /// is what lets one item appear on three forms with one set of statistics. + /// + /// # Arguments + /// + /// * `key` - the option ids keyed correct for this administration. + /// * `distractors` - the option ids offered alongside them. + /// + /// # Returns + /// + /// The named options, or the whole live pool when `distractors` is empty. + /// + /// `distractors` is what says the set was chosen, not `key`. A pre-2.0 + /// record names its key and nothing else — `key: [D]` with no distractor + /// list — and it means "all of them, and D is the right one". Reading that + /// as "administer D alone" would print a one-option paper for every + /// assessment ever recorded. + pub fn administered(&self, key: &[String], distractors: &[String]) -> Vec<&Choice> { + if distractors.is_empty() { + return self + .options + .iter() + .filter(|o| o.retired.is_none()) + .collect(); + } + self.options + .iter() + .filter(|o| key.contains(&o.id) || distractors.contains(&o.id)) + .collect() + } + + /// The options that may still be drawn. + /// + /// # Returns + /// + /// Every option not retired, split into candidate keys and distractors. + pub fn pool(&self) -> (Vec<&Choice>, Vec<&Choice>) { + let live = || self.options.iter().filter(|o| o.retired.is_none()); + ( + live().filter(|o| o.correct).collect(), + live().filter(|o| !o.correct).collect(), + ) + } + + /// A digest of the item as one administration showed it. + /// + /// The pooling key for statistics, and the reason + /// [`Item::fingerprint`] cannot be. A stem with distractors + /// `{third, second, first}` is a measurably easier item than the same stem + /// with `{third, plus-line, line-two}`, so pooling a p-value across both is + /// averaging two different questions. Covers the stem, the administered + /// options, and which of them was keyed — the last because the same option + /// set with a different key is again a different item. + /// + /// # Arguments + /// + /// * `key` - the option ids keyed correct for this administration. + /// * `distractors` - the option ids offered alongside them. + /// + /// # Returns + /// + /// The digest as hex. + pub fn variant_digest(&self, key: &[String], distractors: &[String]) -> String { + let mut parts = vec![self.stem_digest()]; + let mut shown: Vec<&Choice> = self.administered(key, distractors); + shown.sort_by(|a, b| a.id.cmp(&b.id)); + for option in shown { + let keyed = if key.is_empty() { + option.correct + } else { + key.contains(&option.id) + }; + parts.push(format!( + "{}|{}|{}", + option.id, + if keyed { "1" } else { "0" }, + option.text.trim() + )); + } + fingerprint(parts.iter().map(|s| s.as_str())) + } + /// A digest of what the item asks, without its options. /// /// The identity check. [`Item::fingerprint`] covers the options too, which @@ -955,6 +1141,44 @@ impl Item { fingerprint(parts.iter().map(|s| s.as_str())) } + /// The calibration recorded for one option set. + /// + /// # Arguments + /// + /// * `variant` - the digest from [`Item::variant_digest`]. + /// + /// # Returns + /// + /// The record, or `None` when this set has not been calibrated. + pub fn calibration_for(&self, variant: &str) -> Option<&VariantCalibration> { + self.calibration + .as_ref()? + .variants + .iter() + .find(|v| v.variant == variant) + } + + /// Whether a variant's recorded statistics still describe it. + /// + /// Staleness gets *narrower* with a pool rather than wider: rewording one + /// distractor used to invalidate the item's whole calibration, and now it + /// invalidates only the sets that distractor appeared in. + /// + /// # Arguments + /// + /// * `variant` - the digest to check. + /// + /// # Returns + /// + /// `false` only when a record exists for that digest and the digest no + /// longer matches what the option ids now say. + pub fn variant_is_current(&self, variant: &str) -> bool { + match self.calibration_for(variant) { + Some(record) => variant == self.variant_digest(&record.key, &record.distractors), + None => true, + } + } + /// Whether the recorded calibration matches the current content. /// /// # Returns @@ -1131,6 +1355,101 @@ options: assert_eq!(a.fingerprint(), b.fingerprint()); } + #[test] + fn a_pool_administers_a_subset_and_defaults_to_everything() { + let mut it = item(MINIMAL); + let all: Vec = it.options.iter().map(|o| o.id.clone()).collect(); + + // Unstated means the whole pool, which is what a pre-2.0 record meant. + assert_eq!(it.administered(&[], &[]).len(), all.len()); + + // A retired option leaves the pool but not the file. + it.options[1].retired = Some(Retirement { + on: Date::new(2026, 9, 20).unwrap(), + reason: "chosen by 1 of 96 across two administrations".into(), + replaced_by: None, + }); + let shown = it.administered(&[], &[]); + assert_eq!(shown.len(), all.len() - 1); + assert!(!shown.iter().any(|o| o.id == all[1])); + // Still resolvable: a seal and four terms of rows refer to it. + assert!(it.option(&all[1]).is_some()); + + // A key with no distractor list is a pre-2.0 record, and it means all + // of them. Reading it as "administer the key alone" would print a + // one-option paper for every assessment already recorded. + assert_eq!(it.administered(&[all[2].clone()], &[]).len(), all.len() - 1); + + // Named explicitly, bank order is kept whatever order the lists are in. + let shown = it.administered(&[all[2].clone()], &[all[0].clone()]); + assert_eq!( + shown.iter().map(|o| o.id.clone()).collect::>(), + vec![all[0].clone(), all[2].clone()] + ); + } + + #[test] + fn the_variant_digest_tracks_the_option_set_and_the_stem_does_not() { + let it = item(MINIMAL); + let ids: Vec = it.options.iter().map(|o| o.id.clone()).collect(); + + let one = it.variant_digest(&[ids[0].clone()], &[ids[1].clone()]); + let two = it.variant_digest(&[ids[0].clone()], &[ids[2].clone()]); + // A different distractor is a different item: same stem, different + // difficulty, so pooling a p-value across both would average two + // questions. + assert_ne!(one, two, "a swapped distractor is a new variant"); + // The stem is unmoved by any of it. + assert_eq!(it.stem_digest(), item(MINIMAL).stem_digest()); + + // Order of the lists is not part of the identity. + assert_eq!( + it.variant_digest(&[ids[0].clone()], &[ids[2].clone(), ids[1].clone()]), + it.variant_digest(&[ids[0].clone()], &[ids[1].clone(), ids[2].clone()]) + ); + } + + #[test] + fn a_variant_goes_stale_alone_rather_than_taking_the_item_with_it() { + let mut it = item(MINIMAL); + let ids: Vec = it.options.iter().map(|o| o.id.clone()).collect(); + let one = it.variant_digest(&[ids[0].clone()], &[ids[1].clone()]); + let two = it.variant_digest(&[ids[0].clone()], &[ids[2].clone()]); + + it.calibration = Some(Calibration { + variants: vec![ + VariantCalibration { + variant: one.clone(), + key: vec![ids[0].clone()], + distractors: vec![ids[1].clone()], + n_examinees: Some(96), + p_value: Some(0.84), + ..VariantCalibration::default() + }, + VariantCalibration { + variant: two.clone(), + key: vec![ids[0].clone()], + distractors: vec![ids[2].clone()], + n_examinees: Some(32), + ..VariantCalibration::default() + }, + ], + ..Calibration::default() + }); + + assert_eq!(it.calibration_for(&one).unwrap().n_examinees, Some(96)); + assert!(it.calibration_for("nothing-like-this").is_none()); + assert!(it.variant_is_current(&one)); + assert!(it.variant_is_current(&two)); + + // Rewording the option that only the second set used leaves the first + // set's numbers standing. Before the pool, one distractor edit + // invalidated every statistic the item had. + it.options[2].text = "a different distractor".into(); + assert!(it.variant_is_current(&one), "the first set never showed it"); + assert!(!it.variant_is_current(&two)); + } + #[test] fn stale_calibration_is_detectable() { let mut it = item(MINIMAL); diff --git a/src/model/seal.rs b/src/model/seal.rs index 0b5f5cc..d530368 100644 --- a/src/model/seal.rs +++ b/src/model/seal.rs @@ -539,7 +539,9 @@ fn sealed_form(catalog: &Catalog, record: &AssessmentFile, form: &Form) -> Resul for (index, placement) in printed.iter().enumerate() { let entry = catalog.require(&placement.item)?; let item = &entry.item; - let n = item.options.len(); + // The pool is not the paper: seal what this placement administered. + let shown = item.administered(&placement.key, &placement.distractors); + let n = shown.len(); let order = select::option_order(form, &placement.item, n); let canonical_key: BTreeSet = if placement.key.is_empty() { @@ -551,8 +553,7 @@ fn sealed_form(catalog: &Catalog, record: &AssessmentFile, form: &Form) -> Resul let mut options = Vec::with_capacity(n); let mut printed_key = Vec::new(); for (position, source_index) in order.iter().enumerate() { - let canonical = item - .options + let canonical = shown .get(*source_index) .map(|c| c.id.clone()) .unwrap_or_else(|| printed_letter(*source_index));