feat: cooked up something fierce
Pipeline / check (pull_request) Successful in 3m9s
Pipeline / docs (pull_request) Skipped
Pipeline / nightly (pull_request) Skipped
Pipeline / release (pull_request) Skipped

This commit is contained in:
2026-09-26 18:22:25 -04:00
parent eabc98ad31
commit 5ac1e317c0
26 changed files with 1485 additions and 44 deletions
+89 -2
View File
@@ -749,7 +749,8 @@ fn option_schema() -> Value {
},
"selection_rate_expected": proportion(
"How often you expect this to be chosen. Compared against reality."
)
),
"retired": retirement_schema()
}
})
}
@@ -944,6 +945,21 @@ fn calibration_schema() -> Value {
"additionalProperties": option_stat_schema()
},
"irt": irt_schema(),
"variants": {
"type": "array",
"description": "One record per option set ever administered. A stem shown with \
different distractors is a different item, so a p-value pooled \
across both would average two questions.",
"items": variant_calibration_schema()
},
"options": {
"type": "object",
"description": "One record per option, pooled across every set it appeared in. \
Supports one claim — this option draws nobody, anywhere — which \
is what retires a distractor and what one administration cannot \
show.",
"additionalProperties": option_history_schema()
},
"flags": {
"type": "array",
"items": { "type": "string", "enum": strings(&flags) }
@@ -985,6 +1001,63 @@ fn history_schema() -> Value {
})
}
/// The schema for one option set's statistics.
fn variant_calibration_schema() -> Value {
json!({
"type": "object",
"required": ["variant"],
"additionalProperties": false,
"properties": {
"variant": {
"type": "string",
"description": "Digest of the stem, the administered options, and which was keyed."
},
"key": string_array("The option ids keyed correct in this set."),
"distractors": string_array("The option ids offered alongside them."),
"administrations": string_array("The administrations pooled into these numbers."),
"n_examinees": { "type": "integer", "minimum": 0 },
"p_value": proportion("Proportion correct, for this option set only."),
"point_biserial": { "type": "number", "minimum": -1.0, "maximum": 1.0 },
"discrimination_index": { "type": "number", "minimum": -1.0, "maximum": 1.0 },
"option_stats": {
"type": "object",
"description": "Per-option behaviour within this set, by option id.",
"additionalProperties": option_stat_schema()
},
"irt": irt_schema(),
"flags": {
"type": "array",
"items": {
"type": "string",
"enum": strings(
&Flag::ALL.iter().map(|f| f.as_str()).collect::<Vec<&str>>()
)
}
}
}
})
}
/// The schema for one option's cross-variant history.
fn option_history_schema() -> Value {
json!({
"type": "object",
"additionalProperties": false,
"properties": {
"appearances": { "type": "integer", "minimum": 0 },
"n_examinees": { "type": "integer", "minimum": 0 },
"mean_selection_rate": proportion(
"Mean of the within-variant rates. For reading, not for acting on: each rate is \
a share of a different set."
),
"never_chosen": {
"type": "boolean",
"description": "Never chosen, anywhere. The claim that justifies retiring it."
}
}
})
}
/// The schema for a retirement record.
fn retirement_schema() -> Value {
json!({
@@ -1289,7 +1362,21 @@ fn placement_schema() -> Value {
},
"points": { "type": "number", "minimum": 0.0 },
"bonus": { "type": "boolean" },
"key": string_array("Keyed option letters as administered."),
"key": string_array(
"The option ids keyed correct for this administration. One for a \
single_best_answer, chosen from the item's pool of defensible keys."
),
"distractors": string_array(
"The option ids offered alongside the key. Resolved when the assessment is \
assembled and written out explicitly, so a later bank edit cannot change the \
paper. Empty means the whole pool."
),
"variant": {
"type": "string",
"description": "Digest of the item as this administration showed it: stem, \
administered options, and which was keyed. The key statistics \
pool on."
},
"level": level(),
"learning_targets": string_array("Targets as administered."),
"credit_overrides": {
+161 -1
View File
@@ -88,6 +88,16 @@ pub enum Rule {
/// An expectation of low discrimination on a higher-level item.
ContradictoryDesign,
// --- the option pool ---
/// Fewer usable distractors than a form shows.
ThinOptionPool,
/// An option that has never been administered and has not been retired.
UnusedOption,
/// An option retired without saying what it did.
UnjustifiedRetirement,
/// A distractor that has never been chosen, in any set it appeared in.
NonfunctioningDistractor,
// --- evidence ---
/// Statistics describe an older version of the item.
StaleCalibration,
@@ -104,7 +114,7 @@ impl Rule {
///
/// Used by `--list-rules`, and by the test that keeps this list in step with
/// the enum.
pub const ALL: [Rule; 26] = [
pub const ALL: [Rule; 30] = [
Rule::KeyIsLongest,
Rule::UnevenOptionLength,
Rule::WordRepeatCue,
@@ -127,6 +137,10 @@ impl Rule {
Rule::WeakFormatForLevel,
Rule::ScoredBonusLevel,
Rule::ContradictoryDesign,
Rule::ThinOptionPool,
Rule::UnusedOption,
Rule::UnjustifiedRetirement,
Rule::NonfunctioningDistractor,
Rule::StaleCalibration,
Rule::DifficultyMissed,
Rule::DiscriminationMissed,
@@ -152,6 +166,9 @@ impl Rule {
// Statistics attached to text that has since changed are actively
// misleading, which is worse than absent.
R::StaleCalibration => Severity::High,
// An item that cannot fill a form is an item `assemble` will put on
// a paper short an option.
R::ThinOptionPool => Severity::High,
// An unanswerable question for a screen-reader user.
R::AssetWithoutAltText => Severity::High,
@@ -178,6 +195,13 @@ impl Rule {
| R::NoStudentFeedback
| R::DifficultyMissed
| R::DiscriminationMissed => Severity::Low,
// A distractor that draws nobody across several administrations is
// evidence to act on, not a style note.
R::NonfunctioningDistractor => Severity::Medium,
// Both are tidiness: the item still works, but its pool is
// carrying something nobody has accounted for.
R::UnusedOption | R::UnjustifiedRetirement => Severity::Low,
}
}
@@ -206,6 +230,10 @@ impl Rule {
Rule::WeakFormatForLevel => "complete-format-level",
Rule::ScoredBonusLevel => "complete-bonus-policy",
Rule::ContradictoryDesign => "complete-design-conflict",
Rule::ThinOptionPool => "pool-thin",
Rule::UnusedOption => "pool-unused",
Rule::UnjustifiedRetirement => "pool-unjustified-retirement",
Rule::NonfunctioningDistractor => "evidence-nonfunctioning",
Rule::StaleCalibration => "evidence-stale",
Rule::DifficultyMissed => "evidence-difficulty",
Rule::DiscriminationMissed => "evidence-discrimination",
@@ -238,9 +266,11 @@ impl Rule {
| Rule::WeakFormatForLevel
| Rule::ScoredBonusLevel
| Rule::ContradictoryDesign => "completeness",
Rule::ThinOptionPool | Rule::UnusedOption | Rule::UnjustifiedRetirement => "pool",
Rule::StaleCalibration
| Rule::DifficultyMissed
| Rule::DiscriminationMissed
| Rule::NonfunctioningDistractor
| Rule::DuplicateStem => "evidence",
}
}
@@ -270,6 +300,10 @@ impl Rule {
Rule::WeakFormatForLevel,
Rule::ScoredBonusLevel,
Rule::ContradictoryDesign,
Rule::ThinOptionPool,
Rule::UnusedOption,
Rule::UnjustifiedRetirement,
Rule::NonfunctioningDistractor,
Rule::StaleCalibration,
Rule::DifficultyMissed,
Rule::DiscriminationMissed,
@@ -302,6 +336,12 @@ impl Rule {
Rule::WeakFormatForLevel => "true/false at an analytic level",
Rule::ScoredBonusLevel => "a level the policy reserves for bonus is scored",
Rule::ContradictoryDesign => "low expected discrimination on a higher-level item",
Rule::ThinOptionPool => "fewer usable distractors than a form shows",
Rule::UnusedOption => "an option has never been administered and is not retired",
Rule::UnjustifiedRetirement => "an option was retired without saying what it did",
Rule::NonfunctioningDistractor => {
"a distractor has never been chosen in any set it appeared in"
}
Rule::StaleCalibration => "statistics describe an older version of the item",
Rule::DifficultyMissed => "observed difficulty was far from predicted",
Rule::DiscriminationMissed => "observed discrimination contradicted the prediction",
@@ -749,6 +789,87 @@ pub fn lint_item(entry: &Entry, course: &CourseFile, t: &Thresholds) -> Vec<Find
}
}
// --- the option pool
// These only make sense once options are a pool, and the pool is where an
// item's spare parts sit. A bank that never draws from it will not trip any
// of them.
if it.format.has_options() {
let (keys, distractors) = it.pool();
let wanted = course.policy.options_per_item.saturating_sub(1);
if distractors.len() < wanted {
push(
Rule::ThinOptionPool,
Severity::High,
format!(
"has {} usable distractor(s) but a form shows {}, so `assemble` will put \
this on a paper an option short",
distractors.len(),
course.policy.options_per_item
),
);
}
for option in &it.options {
match &option.retired {
Some(retirement) => {
if retirement.reason.trim().len() < 12 {
push(
Rule::UnjustifiedRetirement,
Severity::Low,
format!(
"option `{}` is retired with no real reason. The reason is the \
finding — what it drew, or failed to draw — and it is the only \
part of a retirement worth anything in two years",
option.id
),
);
}
}
None => {
// An option nobody has been shown is a draft, and a draft
// sitting in an approved item's pool will eventually be
// drawn onto a paper without ever having been reviewed
// against data.
if let Some(history) = it
.calibration
.as_ref()
.filter(|c| !c.options.is_empty())
.and_then(|c| c.options.get(&option.id))
{
if history.never_chosen && history.appearances > 1 {
push(
Rule::NonfunctioningDistractor,
Severity::Medium,
format!(
"option `{}` has appeared in {} option set(s) across {} \
examinees and has never been chosen. One administration \
would not show this; several do.",
option.id, history.appearances, history.n_examinees
),
);
}
} else if it
.calibration
.as_ref()
.is_some_and(|c| !c.options.is_empty())
{
push(
Rule::UnusedOption,
Severity::Low,
format!(
"option `{}` has never been administered. Either it is waiting \
its turn, or it was drafted and forgotten — retire it and say \
which.",
option.id
),
);
}
}
}
}
let _ = keys;
}
// --- evidence
if !it.calibration_is_current() {
push(
@@ -1215,6 +1336,45 @@ mod tests {
c
}
#[test]
fn a_pool_too_thin_to_fill_a_form_is_flagged() {
let c = codes(
r#"
id: q-a-001
status: draft
level: 2
stem: Which mechanism best explains the sigmoidal binding curve?
options:
- { id: o-shift, text: Ligand binding shifts the tetramer to a higher-affinity state, correct: true }
- { id: o-fixed, text: Each subunit binds with the same fixed affinity throughout }
"#,
);
// Policy shows four options and the pool can supply two, so `assemble`
// would put this on a paper two short.
assert!(c.contains(&"pool-thin"), "{c:?}");
}
#[test]
fn a_retirement_with_no_finding_is_flagged() {
let c = codes(
r#"
id: q-a-001
status: draft
level: 2
stem: Which mechanism best explains the sigmoidal binding curve?
options:
- { id: o-shift, text: Ligand binding shifts the tetramer to a higher-affinity state, correct: true }
- { id: o-fixed, text: Each subunit binds with the same fixed affinity throughout }
- { id: o-consumed, text: "Ligand is consumed as it binds, depleting the available pool" }
- { id: o-oxidation, text: The heme iron changes oxidation state upon binding }
- { id: o-cooperative, text: Subunits bind independently of one another, retired: { 'on': 2026-09-20, reason: bad } }
"#,
);
assert!(c.contains(&"pool-unjustified-retirement"), "{c:?}");
// Four live distractors is enough for a four-option form.
assert!(!c.contains(&"pool-thin"), "{c:?}");
}
#[test]
fn clean_item_passes() {
let c = codes(
+152 -2
View File
@@ -33,8 +33,9 @@ use crate::course::{CourseFile, SCHEMA_VERSION};
use crate::date::Date;
use crate::error::{Error, Result};
use crate::history::History;
use crate::item::{Choice, Item};
use crate::rng::Rng;
use crate::taxonomy::Level;
use crate::taxonomy::{Format, Level};
/// The result of a draw.
#[derive(Debug, Clone)]
@@ -466,15 +467,23 @@ pub fn to_record(
.chain(selection.bonus.iter().map(|u| (u, true))),
) {
let e = catalog.require(uid)?;
let (key, distractors) = draw_options(
&e.item,
catalog.course.policy.options_per_item,
blueprint.seed.unwrap_or(0),
uid,
);
items.push(Placement {
number,
item: uid.clone(),
version: None,
stem_digest: Some(e.item.stem_digest()),
variant: Some(e.item.variant_digest(&key, &distractors)),
fingerprint: Some(e.item.fingerprint()),
points: Some(e.item.points(default_points)),
bonus: is_bonus || e.item.bonus,
key: e.item.key_letters(),
distractors,
key,
level: Some(e.item.level),
learning_targets: e.item.learning_targets.clone(),
credit_overrides: BTreeMap::new(),
@@ -571,6 +580,85 @@ pub fn layout(record: &AssessmentFile, form: &Form) -> Vec<Placement> {
scored.into_iter().chain(bonus).collect()
}
/// Draws the key and the distractors one placement administers.
///
/// Resolved here, at assembly, and written into the record as explicit lists.
/// Nothing downstream samples: an export that drew its own options would print
/// a different paper every time the bank was touched.
///
/// The draw is seeded on the blueprint and the item, so re-running `assemble`
/// with the same seed produces the same paper, and two items in one assessment
/// draw independently.
///
/// # Arguments
///
/// * `item` - the item, whose options are a pool.
/// * `per_item` - how many options a form shows, from course policy.
/// * `seed` - the blueprint seed.
/// * `uid` - the item id, salting the draw.
///
/// # Returns
///
/// The keyed ids and the distractor ids, each sorted, naming options of `item`.
/// Both empty for an item with no options, which is an open response.
pub fn draw_options(
item: &Item,
per_item: usize,
seed: u64,
uid: &str,
) -> (Vec<String>, Vec<String>) {
let (keys, distractors) = item.pool();
if keys.is_empty() && distractors.is_empty() {
return (Vec::new(), Vec::new());
}
// Multiple response keys every correct option; anything else keys one, and
// when the pool offers several defensible keys the draw picks one so that
// the record says which.
let wanted_keys = match item.format {
Format::MultipleResponse => keys.len(),
_ => 1.min(keys.len()),
};
let mut rng = Rng::from_label(&format!("{seed}/{uid}/options"));
let mut key_ids = pick(&keys, wanted_keys, &mut rng);
key_ids.sort();
// A pool with fewer usable distractors than the policy asks for is a
// finding, not a failure: the form comes out short and `lint` says so,
// rather than `assemble` refusing to build the assessment at all.
let wanted = per_item.saturating_sub(key_ids.len());
let mut distractor_ids = pick(&distractors, wanted.min(distractors.len()), &mut rng);
distractor_ids.sort();
(key_ids, distractor_ids)
}
/// Takes `n` options, preferring the ones that were designed rather than merely
/// written.
///
/// A distractor carrying a misconception and an error type is one you thought
/// about; one carrying neither is filler. When the pool is larger than the form,
/// the thought-about ones go on the paper. The shuffle comes first so that
/// options of equal standing are drawn by seed rather than by declaration
/// order.
fn pick(options: &[&Choice], n: usize, rng: &mut Rng) -> Vec<String> {
if n >= options.len() {
return options.iter().map(|o| o.id.clone()).collect();
}
let mut order: Vec<usize> = (0..options.len()).collect();
rng.shuffle(&mut order);
order.sort_by_key(|&i| {
let o = options[i];
u8::from(o.misconception.is_none()) + u8::from(o.error_type.is_none())
});
order
.into_iter()
.take(n)
.map(|i| options[i].id.clone())
.collect()
}
/// The option order for one item on one form.
///
/// # Arguments
@@ -693,6 +781,68 @@ mod tests {
assert_eq!(form_label(27), "AB");
}
/// An item whose options are given as YAML, so the test needs no literal.
fn pool_item(options: &str) -> Item {
let src = format!(
r#"id: q-x
status: approved
level: 1
cognitive_process: recall
stem: Which line holds the quality scores?
learning_targets: [t-x]
sources: [{{ lecture: L1 }}]
options:
{options}"#
);
serde_yaml_ng::from_str(&src).expect("item parses")
}
const DESIGNED: &str = r#" - { id: o-key, text: right, correct: true }
- { id: o-designed-a, text: a, misconception: mistakes the separator, error_type: recall_confusion }
- { id: o-designed-b, text: b, misconception: confuses the two, error_type: recall_confusion }
- { id: o-filler-a, text: c }
- { id: o-filler-b, text: d }
"#;
#[test]
fn a_draw_prefers_designed_distractors_and_is_reproducible() {
let item = pool_item(DESIGNED);
let (key, distractors) = draw_options(&item, 3, 1103, "q-x");
assert_eq!(key, vec!["o-key".to_string()]);
assert_eq!(distractors.len(), 2);
// Thought-about distractors go on the paper before filler does.
assert!(
distractors.iter().all(|d| d.starts_with("o-designed")),
"{distractors:?}"
);
// Same seed, same paper.
assert_eq!(draw_options(&item, 3, 1103, "q-x"), (key, distractors));
// A retired option is not drawn, and the form comes out of the rest.
let retired = pool_item(&DESIGNED.replace(
"{ id: o-designed-a, text: a,",
"{ id: o-designed-a, text: a, retired: { 'on': 2026-09-20, reason: nonfunctioning },",
));
let (_, after) = draw_options(&retired, 3, 1103, "q-x");
assert!(!after.iter().any(|d| d == "o-designed-a"), "{after:?}");
}
#[test]
fn a_thin_pool_comes_out_short_rather_than_refusing_to_build() {
let item = pool_item(
" - { id: o-key, text: right, correct: true }\n - { id: o-one, text: wrong }\n",
);
let (key, distractors) = draw_options(&item, 4, 7, "q-y");
assert_eq!(key.len(), 1);
assert_eq!(
distractors.len(),
1,
"one usable distractor, so one is drawn"
);
}
#[test]
fn option_order_is_a_reproducible_permutation() {
let form = Form {