diff --git a/.gitignore b/.gitignore index d650d90..5a6fc24 100644 --- a/.gitignore +++ b/.gitignore @@ -1,4 +1,5 @@ preview +scratch /dist/ /THIRD-PARTY-LICENSES.txt diff --git a/Cargo.toml b/Cargo.toml index 2fd4f8b..c53ce87 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -17,10 +17,6 @@ path = "src/main.rs" name = "coursebank" path = "src/lib.rs" -[features] -default = ["parquet"] -parquet = ["dep:parquet", "dep:arrow-array", "dep:arrow-schema"] - [dependencies] clap = { version = "4", features = ["derive"] } csv = "1" @@ -29,9 +25,14 @@ serde_json = "1" serde_yaml_ng = "0.10" thiserror = "2" -arrow-array = { version = "55", optional = true } -arrow-schema = { version = "55", optional = true } -parquet = { version = "55", optional = true } +arrow-array = { version = "55"} +arrow-schema = { version = "55"} +parquet = { version = "55"} +aes-gcm = { version = "0.10"} +pbkdf2 = { version = "0.12", default-features = false, features = ["hmac"]} +sha2 = { version = "0.10"} +base64 = { version = "0.22"} +getrandom = { version = "0.2"} [profile.release] opt-level = 3 diff --git a/docs/guide/assignment.md b/docs/guide/assignment.md new file mode 100644 index 0000000..ee3509c --- /dev/null +++ b/docs/guide/assignment.md @@ -0,0 +1,192 @@ +# An assignment on the web + +This follows a single homework from an assembled record to a published Quarto page whose solutions stay locked until a student enters a password. +It assumes a course directory with approved items and an assembled assessment; if you do not have one, [`setup`](crate::guide::setup) and [`first_exam`](crate::guide::first_exam) build both. + +The site export is the third off-Canvas path, alongside the printed exam in [`typst_export`](crate::guide::typst_export) and the plain-text worksheet from `export practice`. +The difference is where the solutions go: printed on a key you keep, or encrypted into a bundle that ships with the page and unlocks in the browser. + +## Assemble the homework + +Assemble it the same way you assemble an exam, with the platform set to the website rather than paper or Canvas: + +```console +$ coursebank assemble a1.1 \ + --title "Homework 1" \ + --kind homework --platform other \ + --levels 2=1,3=1 --lectures L1.1 --seed 20260210 +``` + +What lands in `assessments/a1.1.yaml` is a record of the draw. +The id `a1.1` is the one thing to choose deliberately here: it becomes the bundle's file name and the name the page's gate points at, so keep it URL-safe and stable. + +A single unshuffled form is fine for homework. +To hand different students different option orders, add `--forms 2` and export each form separately; the printed choices and the letters the solutions refer to move together, because both come from the form's recorded seed. + +## Export for the web + +```console +$ coursebank export site a1.1 --out build/a1.1 --assets build/static +password for a1.1: k7m4-9p2q-r8tx-3wn6 +wrote build/a1.1/_questions.qmd +wrote build/a1.1/a1.1-solutions.json +wrote build/static/questions.css +wrote build/static/solutions.js + +Include in the page with: {{< include _questions.qmd >}} +``` + +That is four files and a password. + +`_questions.qmd` is the partial you include from the page. +`a1.1-solutions.json` is the encrypted bundle, named for the assessment so several assignments can share one site. +The two files under `build/static` style the questions and perform the unlock; they install once for the whole site, not once per page, so `--assets` is something you run the first time and drop afterward. + +The password is printed once and written nowhere. +Record it now. +It cannot be recovered from the files, which is the point: the bundle is useless without it, so losing it means re-exporting rather than reading it back. + +## What the partial holds + +The questions are Quarto fenced divs, with an empty, hidden slot where each solution will land: + +```text +::: {.solutions-gate data-bundle="a1.1-solutions.json"} +::: + +::: {.q #q-enthalpy-qp-001} +:::: {.q-head} +[Question 1]{.q-num} [Single best answer]{.q-kind} [1 point]{.q-points} +:::: + +:::: {.q-stem} +A reaction is run in an open flask, so the system stays at the constant +pressure of the room. The heat the reaction exchanges with its surroundings +equals the change in which quantity? +:::: + +:::: {.q-choices} +1. Enthalpy, $\Delta H$ +2. Internal energy, $\Delta U$ +3. Gibbs free energy, $\Delta G$ +4. Entropy, $\Delta S$ +:::: + +:::: {.qsol data-solution-for="q-enthalpy-qp-001" hidden="true"} +:::: +::: +``` + +The choices are printed without letters, and the stylesheet draws the A, B, C, D from their position. +There is nothing in this file to leak: no `correct` flag, no solution text, no rationale. +The `.qsol` slot is empty until the browser fills it, and the id in `data-solution-for` is the key the bundle looks up. + +Math is written in `$ … $` and passed through untouched, so MathJax typesets it in the page. + +## What the bundle holds + +Every solution is rendered to HTML, then encrypted: + +```json +{ + "v": 1, + "page": "a1.1", + "kdf": { "name": "PBKDF2", "hash": "SHA-256", "iterations": 250000, "salt": "jqbE6xIB..." }, + "cipher": "AES-GCM", + "items": { + "q-enthalpy-qp-001": { "iv": "8jzuxY...", "ct": "gmzNSlhioU...(+tag)" }, + "q-enthalpy-derive-001": { "iv": "1BddCd...", "ct": "sC8czL547/...(+tag)" } + } +} +``` + +The password derives an AES-256 key through PBKDF2-HMAC-SHA256 at 250,000 iterations over a random salt. +Each solution is encrypted under its own random IV, with the authentication tag appended to the ciphertext. +The plaintext HTML never leaves your machine. +Items are listed in the order the questions appear, not sorted, so the reader's browser can decrypt the first one to check the password before touching the rest. + +## Wire it into the site + +Install the two assets once in `_quarto.yml`: + +```yaml +format: + html: + css: + - static/questions.css + include-after-body: + - static/solutions.js +``` + +Put `_questions.qmd` and `a1.1-solutions.json` in the page's own directory, and include the partial from the page: + +```markdown +--- +title: "Homework 1" +--- + +{{< include _questions.qmd >}} +``` + +The gate resolves `data-bundle` relative to the page URL, so keeping the bundle beside the page is enough. +If your build prunes files it does not see linked, add `resources: ["*-solutions.json"]` to the page front matter so the bundle ships with the render. + +Render with `quarto render`. +A reader who opens the page sees the questions and a locked panel; typing the password decrypts the solutions in place, with the math typeset. + +## Hand out the password, and rotate it + +Give the password through a channel students already have, such as the course LMS, rather than the site itself. + +Be clear-eyed about what the lock does. +It keeps solutions off a public page until someone has the password. +It does not make them secret in a strong sense: the whole bundle is downloaded, so anyone with the password, or anyone they share it with, can decrypt every item, and the ciphertext is open to an offline guessing attack. +Eighty bits of password entropy and a quarter-million PBKDF2 iterations make guessing slow, but the right mental model is a lock on a take-home worksheet, not a grading system of record. + +Rotate the password after the due date by re-running the export, which mints a fresh one and a freshly encrypted bundle: + +```console +$ coursebank export site a1.1 --out build/a1.1 +password for a1.1: 2h9k-w4rq-8mnp-x6tv +... +``` + +To set a password yourself instead of generating one, pass `--password`. +Use that only when you have a reason to, such as re-encrypting an unchanged page with a password you already circulated. + +## Doing this from Rust + +The CLI is a thin wrapper over `coursebank::site`, which is compiled only with the `site` feature. +The example below needs `--features site` to build, so it is not run as a doctest: + +```rust,ignore +use std::path::Path; + +use coursebank::assessment::{AssessmentFile, Form}; +use coursebank::site::{self, Options}; +use coursebank::Catalog; + +fn main() -> coursebank::Result<()> { + let catalog = Catalog::load(Path::new("."))?; + let path = catalog.layout.assessments().join("a1.1.yaml"); + let record = AssessmentFile::load(&path)?; + + // An unshuffled form A; declare a form with a seed to shuffle. + let form = Form { id: "A".into(), seed: 0, shuffle_items: false, shuffle_options: false }; + + // `password: None` generates a fresh one; pass `Some(_)` to set your own. + let rendered = site::render(&catalog, &record, Options { form, password: None })?; + + std::fs::write("build/a1.1/_questions.qmd", &rendered.questions_qmd)?; + std::fs::write( + format!("build/a1.1/{}-solutions.json", rendered.page), + &rendered.solutions_json, + )?; + // The password is the one thing not on disk; print it now. + println!("password for {}: {}", rendered.page, rendered.password); + + Ok(()) +} +``` + +`site::assets()` returns the two browser files as name and contents, if you would rather write them from your own code than pass `--assets`. diff --git a/docs/guide/setup.md b/docs/guide/setup.md index 12782ec..1a99c21 100644 --- a/docs/guide/setup.md +++ b/docs/guide/setup.md @@ -26,8 +26,75 @@ That gives you: `build/` and `reports/` are in the generated `.gitignore`. The other four are the repository's content and belong in review. +If the directory already has a `.gitignore`, `init` keeps it and inserts only the patterns it was missing at the top, so running this inside an existing repository costs you nothing. + Add `--with-examples` if you want a filled-in bank to read rather than an empty directory to stare at. +## Declare your texts once + +Every work the course cites goes in `references`, keyed by the citation key you would use in a `.bib` file. + +```yaml +references: + kuriyan2013molecules: + label: KKW + kind: book + role: required + title: 'The molecules of life: Physical and chemical principles' + authors: ['Kuriyan, John', 'Konforti, Boyana', 'Wemmer, David'] + year: 2013 + publisher: W. W. Norton & Company + base_url: https://library.scient.ing/kuriyan2013molecules/ + note: On reserve at the Bevier Engineering Library. +``` + +`label` is the short form a reading list shows, and it has to name one work, because reports print it instead of the key. +`base_url` is what a reading's `path` is joined to, so the key appears once in the file rather than once per reading. + +## Point readings at objectives + +A reading names a location inside a reference and lists the objectives it serves. + +```yaml +lectures: + L1.1: + title: Enthalpy + readings: + - ref: kuriyan2013molecules + locator: '§1.3' + path: '1/A/#3' + objectives: [lo-water-attenuation, lo-coulomb-estimate] + summary: >- + Ionic interactions: favorable in vacuum, attenuated ~80-fold by water. + focus: >- + The two magnitudes and the factor of 80. + skip: >- + Skip the unit-conversion derivation. +``` + +The three prose fields answer three different questions, and each has a different reader. +`summary` says what the section contains, `focus` says what to take from it, and `skip` says what to ignore. +A student report quotes `focus` at somebody who missed the objective; a lecture page prints all three. + +The mapping lives on the reading rather than on the objective because objectives outlive editions. +When a textbook renumbers its sections, one block of `readings` changes and `learning_objectives` does not. +Going the other way is a scan: `coursebank lecture coverage` lists the readings behind each objective and flags the ones with none. + +Set `order` on each objective if you want a lecture page to number them in teaching order. +The registry is a map, so declaration order is lost on load, and sorting by id would put `lo-enthalpy` ahead of `lo-first-law`. + +A reading written as a plain string, which is what this field held before, still loads and is written back out unchanged. + +## Generate the reading list + +```console +$ coursebank lecture readings L1.1 --out lectures/l1_1-readings.qmd +wrote lectures/l1_1-readings.qmd +``` + +Objective numbers in the generated page (`_(LO 4, 7)_`) are positional, so they are computed at render time rather than written down. +Insert an objective and everything after it renumbers on the next build. + ## Point your editor at the schemas The schemas are the difference between authoring items and looking up field names. diff --git a/src/analysis.rs b/src/analysis.rs index 2d24d25..baa66ba 100644 --- a/src/analysis.rs +++ b/src/analysis.rs @@ -35,5 +35,6 @@ pub mod calibrate; pub mod classical; +pub mod diagnostic; pub mod irt; pub mod students; diff --git a/src/analysis/calibrate.rs b/src/analysis/calibrate.rs index 880a87b..7bd914a 100644 --- a/src/analysis/calibrate.rs +++ b/src/analysis/calibrate.rs @@ -28,18 +28,19 @@ use std::collections::{BTreeMap, BTreeSet}; use std::path::PathBuf; +use crate::SCHEMA_VERSION; use crate::assessment::AssessmentFile; -use crate::bank::BankFile; +use crate::calibration::{CalibrationFile, Measurement, MeasurementFile, MeasurementMeta}; use crate::catalog::Catalog; use crate::classical::{self, Analysis, ItemAnalysis, Thresholds}; use crate::date::Date; use crate::error::{Error, Result}; use crate::irt::{self, Fit}; -use crate::item::{Calibration, IrtParams, OptionStat}; +use crate::item::{Calibration, IrtParams, Item, OptionStat, VariantCalibration}; +use crate::layout::Layout; use crate::responses::ResponseSet; use crate::store::Store; use crate::taxonomy::Flag; -use crate::yaml; /// What calibration would change about one item. #[derive(Debug, Clone)] @@ -267,6 +268,35 @@ pub fn plan(catalog: &Catalog, store: &Store, opts: &Options) -> Result { )); } + // Splitting by option set means an item administered three times with three + // different sets has three cells of 24 rather than one of 72. That is the + // honest picture, and it is worth saying out loud rather than leaving + // someone to read an IRT fit that was never possible. + for (uid, appearances) in &by_item { + let mut sizes: Vec = Vec::new(); + for (_, analysis) in appearances { + if analysis.variant.is_some() { + sizes.push(analysis.n); + } + } + if sizes.len() > 1 { + let distinct: BTreeSet<&str> = appearances + .iter() + .filter_map(|(_, a)| a.variant.as_deref()) + .collect(); + if distinct.len() > 1 { + warnings.push(format!( + "{uid}: {} option sets across {} administrations, largest n = {}. \ + Statistics are kept per set, because a stem shown with different \ + distractors is a different item.", + distinct.len(), + appearances.len(), + sizes.iter().copied().max().unwrap_or(0) + )); + } + } + } + let mut changes = Vec::new(); for (uid, appearances) in &by_item { @@ -304,6 +334,12 @@ pub fn plan(catalog: &Catalog, store: &Store, opts: &Options) -> Result { option_stats: pooled.option_stats.clone(), irt: irt_params, flags: pooled.flags.clone(), + variants: variant_records(&entry.item, appearances, previous_variants(entry)), + options: BTreeMap::new(), + }; + let calibration = Calibration { + options: option_histories(&calibration.variants), + ..calibration }; let previous = entry.item.calibration.as_ref(); @@ -580,60 +616,123 @@ fn diff_calibration(previous: Option<&Calibration>, next: &Calibration) -> Vec Result> { - // Group by file so each is read and written once. - let mut by_file: BTreeMap<&PathBuf, Vec<&Change>> = BTreeMap::new(); +/// Returns [`Error::Io`] on a write failure and [`Error::Yaml`] if the existing +/// store does not parse. +pub fn apply(layout: &Layout, plan: &Plan) -> Result { + let path = layout.calibration_file(); + let mut store = CalibrationFile::load(&path)?; + for change in &plan.changes { - by_file.entry(&change.path).or_default().push(change); + store + .items + .insert(change.uid.clone(), change.calibration.clone()); } - let mut written = Vec::new(); - for (path, changes) in by_file { - let mut bank: BankFile = yaml::read(path)?; - for change in changes { - // The uid is `bank::item`; match on the item part. - let item_id = change - .uid - .split_once("::") - .map(|(_, id)| id) - .unwrap_or(&change.uid); - let target = bank.items.iter_mut().find(|i| i.id == item_id); - match target { - Some(item) => item.calibration = Some(change.calibration.clone()), - None => { - return Err(Error::Unresolved { - kind: "item", - id: change.uid.clone(), - context: Some(format!( - "{} — the bank changed since the plan was built; re-run calibration", - path.display() - )), - }); - } - } - } - yaml::write(path, &bank)?; - written.push(path.clone()); + if let Some(parent) = path.parent() { + std::fs::create_dir_all(parent).map_err(|e| Error::io(parent, e))?; } - Ok(written) + store.save(&path)?; + Ok(path) +} + +/// Records what one administration measured, as a file that is never rewritten. +/// +/// The audit trail under the pooled store: these are the numbers one exam +/// produced, on a day, under a named model. Keeping them means the history +/// survives losing `data/`, which is ignored by git precisely because every row +/// of it carries a student. +/// +/// # Arguments +/// +/// * `layout` - the course layout. +/// * `administration` - the administration id, which names the file. +/// * `analysis` - the classical analysis of that administration. +/// * `catalog` - the loaded course, for the digests each item was measured +/// against. +/// * `record` - the assessment record, for the term and date. +/// +/// # Returns +/// +/// The two files written: one row per question, one row per question and +/// option. +/// +/// # Errors +/// +/// Returns [`Error::Usage`] when a record for this administration already +/// exists, since an administration happened once. +pub fn record_measurements( + layout: &Layout, + administration: &str, + analysis: &Analysis, + catalog: &Catalog, + record: Option<&AssessmentFile>, +) -> Result> { + let mut items = Vec::new(); + for item in &analysis.items { + let Some(uid) = &item.item_ref else { continue }; + let entry = catalog.get(uid); + items.push(Measurement { + item: uid.clone(), + number: item.number, + variant: item.variant.clone(), + stem_digest: entry.map(|e| e.item.stem_digest()), + n: item.n, + p_value: Some(round4(item.p_value)), + point_biserial: item.point_biserial.map(round4), + discrimination_index: item.discrimination_index.map(round4), + key: item.key.clone(), + option_stats: item + .options + .iter() + .map(|(id, o)| (id.clone(), o.to_option_stat())) + .collect(), + irt: None, + flags: item.flags.clone(), + }); + } + + let file = MeasurementFile { + schema_version: SCHEMA_VERSION.to_string(), + administration: MeasurementMeta { + id: administration.to_string(), + assessment: record + .map(|r| r.assessment.id.clone()) + .unwrap_or_else(|| administration.to_string()), + term: record.and_then(|r| r.assessment.term.clone()), + date: record.and_then(|r| r.assessment.date), + forms: record + .map(|r| r.forms.iter().map(|f| f.id.clone()).collect()) + .unwrap_or_default(), + // The cohort, taken as the largest per-item n: a student who + // skipped question 7 still sat the exam. + n_examinees: analysis.items.iter().map(|i| i.n).max().unwrap_or(0), + model: None, + generated: Some(Date::today()), + coursebank: Some(crate::VERSION.to_string()), + }, + items, + }; + + file.write_csv(&layout.measurements()) } /// Builds a plan for a single administration, from an in-memory analysis. @@ -703,6 +802,16 @@ pub fn plan_from_analysis( .collect(), irt: irt_params, flags: item.flags.clone(), + variants: variant_records( + &entry.item, + &[(admin.clone(), item.clone())], + previous_variants(entry), + ), + options: BTreeMap::new(), + }; + let calibration = Calibration { + options: option_histories(&calibration.variants), + ..calibration }; let previous = entry.item.calibration.as_ref(); @@ -730,6 +839,171 @@ pub fn plan_from_analysis( } } +/// The variant records an item already has, to be merged with the new ones. +fn previous_variants(entry: &crate::catalog::Entry) -> Vec { + entry + .item + .calibration + .as_ref() + .map(|c| c.variants.clone()) + .unwrap_or_default() +} + +/// Builds one calibration record per option set the item was administered in. +/// +/// The records this pass computes replace the stored ones for the same variant +/// and leave the rest alone. That is what makes a partial recalibration safe: +/// a pass given only this term's data must not silently discard the numbers for +/// an option set that was retired two terms ago. +/// +/// Statistics come from the same [`pool`] used for the flat summary, so the two +/// agree for an item whose pool is its form — the case every pre-2.0 item is in. +/// +/// # Arguments +/// +/// * `item` - the bank item. +/// * `appearances` - the administration id and analysis of each appearance. +/// * `previous` - the records already stored. +/// +/// # Returns +/// +/// The merged records, in variant order. +fn variant_records( + item: &Item, + appearances: &[(String, ItemAnalysis)], + previous: Vec, +) -> Vec { + // Appearances whose administration mixed two option sets under one question + // number carry no variant, and there is no set for them to describe. + let mut by_variant: BTreeMap> = BTreeMap::new(); + for (admin, analysis) in appearances { + if let Some(variant) = &analysis.variant { + by_variant + .entry(variant.clone()) + .or_default() + .push((admin.clone(), analysis.clone())); + } + } + + let mut merged: BTreeMap = previous + .into_iter() + .map(|v| (v.variant.clone(), v)) + .collect(); + + for (variant, group) in by_variant { + let pooled = pool(&group); + // The option set is recovered from the digest's own record when the + // stored one has it, and from the options that were actually chosen + // otherwise, so a record written from data alone still says what it + // describes. + let (key, distractors) = describe(item, &variant, &group, merged.get(&variant)); + merged.insert( + variant.clone(), + VariantCalibration { + variant, + key, + distractors, + administrations: group.iter().map(|(a, _)| a.clone()).collect(), + n_examinees: Some(pooled.n), + p_value: Some(round4(pooled.p_value)), + point_biserial: pooled.point_biserial.map(round4), + discrimination_index: pooled.discrimination_index.map(round4), + option_stats: pooled.option_stats.clone(), + irt: None, + flags: pooled.flags.clone(), + }, + ); + } + + merged.into_values().collect() +} + +/// Which options a variant administered. +/// +/// Prefers what a stored record already says. Failing that, the options that +/// appear in the statistics are the ones students saw, and the item says which +/// of those are keyed. +fn describe( + item: &Item, + variant: &str, + group: &[(String, ItemAnalysis)], + stored: Option<&VariantCalibration>, +) -> (Vec, Vec) { + if let Some(stored) = stored { + if !stored.key.is_empty() + && item.variant_digest(&stored.key, &stored.distractors) == variant + { + return (stored.key.clone(), stored.distractors.clone()); + } + } + + let mut seen: BTreeSet = BTreeSet::new(); + for (_, analysis) in group { + seen.extend(analysis.options.keys().cloned()); + } + let keyed: BTreeSet = group + .iter() + .flat_map(|(_, a)| a.key.iter().cloned()) + .collect(); + + let key: Vec = seen + .iter() + .filter(|o| keyed.contains(*o)) + .cloned() + .collect(); + let distractors: Vec = seen + .iter() + .filter(|o| !keyed.contains(*o)) + .cloned() + .collect(); + (key, distractors) +} + +/// Summarizes what each option has done across every set it appeared in. +/// +/// Deliberately coarse, because selection rates are shares of a fixed set and +/// averaging them across different sets is not a statistic. The one claim this +/// view supports is the one worth having: an option that draws nobody in any +/// set it has appeared in is not doing anything, and that is the evidence for +/// retiring it — evidence a single administration cannot provide. +/// +/// # Arguments +/// +/// * `variants` - the per-variant records. +/// +/// # Returns +/// +/// One history per option id. +fn option_histories( + variants: &[VariantCalibration], +) -> BTreeMap { + let mut out: BTreeMap = BTreeMap::new(); + let mut rates: BTreeMap> = BTreeMap::new(); + + for variant in variants { + let n = variant.n_examinees.unwrap_or(0); + for (option, stat) in &variant.option_stats { + let entry = out.entry(option.clone()).or_default(); + entry.appearances += 1; + entry.n_examinees += n; + if let Some(rate) = stat.selection_rate { + rates.entry(option.clone()).or_default().push(rate); + } + } + } + + for (option, entry) in &mut out { + if let Some(seen) = rates.get(option) { + if !seen.is_empty() { + entry.mean_selection_rate = + Some(round4(seen.iter().sum::() / seen.len() as f64)); + entry.never_chosen = seen.iter().all(|r| *r <= f64::EPSILON); + } + } + } + out +} + /// Rounds to four decimals. fn round4(x: f64) -> f64 { (x * 1e4).round() / 1e4 @@ -754,9 +1028,77 @@ mod tests { option_stats: BTreeMap::new(), irt: None, flags: Vec::new(), + variants: Vec::new(), + options: BTreeMap::new(), } } + fn stat(rate: f64) -> OptionStat { + OptionStat { + selection_rate: Some(rate), + point_biserial: None, + upper_group_rate: None, + lower_group_rate: None, + } + } + + fn variant(id: &str, n: usize, dead_rate: f64) -> VariantCalibration { + VariantCalibration { + variant: id.into(), + n_examinees: Some(n), + option_stats: [ + ("o-key".to_string(), stat(0.8)), + ("o-dead".to_string(), stat(dead_rate)), + ] + .into_iter() + .collect(), + ..VariantCalibration::default() + } + } + + #[test] + fn an_option_that_draws_nobody_is_only_visible_across_sets() { + let histories = option_histories(&[variant("v1", 50, 0.0), variant("v2", 46, 0.0)]); + + let dead = &histories["o-dead"]; + assert_eq!(dead.appearances, 2); + assert_eq!(dead.n_examinees, 96); + // The claim the cross-variant view exists to support, and the one a + // single administration cannot make. + assert!(dead.never_chosen); + + assert!(!histories["o-key"].never_chosen); + assert_eq!(histories["o-key"].mean_selection_rate, Some(0.8)); + + // One set where it drew is enough to stop the claim. + let mixed = option_histories(&[variant("v1", 50, 0.0), variant("v2", 46, 0.04)]); + assert!(!mixed["o-dead"].never_chosen); + } + + #[test] + fn a_pass_with_no_data_for_an_item_keeps_the_records_it_has() { + let item: Item = serde_yaml_ng::from_str( + r#"id: q-x +status: approved +level: 1 +stem: s +options: + - { id: o-key, text: right, correct: true } + - { id: o-one, text: wrong } +"#, + ) + .expect("item parses"); + + let kept = variant_records(&item, &[], vec![variant("v-old", 96, 0.0)]); + assert_eq!( + kept.len(), + 1, + "a pass given nothing must not delete history" + ); + assert_eq!(kept[0].variant, "v-old"); + assert_eq!(kept[0].n_examinees, Some(96)); + } + #[test] fn a_first_calibration_is_all_new() { let next = calibration(0.7, Some(0.3), 24, "abc"); diff --git a/src/analysis/classical.rs b/src/analysis/classical.rs index a46f135..233307b 100644 --- a/src/analysis/classical.rs +++ b/src/analysis/classical.rs @@ -61,6 +61,13 @@ pub struct Thresholds { pub nonfunctioning: f64, /// How far observed difficulty may drift from the authored expectation. pub design_tolerance: f64, + /// How many examinees an item's calibration needs before its recorded + /// expectations are treated as evidence rather than as the author's guess. + /// + /// Fifty is the point at which the standard error of a proportion near 0.5 + /// drops to about 0.07, which is small enough that a quarter-point miss is + /// about the item rather than about the sample. + pub calibrated_n: usize, /// Fraction of the class in the upper and lower comparison groups. Kelley's /// 0.27 maximizes the difference between the groups for a normal /// distribution, and it remains the convention. @@ -78,6 +85,7 @@ impl Default for Thresholds { negative_discrimination: -0.05, nonfunctioning: 0.05, design_tolerance: 0.25, + calibrated_n: 50, group_fraction: 0.27, small_sample: 100, } @@ -126,6 +134,14 @@ pub struct ItemAnalysis { pub number: u32, /// The item's global id, when known. pub item_ref: Option, + /// The option set administered, when the rows agree on one. + /// + /// Within one administration an item has one variant, because a placement's + /// distractors are drawn once and shared by every form — only the printed + /// order differs. `None` means the rows disagreed, which happens when a + /// course opts into drawing distractors per form; the statistics below then + /// describe a mixture and cannot be pooled by option set. + pub variant: Option, /// How many students the item was administered to. pub n: usize, /// How many gave a non-blank response. @@ -154,6 +170,38 @@ pub struct ItemAnalysis { pub flags: Vec, /// Human-readable explanations tied to the flags. pub notes: Vec, + /// How the item behaved against what its author predicted, when the item + /// records a prediction. + pub prediction: Option, +} + +/// An authored expectation, checked against what happened. +/// +/// Kept apart from [`ItemAnalysis::flags`] on purpose. Before an item has been +/// administered, `design.expected_difficulty` is the author's guess, and a guess +/// that turns out wrong says something about the guess rather than about the +/// item. Flagging it anyway is how a report ends up with thirty +/// `design_mismatch` findings and no way to see the four that matter. So the +/// discrepancy is always recorded here, and it only becomes a +/// [`Flag::DesignMismatch`] once the expectation has data behind it. +#[derive(Debug, Clone)] +pub struct Prediction { + /// The difficulty the author expected. + pub expected_p: Option, + /// The discrimination band the author expected, as `(low, high)`. + pub expected_band: Option<(f64, f64)>, + /// Whether the expectation rests on a calibration with enough examinees + /// behind it, rather than on the author's judgement alone. + pub calibrated: bool, + /// Signed difficulty error, observed minus expected. Positive means the item + /// was easier than predicted. + pub p_error: Option, + /// Whether observed difficulty landed inside the tolerance. + pub p_within: Option, + /// Whether observed discrimination landed inside the expected band. + pub band_hit: Option, + /// What to say about it, phrased for whichever case applies. + pub notes: Vec, } impl ItemAnalysis { @@ -356,6 +404,39 @@ pub fn analyze( )); } + // A drop changes every student's percentage, so the two ways to get it wrong + // are worth saying out loud. Both are silent otherwise: the numbers simply + // come out different from the platform's. + if let Some(record) = record { + for placement in record.items.iter().filter(|p| p.dropped) { + let rows = set.for_item(placement.number); + if rows.is_empty() { + continue; + } + let all_credited = rows.iter().all(|r| r.credit >= 0.999); + if placement.dropped_with_credit() && !all_credited { + let short = rows.iter().filter(|r| r.credit < 0.999).count(); + warnings.push(format!( + "question {} is marked `dropped_as: full_credit`, but {short} of {} responses \ + carry less than full credit. Either the platform was not regraded or the \ + export predates the regrade; until one of those is fixed this report's \ + percentages will sit below the grade of record", + placement.number, + rows.len() + )); + } + if !placement.dropped_with_credit() && all_credited { + warnings.push(format!( + "question {} is dropped and every response carries full credit, which is what \ + crediting every option on the platform looks like. It is being removed from \ + the denominator here, so this report will read slightly lower than the \ + platform. Set `dropped_as: full_credit` if the platform kept the point", + placement.number + )); + } + } + } + let mut items = Vec::new(); let mut p_values = Vec::new(); let mut rpbs = Vec::new(); @@ -406,13 +487,13 @@ pub fn analyze( let mut credits: BTreeMap> = BTreeMap::new(); let mut blank = 0usize; for r in &rows { - if r.selected.is_empty() { + if r.chosen().is_empty() { blank += 1; continue; } // A multiple-response item is credited to the joined set, so that // "chose A and C" is one response pattern rather than two options. - let label = r.selected.join("+"); + let label = r.chosen().join("+"); if let Some(&si) = student_index.get(r.student_key.as_str()) { chose.entry(label.clone()).or_default().push(si); } @@ -480,6 +561,7 @@ pub fn analyze( item_ref: record .and_then(|r| r.placement(*number)) .map(|p| p.item.clone()), + variant: one_variant(set, *number), n, n_answered, blank_rate: blank as f64 / responded as f64, @@ -493,9 +575,29 @@ pub fn analyze( options, flags: Vec::new(), notes: Vec::new(), + prediction: None, }; - flag_item(&mut analysis, t, design.as_ref(), &rows); + // Whether the authored expectation is evidence or a guess. An item that + // has never been administered has no calibration block, and one edited + // since its last calibration has a fingerprint that no longer matches. + let calibrated = record + .and_then(|r| r.placement(*number)) + .and_then(|p| catalog.and_then(|c| c.get(&p.item))) + .and_then(|entry| { + let cal = entry.item.calibration.as_ref()?; + let enough = cal.n_examinees.unwrap_or(0) >= t.calibrated_n; + let current = match &cal.fingerprint { + Some(recorded) => *recorded == entry.item.fingerprint(), + // An older calibration block with no fingerprint cannot be + // shown stale, so it is taken at its word. + None => true, + }; + Some(enough && current) + }) + .unwrap_or(false); + + flag_item(&mut analysis, t, design.as_ref(), &rows, calibrated); p_values.push(p_value); if let Some(r) = rpb { @@ -521,11 +623,13 @@ pub fn analyze( /// * `t` - the thresholds. /// * `design` - the authored expectation, when available. /// * `rows` - the raw responses, for partial-credit detection. +/// * `calibrated` - whether that expectation rests on prior data. fn flag_item( a: &mut ItemAnalysis, t: &Thresholds, design: Option<&Design>, rows: &[&crate::responses::Response], + calibrated: bool, ) { // Discrimination first: it is the finding that changes what you do. match a.point_biserial { @@ -678,31 +782,71 @@ fn flag_item( } } - // Did the item behave as authored? + // Did the item behave as authored? This is the one check whose meaning + // depends on where the expectation came from, so it is recorded either way + // and flagged only when the expectation had data behind it. if let Some(d) = design { + let mut prediction = Prediction { + expected_p: d.expected_difficulty, + expected_band: d.expected_discrimination.map(|b| b.expected_band()), + calibrated, + p_error: None, + p_within: None, + band_hit: None, + notes: Vec::new(), + }; + if let Some(expected) = d.expected_difficulty { - if (expected - a.p_value).abs() > t.design_tolerance { - a.flags.push(Flag::DesignMismatch); - a.notes.push(format!( - "you expected about {:.0}% correct and observed {:.0}%. Worth knowing whether \ - your model of the students or the item is off.", - expected * 100.0, - a.p_value * 100.0 - )); + let error = a.p_value - expected; + let within = error.abs() <= t.design_tolerance; + prediction.p_error = Some(error); + prediction.p_within = Some(within); + if !within { + if calibrated { + a.flags.push(Flag::DesignMismatch); + prediction.notes.push(format!( + "this item is calibrated at about {:.0}% correct and came out at {:.0}%. \ + Something changed: the cohort, the teaching, or the item.", + expected * 100.0, + a.p_value * 100.0 + )); + } else { + prediction.notes.push(format!( + "you predicted about {:.0}% correct and observed {:.0}%. This is the \ + first data on the item, so it corrects the prediction rather than \ + condemning the item.", + expected * 100.0, + a.p_value * 100.0 + )); + } } } + if let (Some(band), Some(r)) = (d.expected_discrimination, a.point_biserial) { let (low, high) = band.expected_band(); - if r < low || r > high { - if !a.flags.contains(&Flag::DesignMismatch) { - a.flags.push(Flag::DesignMismatch); + let hit = r >= low && r <= high; + prediction.band_hit = Some(hit); + if !hit { + if calibrated { + if !a.flags.contains(&Flag::DesignMismatch) { + a.flags.push(Flag::DesignMismatch); + } + prediction.notes.push(format!( + "calibrated for {} discrimination ({low:.2} to {high:.2}), observed \ + {r:.2}.", + format!("{band:?}").to_lowercase() + )); + } else { + prediction.notes.push(format!( + "you predicted {} discrimination ({low:.2} to {high:.2}) and observed \ + {r:.2}.", + format!("{band:?}").to_lowercase() + )); } - a.notes.push(format!( - "you expected {} discrimination ({low:.2} to {high:.2}) and observed {r:.2}.", - format!("{band:?}").to_lowercase() - )); } } + + a.prediction = Some(prediction); } a.flags.sort(); @@ -771,11 +915,30 @@ fn reliability(coded: &[Vec], totals: &[f64], p_values: &[f64], rpbs: &[f64 /// # Returns /// /// The letters that appear on full-credit responses. +/// The single variant every row for one question names, if they agree. +/// +/// Disagreement is not an error, it is a fact about the administration: a course +/// that draws distractors per form has two option sets under one question +/// number, and no pooled statistic describes both. Returning `None` is what +/// keeps the per-variant records from claiming otherwise. +fn one_variant(set: &ResponseSet, number: u32) -> Option { + let mut seen: Option<&str> = None; + for row in set.rows.iter().filter(|r| r.item_number == number) { + let variant = row.variant.as_deref()?; + match seen { + None => seen = Some(variant), + Some(first) if first == variant => {} + Some(_) => return None, + } + } + seen.map(str::to_string) +} + fn infer_key(rows: &[&crate::responses::Response]) -> Vec { let mut out: BTreeSet = BTreeSet::new(); for r in rows { if r.credit >= 0.999 { - for letter in &r.selected { + for letter in r.chosen() { out.insert(letter.clone()); } } @@ -857,6 +1020,7 @@ mod tests { assessment_id: "a".into(), date: None, form: None, + form_position: None, student_key: student.into(), sid: None, name: None, @@ -865,22 +1029,26 @@ mod tests { item_number: number, item_ref: None, item_version: None, + variant: None, selected: if letter.is_empty() { vec![] } else { vec![letter.to_string()] }, + selected_source: vec![], eliminated: vec![], + eliminated_source: vec![], correct: Some(credit >= 0.999), credit, points_possible: 1.0, score: credit, response_time_seconds: None, level: None, - learning_objectives: vec![], + learning_targets: vec![], topics: vec![], bonus: false, dropped: false, + dropped_full_credit: false, } } diff --git a/src/analysis/diagnostic.rs b/src/analysis/diagnostic.rs new file mode 100644 index 0000000..75810ea --- /dev/null +++ b/src/analysis/diagnostic.rs @@ -0,0 +1,2526 @@ +// SPDX-License-Identifier: Prosperity-3.0.0 +// Copyright Scientific Computing Studio +// Source: https://git.scient.ing/education/coursebank + +//! What to tell a student, and what to tell yourself. +//! +//! [`crate::students`] computes mastery. [`crate::classical`] computes item +//! statistics. Neither decides what belongs in a document, and that decision is +//! not a formatting concern: it determines what a student is able to reconstruct +//! from the page in front of them. +//! +//! So this module assembles two view models, and the interesting property of the +//! first one is what it does not contain. +//! +//! # The withheld-question invariant +//! +//! [`StudentDiagnostic`] has no field for a stem and no field for option text. +//! Not an empty one, not one gated behind a flag: the struct has no such field, so +//! no template can print what it was never given. This is the same argument +//! [`crate::typst::config::Reveal`] makes about the exam paper, for the same +//! reason — a flag is one forgotten `if` away from a bad afternoon, and a +//! diagnostic that carries the questions cannot be handed back before the makeup +//! exam is given. +//! +//! What a student does get, per missed question: the number, its level, the +//! objectives it measured and what those objectives ask, whether they answered +//! it, the feedback written for the specific option they chose, the hint written +//! for that same option, and the item's own `review` citations. That is why +//! authoring distractors carefully pays off twice. +//! +//! Two further fields are available and off by default, because they are the two +//! that trade a student's understanding against reusing the question. The +//! misconception is written to you about the student; the worked solution is the +//! solutions document. See [`Options::misconceptions`] and [`Options::solutions`]. +//! +//! It is worth being clear about what the default already discloses. The +//! per-option feedback on a missed question routinely names the right answer, +//! because that is what makes it useful. A report handed to sixty students is +//! therefore already a partial answer key for the questions those students +//! missed, with or without the two optional fields. +//! +//! Two consequences of that invariant are worth stating because they are easy to +//! undo by accident: +//! +//! *Only student-facing feedback is used.* [`crate::item::Choice::student_text`] +//! falls back to `explanation`, which is instructor-facing and routinely says +//! which option is right. This module reads `feedback_student` and `misconception` +//! and nothing else. +//! +//! *Feedback is looked up by bank letter, not printed letter.* On a shuffled form +//! those differ, and looking up the printed letter returns another option's +//! misconception — confident, specific, and about a question the student did not +//! answer that way. See [`crate::decode`]. +//! +//! # What to study +//! +//! A list of missed objectives is a diagnosis, not a prescription. The study plan +//! resolves each weak objective through the course's own reading registry, so a +//! student is pointed at `KKW §6.2` with the sentence you wrote about what to take +//! from it, rather than at the name of a chapter. +//! +//! [`LectureFocus`] answers the question a student actually asks, which is where +//! to start. It ranks the lectures behind the missed questions by how many +//! objectives went wrong in each, so a reading list of eleven sections becomes an +//! ordered afternoon. + +use std::collections::{BTreeMap, BTreeSet}; + +use serde::Serialize; + +use crate::assessment::AssessmentFile; +use crate::catalog::Catalog; +use crate::classical::Analysis; +use crate::course::{CourseFile, ReadingRole}; +use crate::irt::Fit; +use crate::item::Citation; +use crate::responses::{Response, ResponseSet}; +use crate::students::{Cohort, Mastery, StudentSummary}; +use crate::taxonomy::{Level, Tier}; + +/// What to assemble. +#[derive(Debug, Clone)] +pub struct Options { + /// Whether to compare the student to the class. + pub comparison: bool, + /// Whether to include the per-question map. + /// + /// The map names question numbers, never their content. It is what lets a + /// student who has their paper back line the two up. + pub questions: bool, + /// Whether to include the feedback written for the option the student chose. + pub feedback: bool, + /// Whether to include the hint written for that option. + /// + /// On by default. A hint is the question you would ask a student who was + /// reconsidering that option, so it gives them somewhere to start rather than + /// a verdict to accept. + pub hints: bool, + /// Whether to name the misconception the chosen distractor was written to + /// catch. + /// + /// Off by default, and not because it is unsafe. The text is written to you, + /// about the student, in the third person, and next to the feedback written + /// for them it reads like a chart note. + pub misconceptions: bool, + /// Whether to include the worked solution for a missed question. + /// + /// Off by default. [`crate::item::Solution::explanation`] is the derivation, + /// the estimate, and the argument for the key over its neighbours: the body of + /// the solutions document. Turning this on hands that to every student who + /// missed the question, which is the right call for a question you will not + /// use again and the wrong one for a bank you reuse each term. + pub solutions: bool, + /// How many objectives to build a study plan for. + pub focus_limit: usize, + /// How many readings to list per objective. + pub readings_per_objective: usize, + /// Whether to include the IRT ability estimate. + pub ability: bool, +} + +impl Default for Options { + fn default() -> Options { + Options { + comparison: true, + questions: true, + feedback: true, + hints: true, + misconceptions: false, + solutions: false, + focus_limit: 4, + readings_per_objective: 2, + ability: false, + } + } +} + +/// Everything one student's diagnostic says. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct StudentDiagnostic { + /// The grouping key, which is a pseudonym when the store is pseudonymized. + pub student_key: String, + /// The display name, when identifiers were kept. + #[serde(skip_serializing_if = "Option::is_none")] + pub name: Option, + /// The institutional id, when identifiers were kept. + #[serde(skip_serializing_if = "Option::is_none")] + pub sid: Option, + /// Their email, when the export carried one and identifiers were kept. + /// + /// Absent rather than blank when the platform did not report it, so a + /// template prints nothing instead of an empty label. + #[serde(skip_serializing_if = "Option::is_none")] + pub email: Option, + /// Which form they sat. + #[serde(skip_serializing_if = "Option::is_none")] + pub form: Option, + /// Their score. + pub score: Score, + /// How they compare to the class, when comparison is on. + #[serde(skip_serializing_if = "Option::is_none")] + pub standing: Option, + /// Per-level performance. + pub levels: Vec, + /// Per-objective standing, in the course's own order. + /// + /// Objectives only, never their targets. This is the table that makes a + /// claim, and a claim needs a denominator: an objective's row aggregates + /// every item tagged to any of its targets, while a target's row usually + /// rests on one question and could only ever read "not enough questions to + /// say". Mixing the two produced a three-page table where most rows carried + /// that mark and the few real classifications were lost among them. The + /// specifics live in the two sections built for them: which lectures to go + /// back to, and the notes on missed questions. + pub objectives: Vec, + /// Objectives they are clearly meeting, worst first among the confident ones. + pub strengths: Vec, + /// Objectives to work on, worst first. + pub focus: Vec, + /// One row per question, with no question in it. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub questions: Vec, + /// Questions thrown out, in number order. + /// + /// Reported separately from the question rows so the document can say once, + /// in prose, what happened to them. The two kinds need different sentences: + /// a removed question is gone from the denominator, and a full-credit + /// question is still in it. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub dropped_questions: Vec, + /// Which lectures to go back to, the one that would repay the most time + /// first. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub review_lectures: Vec, + /// What to read, grouped by objective. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub study: Vec, +} + +/// A score, with the denominators spelled out. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct Score { + /// Points earned on scored items. + pub points: f64, + /// Points available on scored items. + pub points_possible: f64, + /// Percentage on scored items. + pub percent: f64, + /// Bonus points earned. + pub bonus_points: f64, + /// Items answered correctly. + pub correct: usize, + /// Scored items administered. + pub n_items: usize, +} + +/// Where a score sits in the class. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct Standing { + /// The class mean percentage. + pub class_mean: f64, + /// The standard deviation of class percentages. + pub class_sd: f64, + /// A coarse band, never a rank. + pub band: String, + /// The IRT ability estimate, when asked for. + #[serde(skip_serializing_if = "Option::is_none")] + pub theta: Option, + /// Its standard error. + #[serde(skip_serializing_if = "Option::is_none")] + pub theta_se: Option, +} + +/// One cognitive level. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct LevelRow { + /// The level code, 1 through 5. + pub level: u8, + /// The level name. + pub name: String, + /// What that level asks of a student, in one phrase. + pub blurb: String, + /// How many items at this level. + pub n_items: usize, + /// The student's rate. + pub rate: f64, + /// The class rate, when comparison is on. + #[serde(skip_serializing_if = "Option::is_none")] + pub class_rate: Option, + /// A plain-language comparison, when comparison is on. + #[serde(skip_serializing_if = "Option::is_none")] + pub comparison: Option, +} + +/// One objective's standing. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct ObjectiveRow { + /// The objective id. + pub id: String, + /// The objective text. + pub text: String, + /// The unit it belongs to, when the course declares one. + #[serde(skip_serializing_if = "Option::is_none")] + pub unit: Option, + /// How many items measured it. + pub n_items: usize, + /// Credit earned across them. + pub credit: f64, + /// The observed rate. + pub rate: f64, + /// The lower bound of the 95% Wilson interval. + pub lower: f64, + /// The upper bound. + pub upper: f64, + /// The class rate, when comparison is on. + #[serde(skip_serializing_if = "Option::is_none")] + pub class_rate: Option, + /// The mastery classification: `meeting`, `developing`, `not yet`, or + /// `not enough evidence`. + pub status: String, + /// A compact symbol for the same thing. + pub symbol: String, + /// Whether the interval, not just the estimate, clears the threshold. + pub confident: bool, + /// Whether too few items measured it to classify at all. This is a fact about + /// the exam, and a report that says so is being honest rather than vague. + pub thin_evidence: bool, + /// The levels it was assessed at, as codes. + pub levels: Vec, +} + +/// An objective named in a list. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct ObjectiveRef { + /// The objective id. + pub id: String, + /// The objective text. + pub text: String, + /// The observed rate. + pub rate: f64, + /// How many items measured it. + pub n_items: usize, +} + +/// What one question measured, named at both tiers. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct Measured { + /// The objective this question's result rolls up to. + /// + /// `None` when the tagged id is an objective with no targets of its own, so + /// that a report does not print the same sentence twice. + #[serde(skip_serializing_if = "Option::is_none")] + pub objective: Option, + /// The target the question was written against. + pub target: String, +} + +/// One question, described without being reproduced. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct QuestionRow { + /// The recorded question number, which is what is printed on the paper. + pub number: u32, + /// Where it sat on this student's form, when the forms differ. + #[serde(skip_serializing_if = "Option::is_none")] + pub position: Option, + /// The level code. + #[serde(skip_serializing_if = "Option::is_none")] + pub level: Option, + /// The learning targets it measured, by id. + pub targets: Vec, + /// What this question measured, at both tiers. + /// + /// Both, because each answers a different question a student has in front of + /// a missed item. The target says what this question actually asked of them, + /// which is the specific thing to go and practise. The objective says which + /// row of the table above the mark landed in, which is how they tell whether + /// one slip cost them a claim or whether it was one of several. Printing the + /// target alone left them unable to connect the note to the table; printing + /// the objective alone described something broader than the question. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub measured: Vec, + /// Whether it was answered correctly. + #[serde(skip_serializing_if = "Option::is_none")] + pub correct: Option, + /// Credit earned, as a fraction. + pub credit: f64, + /// Whether it was a bonus question. + #[serde(skip_serializing_if = "std::ops::Not::not")] + pub bonus: bool, + /// Whether it was dropped from scoring after the fact. + /// + /// A dropped question still appears in the map, because the student has the + /// paper in front of them and will look for it. What it must not do is + /// appear as an error they made. + #[serde(skip_serializing_if = "std::ops::Not::not")] + pub dropped: bool, + /// Whether the student left it blank. + pub blank: bool, + /// The share of the class that answered it correctly, when comparison is on. + #[serde(skip_serializing_if = "Option::is_none")] + pub class_rate: Option, + /// The feedback written for the option this student chose. + #[serde(skip_serializing_if = "Option::is_none")] + pub feedback: Option, + /// The hint written for that option: where to look, not what the answer was. + #[serde(skip_serializing_if = "Option::is_none")] + pub hint: Option, + /// The misconception that option was written to catch, when + /// [`Options::misconceptions`] is on. + #[serde(skip_serializing_if = "Option::is_none")] + pub misconception: Option, + /// The worked solution, when [`Options::solutions`] is on. + #[serde(skip_serializing_if = "Option::is_none")] + pub worked: Option, + /// Where the material was taught. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub taught_in: Vec, + /// What to read again about this question, from the item's own `review` + /// citations rather than from the objective's reading list. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub review: Vec, +} + +/// One question thrown out after the exam. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct DroppedQuestion { + /// The recorded question number. + pub number: u32, + /// Whether the drop was applied by crediting every option, in which case the + /// question is still in the points of record. + pub full_credit: bool, +} + +/// One citation to read again after missing a question. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct ItemReading { + /// A short citation, e.g. `KKW §2.1`. + pub citation: String, + /// The work's full title, for a student who does not recognise the label. + #[serde(skip_serializing_if = "Option::is_none")] + pub title: Option, + /// A link, when the citation resolves to one. + #[serde(skip_serializing_if = "Option::is_none")] + pub url: Option, +} + +/// One lecture worth going back to, with the evidence for saying so. +/// +/// Ranked by how many *objectives* went wrong rather than how many questions +/// did. Missing four questions on one objective is one thing to relearn; missing +/// four questions across four objectives is four, and the second is the lecture +/// to reread first even though the arithmetic looks identical. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct LectureFocus { + /// The lecture id, e.g. `L1.4`. + pub lecture: String, + /// Its title. + pub title: String, + /// Where the slides live, when the course records that. + #[serde(skip_serializing_if = "Option::is_none")] + pub url: Option, + /// How many distinct learning targets from this lecture were missed. The + /// ranking key. + /// + /// Counting targets rather than objectives keeps the ranking informative: a + /// lecture where four separate performances went wrong needs more time than + /// one where a single performance was missed twice, and counting objectives + /// would score those the same. + pub n_targets: usize, + /// How many questions from this lecture were missed. + pub n_questions: usize, + /// Which questions, so a student can line this up with their paper. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub questions: Vec, + /// The slides those questions came from, when the items record them. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub slides: Vec, + /// The learning targets that went wrong here, in the course's own words. + /// + /// Targets rather than objectives, because this section answers "what do I + /// go and restudy". "You missed the objective on binding" sends a student to + /// a whole lecture; "you missed reading a dissociation constant off an + /// isotherm" sends them to one page of it. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub targets: Vec, +} + +/// What to read about one objective. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct StudyGroup { + /// The objective id. + pub objective: String, + /// The objective text. + pub text: String, + /// The observed rate, so the list is ordered by need. + pub rate: f64, + /// The readings. + pub readings: Vec, +} + +/// One reading to revisit. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct StudyReading { + /// A short citation, e.g. `KKW §6.2`. + pub citation: String, + /// The lecture it was assigned for. + pub lecture: String, + /// That lecture's title. + pub lecture_title: String, + /// A link, when the reference resolves to one. + #[serde(skip_serializing_if = "Option::is_none")] + pub url: Option, + /// What to take from it, which is the sentence worth quoting at someone who + /// missed the objective. + #[serde(skip_serializing_if = "Option::is_none")] + pub focus: Option, + /// What the section covers. + #[serde(skip_serializing_if = "Option::is_none")] + pub summary: Option, + /// Whether it was assigned or offered alongside. + pub supplemental: bool, +} + +/// Builds one student's diagnostic. +/// +/// # Arguments +/// +/// * `summary` - the student's computed summary. +/// * `cohort` - the class context. +/// * `catalog` - the loaded course, for objective text, feedback, and readings. +/// * `set` - the responses, for this student's per-question rows. +/// * `analysis` - the item analysis, for class rates per question. Optional. +/// * `opts` - what to include. +/// +/// # Returns +/// +/// The diagnostic. +pub fn student( + summary: &StudentSummary, + cohort: &Cohort, + catalog: &Catalog, + set: &ResponseSet, + analysis: Option<&Analysis>, + opts: &Options, +) -> StudentDiagnostic { + let course = &catalog.course; + let rows = set.for_student(&summary.student_key); + let form = rows.first().and_then(|r| r.form.clone()); + + // The summary carries the name; the email only ever existed on the response + // rows, and both are already absent from a pseudonymized store, so neither + // needs a flag here. + let name = summary + .name + .clone() + .or_else(|| rows.iter().find_map(|r| r.name.clone())); + let email = rows.iter().find_map(|r| r.email.clone()); + + let levels = summary + .levels + .iter() + .map(|profile| LevelRow { + level: profile.level.code(), + name: profile.level.name().to_string(), + blurb: profile.level.blurb().to_string(), + n_items: profile.n_items, + rate: profile.rate, + class_rate: opts.comparison.then_some(profile.cohort_rate), + comparison: opts.comparison.then(|| profile.comparison().to_string()), + }) + .collect(); + + let objectives: Vec = summary + .objectives + .iter() + // Objectives only; see `StudentDiagnostic::objectives`. + .filter(|mastery| mastery.tier == Tier::Objective) + .map(|mastery| ObjectiveRow { + id: mastery.id.clone(), + text: mastery.text.clone(), + unit: course.objective_unit(&mastery.id).map(str::to_string), + n_items: mastery.n_items, + credit: mastery.credit, + rate: mastery.rate, + lower: mastery.wilson_lower, + upper: mastery.wilson_upper, + class_rate: opts.comparison.then_some(mastery.cohort_rate), + status: mastery.status.label().to_string(), + symbol: mastery.status.symbol().to_string(), + confident: mastery.confident, + thin_evidence: mastery.status == Mastery::NotEnoughEvidence, + levels: mastery.levels.iter().map(|l| l.code()).collect(), + }) + .collect(); + + let refs = |ids: &[String]| -> Vec { + ids.iter() + .filter_map(|id| objectives.iter().find(|o| &o.id == id)) + .map(|o| ObjectiveRef { + id: o.id.clone(), + text: o.text.clone(), + rate: o.rate, + n_items: o.n_items, + }) + .collect() + }; + + let strengths = refs(&summary.strengths); + let focus = refs(&summary.focus); + + let class_rates: BTreeMap = analysis + .map(|a| a.items.iter().map(|i| (i.number, i.p_value)).collect()) + .unwrap_or_default(); + + let questions = if opts.questions { + rows.iter() + .map(|row| question_row(row, catalog, &class_rates, opts)) + .collect() + } else { + Vec::new() + }; + + let mut dropped_questions: Vec = rows + .iter() + .filter(|r| r.dropped) + .map(|r| DroppedQuestion { + number: r.item_number, + full_credit: r.dropped_full_credit, + }) + .collect(); + dropped_questions.sort_by_key(|d| d.number); + dropped_questions.dedup_by_key(|d| d.number); + + // Built from the response rows rather than from `questions`, so a report with + // `--no-questions` still says where to go back to; it just does not name the + // question numbers. + let review_lectures = lecture_focus(catalog, &rows, opts); + + let study = focus + .iter() + .take(opts.focus_limit) + .map(|objective| StudyGroup { + objective: objective.id.clone(), + text: objective.text.clone(), + rate: objective.rate, + readings: readings_for(course, &objective.id, opts.readings_per_objective), + }) + .filter(|group| !group.readings.is_empty()) + .collect(); + + StudentDiagnostic { + student_key: summary.student_key.clone(), + name, + sid: summary.sid.clone(), + email, + form, + score: Score { + points: summary.points, + points_possible: summary.points_possible, + percent: summary.percent, + bonus_points: summary.bonus_points, + correct: summary.correct, + n_items: summary.n_items, + }, + standing: opts.comparison.then(|| Standing { + class_mean: cohort.mean_percent, + class_sd: cohort.sd_percent, + band: summary.band.clone(), + theta: opts.ability.then_some(summary.theta).flatten(), + theta_se: opts.ability.then_some(summary.theta_se).flatten(), + }), + levels, + objectives, + strengths, + focus, + questions, + dropped_questions, + review_lectures, + study, + } +} + +/// Builds one question's row. +/// +/// The feedback lookup uses [`Response::chosen`], which prefers the bank letters +/// written at ingest. Falling back to the printed letters is right for an +/// unshuffled form and wrong for a shuffled one, which is why ingest translates +/// rather than leaving it to here. +fn question_row( + row: &Response, + catalog: &Catalog, + class_rates: &BTreeMap, + opts: &Options, +) -> QuestionRow { + let blank = row.selected.is_empty() && row.eliminated.is_empty(); + // A dropped question cannot be missed. Without `counts()` here, a question + // thrown out after the exam still collects per-option feedback explaining an + // error the student is no longer being charged for. + let missed = row.credit < 0.999 && row.counts(); + let mut feedback = None; + let mut hint = None; + let mut misconception = None; + let mut worked = None; + let mut taught_in = Vec::new(); + let mut review = Vec::new(); + + if let Some(uid) = row.item_ref.as_deref() { + if let Some(entry) = catalog.get(uid) { + if missed { + if let Some(letter) = row.chosen().first() { + if let Some(choice) = entry.item.option(letter) { + if opts.feedback { + // `student_text` falls back to `explanation`, which is + // written for a grader and often names the right + // answer. A student report must not print it. + feedback = choice + .feedback_student + .clone() + .or_else(|| choice.misconception.clone()); + } + if opts.hints { + hint = choice.hint.clone(); + } + // When an option carries no student feedback, the + // fallback above already printed this text. The same + // sentence twice under two labels reads as a bug. + if opts.misconceptions && choice.misconception != feedback { + misconception = choice.misconception.clone(); + } + } + } + + if let Some(solution) = entry.item.solution.as_ref() { + if opts.solutions { + worked = solution.explanation.clone(); + } + review = item_readings(&catalog.course, &solution.review); + } + } + for source in &entry.item.sources { + let title = catalog + .course + .lectures + .get(&source.lecture) + .map(|l| l.title.clone()) + .unwrap_or_else(|| source.lecture.clone()); + if source.slides.is_empty() { + taught_in.push(format!("{} ({})", title, source.lecture)); + } else { + let slides: Vec = source.slides.iter().map(|s| s.to_string()).collect(); + taught_in.push(format!( + "{} ({}), slide{} {}", + title, + source.lecture, + if source.slides.len() == 1 { "" } else { "s" }, + slides.join(", ") + )); + } + } + } + } + + QuestionRow { + number: row.item_number, + position: row.form_position.filter(|p| *p != row.item_number), + level: row.level.map(|l| l.code()), + targets: row.learning_targets.clone(), + measured: if missed { + let course = &catalog.course; + row.learning_targets + .iter() + .map(|id| { + let objective = course.objective_for(id); + Measured { + // An objective with no targets of its own is tagged + // directly, and then the two tiers are the same row. + // Saying it twice would read as an error, so the + // objective is left out. + objective: (objective != id).then(|| course.text_for(objective)), + target: course.text_for(id), + } + }) + .collect() + } else { + // Only where it earns its space. Every question already carries its + // target ids, and a correct answer needs no explaining. + Vec::new() + }, + correct: row.correct, + credit: row.credit, + bonus: row.bonus, + dropped: row.dropped, + blank, + class_rate: opts + .comparison + .then(|| class_rates.get(&row.item_number).copied()) + .flatten(), + feedback, + hint, + misconception, + worked, + taught_in, + review, + } +} + +/// Resolves an item's `review` citations against the course reference registry. +/// +/// # Arguments +/// +/// * `course` - the course, for its reference labels and base URLs. +/// * `citations` - the item's citations. +/// +/// # Returns +/// +/// One entry per citation that resolves to something printable. +fn item_readings(course: &CourseFile, citations: &[Citation]) -> Vec { + let mut out = Vec::new(); + for citation in citations { + let reference = citation + .reference + .as_deref() + .and_then(|key| course.references.get(key).map(|r| (key, r))); + + let (label, title, url) = match reference { + Some((key, reference)) => ( + reference.label_or(key).to_string(), + Some(reference.title.clone()), + citation.href(reference), + ), + None => (citation.display(), None, citation.url.clone()), + }; + if label.is_empty() { + continue; + } + + let citation_text = match (&citation.text, &citation.locator) { + (Some(text), _) => text.clone(), + (None, Some(locator)) => format!("{label} {locator}"), + (None, None) => label, + }; + out.push(ItemReading { + citation: citation_text, + title, + url, + }); + } + out +} + +/// Ranks the lectures behind a student's missed questions. +/// +/// A lecture earns its place by how many distinct objectives went wrong in it, +/// then by how many questions, then by id so the order is stable between runs. +/// +/// # Arguments +/// +/// * `catalog` - the loaded course, for objective and lecture titles. +/// * `rows` - this student's responses. +/// * `opts` - what to include; `questions` decides whether numbers are named. +/// +/// # Returns +/// +/// The lectures, the one that would repay the most time first. +fn lecture_focus(catalog: &Catalog, rows: &[&Response], opts: &Options) -> Vec { + /// What has accumulated for one lecture so far. + #[derive(Default)] + struct Tally { + targets: BTreeSet, + questions: BTreeSet, + slides: BTreeSet, + } + + let course = &catalog.course; + let mut tallies: BTreeMap = BTreeMap::new(); + + for row in rows.iter().filter(|r| r.counts() && r.credit < 0.999) { + // Two routes to a lecture, and both are wanted. The registry knows + // which lectures develop a target; the item knows which lecture it was + // written from, which is the finer answer when a target spans several. + let mut lectures: BTreeSet = BTreeSet::new(); + let mut slides: BTreeMap> = BTreeMap::new(); + + if let Some(entry) = row.item_ref.as_deref().and_then(|uid| catalog.get(uid)) { + for source in &entry.item.sources { + lectures.insert(source.lecture.clone()); + slides + .entry(source.lecture.clone()) + .or_default() + .extend(source.slides.iter().copied()); + } + } + for target in &row.learning_targets { + lectures.extend(course.lectures_for(target).iter().cloned()); + } + + for lecture in lectures { + let tally = tallies.entry(lecture.clone()).or_default(); + tally.targets.extend(row.learning_targets.clone()); + tally.questions.insert(row.item_number); + if let Some(numbers) = slides.get(&lecture) { + tally.slides.extend(numbers.iter().copied()); + } + } + } + + let mut out: Vec = tallies + .into_iter() + .map(|(lecture, tally)| { + let record = course.lectures.get(&lecture); + LectureFocus { + title: record + .map(|l| l.title.clone()) + .unwrap_or_else(|| lecture.clone()), + url: record.and_then(|l| l.slides_url.clone()), + n_targets: tally.targets.len(), + n_questions: tally.questions.len(), + questions: if opts.questions { + tally.questions.iter().copied().collect() + } else { + Vec::new() + }, + slides: tally.slides.iter().copied().collect(), + targets: tally.targets.iter().map(|id| course.text_for(id)).collect(), + lecture, + } + }) + .collect(); + + out.sort_by(|a, b| { + b.n_targets + .cmp(&a.n_targets) + .then(b.n_questions.cmp(&a.n_questions)) + .then(a.lecture.cmp(&b.lecture)) + }); + out +} + +/// Resolves an objective to readings. +/// +/// # Arguments +/// +/// * `course` - the course registry. +/// * `objective` - the objective id. +/// * `limit` - how many readings to keep. +/// +/// # Returns +/// +/// The readings, assigned ones first. +fn readings_for(course: &CourseFile, objective: &str, limit: usize) -> Vec { + let mut out = Vec::new(); + for (lecture_id, reading) in course.readings_for_objective(objective) { + let lecture_title = course + .lectures + .get(lecture_id) + .map(|l| l.title.clone()) + .unwrap_or_else(|| lecture_id.to_string()); + + let (citation, url) = match reading.reference.as_deref() { + Some(key) => match course.references.get(key) { + Some(reference) => (reading.cite(key, reference), reading.resolve_url(reference)), + None => ( + reading + .text + .clone() + .or_else(|| reading.locator.clone()) + .unwrap_or_else(|| key.to_string()), + reading.url.clone(), + ), + }, + None => ( + reading + .text + .clone() + .or_else(|| reading.locator.clone()) + .unwrap_or_else(|| lecture_title.clone()), + reading.url.clone(), + ), + }; + + out.push(StudyReading { + citation, + lecture: lecture_id.to_string(), + lecture_title, + url, + focus: reading.focus.clone(), + summary: reading.summary.clone(), + supplemental: reading.role == ReadingRole::Supplemental, + }); + } + + // Assigned before supplemental, otherwise the order the course declares. + out.sort_by_key(|r| r.supplemental); + out.truncate(limit); + out +} + +/// Everything the class diagnostic says. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct CohortDiagnostic { + /// How many students sat it. + pub n_students: usize, + /// How many scored items. + pub n_items: usize, + /// The score distribution. + pub distribution: Distribution, + /// Whole-test reliability. + pub reliability: ReliabilityRow, + /// Per-level class performance. + pub levels: Vec, + /// Per-objective class performance, worst first. + pub objectives: Vec, + /// Objectives the class as a whole did not meet. + pub gaps: Vec, + /// The distribution binned by the course's letter-grade scale. Empty when + /// `course.yaml` sets no scale, in which case the ten-point bins stand. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub grades: Vec, + /// Per-lecture class performance, worst first. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub lectures: Vec, + /// Per-question statistics. + pub questions: Vec, + /// Questions thrown out, which are therefore absent from every table above. + /// Recorded so the report says why rather than leaving a gap in the + /// numbering. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub dropped_questions: Vec, + /// What to do about each question that raised something. + pub triage: Triage, + /// How the authored expectations did. + pub predictions: PredictionSummary, + /// Questions worth revisiting before reuse, worst first. + pub revise: Vec, + /// The dropped items, described but not measured. + /// + /// Kept out of [`CohortDiagnostic::questions`] so that no statistic above + /// silently includes an item that was thrown out, and reported alongside it + /// in the evidence section so that dropping a question does not erase the + /// evidence for having dropped it. + pub dropped_detail: Vec, + /// One row per form, when more than one was given. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub forms: Vec, + /// Where the form built differs from the blueprint it was drawn against. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub blueprint: Vec, + /// Response profiles the class falls into. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub patterns: Vec, + /// Cautions about the analysis itself. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub warnings: Vec, +} + +/// The score distribution, binned. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct Distribution { + /// Mean percentage. + pub mean: f64, + /// Median percentage. + pub median: f64, + /// Standard deviation. + pub sd: f64, + /// Lowest percentage. + pub min: f64, + /// Highest percentage. + pub max: f64, + /// Counts in ten-point bins, from 0-9 through 90-100. + pub bins: Vec, +} + +/// One histogram bin. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct Bin { + /// Inclusive lower bound, in percent. + pub low: u32, + /// Exclusive upper bound, in percent, except the last bin which includes 100. + pub high: u32, + /// How many students fell in it. + pub count: usize, +} + +/// Reliability, flattened for a template. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct ReliabilityRow { + /// KR-20, when it could be computed. + #[serde(skip_serializing_if = "Option::is_none")] + pub alpha: Option, + /// The standard error of measurement, in items. + #[serde(skip_serializing_if = "Option::is_none")] + pub sem: Option, + /// Mean p-value across items. + pub mean_p: f64, + /// Mean point-biserial across items that had one. + #[serde(skip_serializing_if = "Option::is_none")] + pub mean_point_biserial: Option, + /// What the alpha value means for a test this length in a class this size. + pub interpretation: String, +} + +/// One level, class-wide. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct CohortLevelRow { + /// The level code. + pub level: u8, + /// The level name. + pub name: String, + /// How many items sat at this level. + pub n_items: usize, + /// The class rate. + pub rate: f64, +} + +/// One objective, class-wide. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct CohortObjectiveRow { + /// The objective id. + pub id: String, + /// The objective text. + pub text: String, + /// How many items measured it. + pub n_items: usize, + /// The class rate. + pub rate: f64, + /// How many students met it. + pub meeting: usize, + /// How many students are developing on it. + pub developing: usize, + /// How many students are not yet meeting it. + pub not_yet: usize, + /// How many had too few items to classify. + pub thin: usize, + /// Whether the class rate is below the course's mastery threshold. + pub below_threshold: bool, +} + +/// One question, class-wide. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct CohortQuestionRow { + /// The recorded question number. + pub number: u32, + /// The item's global id. + #[serde(skip_serializing_if = "Option::is_none")] + pub item: Option, + /// The level code. + #[serde(skip_serializing_if = "Option::is_none")] + pub level: Option, + /// The learning targets it measured, by id. + pub targets: Vec, + /// What those targets ask, in the course's own words. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub target_texts: Vec, + /// Whether the item was dropped from scoring. + /// + /// A dropped item carries no p, r, or D, and appears in none of the + /// statistics above. It still appears in the evidence section, because the + /// option spread that justified dropping it is the record of why, and that + /// record should survive re-running the report afterwards. + #[serde(skip_serializing_if = "std::ops::Not::not")] + pub dropped: bool, + /// Whether the drop was applied as full credit to everyone. + #[serde(skip_serializing_if = "std::ops::Not::not")] + pub dropped_full_credit: bool, + /// The question as written. + /// + /// The instructor report reads better with it than without: a row of option + /// shares says a distractor drew 44% of the class, and only the stem says + /// whether that is a second defensible reading. Absent when the item has + /// left the bank, and withheld from any report that is not the instructor + /// copy. + #[serde(skip_serializing_if = "Option::is_none")] + pub stem: Option, + /// Where the item was taught, as lecture titles and slide numbers. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub taught_in: Vec, + /// The lecture ids alone, for a table column where only `L1.4` fits. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub lectures: Vec, + /// Which difficulty band it fell in: `too easy`, `moderate`, or `hard`. + pub difficulty_band: String, + /// Which discrimination band it fell in, on the conventional cut points: + /// `excellent`, `good`, `marginal`, `poor`, or `negative`. + pub discrimination_band: String, + /// Proportion correct. + pub p_value: f64, + /// Corrected item-total point-biserial. + #[serde(skip_serializing_if = "Option::is_none")] + pub point_biserial: Option, + /// Upper minus lower group proportion correct. + #[serde(skip_serializing_if = "Option::is_none")] + pub discrimination: Option, + /// Fraction who left it blank. + pub blank_rate: f64, + /// The keyed letters. + pub key: Vec, + /// Per-option selection, in letter order. + pub options: Vec, + /// Machine-detected problems. + pub flags: Vec, + /// What those flags mean. + pub notes: Vec, + /// How the item did against its author's expectation. Separate from `notes` + /// because an unmet prediction on an uncalibrated item is a fact about the + /// prediction. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub prediction_notes: Vec, + /// Whether that expectation rested on a prior calibration. + pub calibrated: bool, + /// Per-form proportion correct, when more than one form was given. A gap here + /// on one question, with the rest of the exam in step, points at that + /// question's permutation rather than at the cohort. + #[serde(skip_serializing_if = "BTreeMap::is_empty")] + pub by_form: BTreeMap, +} + +/// One option's selection statistics. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct OptionRow { + /// The bank letter, which is the one every statistic is keyed by. + pub letter: String, + /// The option as written, for a report that shows the question. + /// + /// Absent when the item is no longer in the bank, which is why this is an + /// option rather than an empty string: a missing option and an empty one + /// are different facts. + #[serde(skip_serializing_if = "Option::is_none")] + pub text: Option, + /// What this option was lettered on each printed form, worst case one entry + /// per form. + /// + /// Shuffling means the bank's option C is a different letter on every form, + /// so a statistic reported against C cannot be checked against a student's + /// paper without this map. It is the first thing anyone needs when a student + /// brings a paper to office hours, and working it out by hand from a seal is + /// the kind of task that gets done wrong once and then trusted. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub printed: Vec, + /// How many chose it. + pub count: usize, + /// The share who chose it. + pub rate: f64, + /// Whether it is keyed. + pub is_key: bool, + /// Correlation between choosing it and scoring well elsewhere. + #[serde(skip_serializing_if = "Option::is_none")] + pub point_biserial: Option, + /// Whether it drew nobody, and is therefore doing no work. + pub nonfunctioning: bool, +} + +/// What one bank option was lettered on one form. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct PrintedLetter { + /// The form id. + pub form: String, + /// The letter this option carried on that form's paper. + pub letter: String, +} + +/// One form's summary. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct FormRow { + /// The form id. + pub id: String, + /// How many students sat it. + pub n_students: usize, + /// Their mean percentage. + pub mean: f64, + /// The standard deviation of their percentages. + pub sd: f64, +} + +/// One response profile. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct PatternRow { + /// A label describing the pattern. + pub label: String, + /// How many students fit it. + pub n_students: usize, + /// Mean rate at each level, by level code. + pub level_means: BTreeMap, +} + +/// The class's scores binned by the course's own letter-grade scale. +/// +/// A ten-point histogram is the default because it needs no course +/// configuration, but nobody acts on "nineteen students in the fifties". They +/// act on "nineteen students are failing", and that sentence needs the scale +/// from `course.yaml`. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct GradeRow { + /// The letter. + pub letter: String, + /// The lowest percentage in the band. + pub low: f64, + /// The highest percentage in the band, which is just under the next band's + /// floor, or 100 for the top band. + pub high: f64, + /// Grade points, when the scale records them. + #[serde(skip_serializing_if = "Option::is_none")] + pub gpa: Option, + /// The attainment word, when the scale records one. + #[serde(skip_serializing_if = "Option::is_none")] + pub attainment: Option, + /// The colour group, so A, A- and A+ can be tinted together. + pub group: String, + /// How many students landed in the band. + pub count: usize, + /// Their share of the class, in `0.0..=1.0`. + pub share: f64, + /// How many students are in this band or a higher one. + pub at_or_above: usize, +} + +/// One lecture's showing, aggregated from the items written against it. +/// +/// The objective table answers "which objective went wrong". This answers "which +/// class meeting went wrong", which is the question that maps onto next week. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct CohortLectureRow { + /// The lecture id. + pub lecture: String, + /// Its title. + pub title: String, + /// How many scored items traced back to it. + pub n_items: usize, + /// How many distinct objectives those items measured. + pub n_objectives: usize, + /// How many of those objectives the class did not meet. + pub n_objectives_below: usize, + /// Mean proportion correct across its items. + pub rate: f64, + /// The questions, so the row can be checked against the item table. + pub questions: Vec, + /// The worst objective under this lecture, by class rate. + #[serde(skip_serializing_if = "Option::is_none")] + pub worst_objective: Option, +} + +/// What to do about one question, and why. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct TriageRow { + /// The question number. + pub number: u32, + /// The item id. + #[serde(skip_serializing_if = "Option::is_none")] + pub item: Option, + /// The level code. + #[serde(skip_serializing_if = "Option::is_none")] + pub level: Option, + /// Proportion correct. + pub p_value: f64, + /// Corrected item-total correlation. + #[serde(skip_serializing_if = "Option::is_none")] + pub point_biserial: Option, + /// Upper minus lower group. + #[serde(skip_serializing_if = "Option::is_none")] + pub discrimination: Option, + /// What the question measured, in the course's words: its learning targets. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub targets: Vec, + /// Where it was taught. + #[serde(skip_serializing_if = "Vec::is_empty")] + pub taught_in: Vec, + /// The specific option this recommendation is about, when it is about one. + #[serde(skip_serializing_if = "Option::is_none")] + pub option: Option, + /// That option's share of responses. + #[serde(skip_serializing_if = "Option::is_none")] + pub option_share: Option, + /// That option's correlation with total score. + #[serde(skip_serializing_if = "Option::is_none")] + pub option_point_biserial: Option, + /// The evidence, one clause per line. + pub reasons: Vec, +} + +/// Every question sorted into what to do with it. +/// +/// The first four lists are decisions about items and are mutually exclusive: a +/// question appears in the most severe one that fits, because there is no point +/// rewriting a distractor on an item you are about to discard. `reteach` is not +/// a decision about an item at all, so a question can appear there as well as in +/// one of the others. +#[derive(Debug, Clone, Default, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct Triage { + /// Broken: the evidence says these did not measure what they were scored on. + pub discard: Vec, + /// A second defensible answer with statistical support behind it. + pub rekey: Vec, + /// Weak but salvageable, worth rewriting before reuse. + pub revise: Vec, + /// Sound items the class got wrong. A teaching finding, not an item finding. + pub reteach: Vec, + /// Items whose low discrimination is explained by their difficulty rather + /// than by a fault. Listed so they are not mistaken for work to do. + pub bounded: Vec, + /// How many questions raised nothing at all. + pub clean: usize, +} + +/// How the authored expectations did against the data. +/// +/// This exists so that an uncalibrated bank does not produce one +/// `design_mismatch` per item. Before an item has data, its expected difficulty +/// is a prediction by its author, and the useful summary is whether those +/// predictions run optimistic or pessimistic as a set. +#[derive(Debug, Clone, Default, Serialize)] +#[serde(rename_all = "kebab-case")] +pub struct PredictionSummary { + /// How many items recorded an expected difficulty. + pub n_predicted: usize, + /// How many of those expectations rest on a prior calibration. + pub n_calibrated: usize, + /// Mean of observed minus expected difficulty. Positive means the items came + /// out easier than predicted. + #[serde(skip_serializing_if = "Option::is_none")] + pub mean_signed_error: Option, + /// Mean absolute difficulty error, which is the size of a typical miss. + #[serde(skip_serializing_if = "Option::is_none")] + pub mean_abs_error: Option, + /// How many landed inside the tolerance. + pub n_within: usize, + /// How many items recorded an expected discrimination band. + pub n_band: usize, + /// How many of those landed inside it. + pub n_band_hit: usize, + /// The largest single surprise, as `(question, expected, observed)`. + #[serde(skip_serializing_if = "Option::is_none")] + pub biggest_surprise: Option<(u32, f64, f64)>, +} + +/// Builds the class diagnostic. +/// +/// # Arguments +/// +/// * `analysis` - the classical item analysis. +/// * `cohort` - the per-student summaries and class rates. +/// * `catalog` - the loaded course. +/// * `record` - the assessment record, for the blueprint check. +/// * `set` - the responses, for per-form and per-option breakdowns. +/// * `fit` - an IRT fit, when one was computed. +/// +/// # Returns +/// +/// The diagnostic. +pub fn cohort( + analysis: &Analysis, + cohort: &Cohort, + catalog: &Catalog, + record: &AssessmentFile, + set: &ResponseSet, + fit: Option<&Fit>, +) -> CohortDiagnostic { + let _ = fit; + let course = &catalog.course; + let threshold = course.policy.mastery_threshold; + + let percents: Vec = cohort.students.iter().map(|s| s.percent).collect(); + + let level_counts: BTreeMap = { + let mut counts: BTreeMap> = BTreeMap::new(); + for row in set.rows.iter().filter(|r| r.counts()) { + if let Some(level) = row.level { + counts.entry(level).or_default().insert(row.item_number); + } + } + counts.into_iter().map(|(k, v)| (k, v.len())).collect() + }; + + let levels = Level::ALL + .iter() + .filter_map(|level| { + let rate = cohort.level_rates.get(level).copied()?; + Some(CohortLevelRow { + level: level.code(), + name: level.name().to_string(), + n_items: level_counts.get(level).copied().unwrap_or(0), + rate, + }) + }) + .collect(); + + // Objective counts come from the responses so that an objective assessed by + // two items is not reported as if it had one. Counted per objective, and by + // item number, so a question tagged with two of that objective's targets is + // one item here. + let mut objective_items: BTreeMap> = BTreeMap::new(); + for row in set.rows.iter().filter(|r| r.counts()) { + for objective in &row.learning_targets { + objective_items + .entry(course.objective_for(objective).to_string()) + .or_default() + .insert(row.item_number); + } + } + + // Objectives, not targets: this table is the reteaching queue, and it is + // only usable if it is short enough to read and each row rests on enough + // items to believe. + let mut objectives: Vec = cohort + .objective_rates + .iter() + .map(|(id, rate)| { + let mut meeting = 0; + let mut developing = 0; + let mut not_yet = 0; + let mut thin = 0; + for student in &cohort.students { + if let Some(row) = student.objectives.iter().find(|o| &o.id == id) { + match row.status { + Mastery::Meeting => meeting += 1, + Mastery::Developing => developing += 1, + Mastery::NotYet => not_yet += 1, + Mastery::NotEnoughEvidence => thin += 1, + } + } + } + CohortObjectiveRow { + id: id.clone(), + text: course.text_for(id), + n_items: objective_items.get(id).map(|s| s.len()).unwrap_or(0), + rate: *rate, + meeting, + developing, + not_yet, + thin, + below_threshold: *rate < threshold, + } + }) + .collect(); + objectives.sort_by(|a, b| { + a.rate + .partial_cmp(&b.rate) + .unwrap_or(std::cmp::Ordering::Equal) + .then_with(|| a.id.cmp(&b.id)) + }); + let gaps: Vec = objectives + .iter() + .filter(|o| o.below_threshold) + .cloned() + .collect(); + + let by_form = per_form_p_values(set); + let item_meta: BTreeMap, Vec)> = record + .items + .iter() + .map(|p| { + ( + p.number, + (p.level.map(|l| l.code()), p.learning_targets.clone()), + ) + }) + .collect(); + + // The printed lettering per form, computed from the same two functions the + // exporter and the seal use, so the letters here are the letters on the + // paper rather than a second guess at them. + let forms: Vec<&crate::assessment::Form> = record.forms.iter().collect(); + + let questions: Vec = analysis + .items + .iter() + .map(|item| { + let meta = item_meta.get(&item.number); + let entry = item.item_ref.as_deref().and_then(|uid| catalog.get(uid)); + CohortQuestionRow { + number: item.number, + item: item.item_ref.clone(), + level: meta.and_then(|m| m.0), + targets: meta.map(|m| m.1.clone()).unwrap_or_default(), + target_texts: meta + .map(|m| m.1.iter().map(|id| course.text_for(id)).collect()) + .unwrap_or_default(), + taught_in: item + .item_ref + .as_deref() + .map(|uid| taught_in(catalog, uid)) + .unwrap_or_default(), + lectures: item + .item_ref + .as_deref() + .and_then(|uid| catalog.get(uid)) + .map(|entry| { + entry + .item + .sources + .iter() + .map(|source| source.lecture.clone()) + .collect() + }) + .unwrap_or_default(), + dropped: false, + dropped_full_credit: false, + stem: entry.map(|e| e.item.stem.clone()), + difficulty_band: difficulty_band(item.p_value).to_string(), + discrimination_band: discrimination_band(item.point_biserial).to_string(), + p_value: item.p_value, + point_biserial: item.point_biserial, + discrimination: item.discrimination_index, + blank_rate: item.blank_rate, + key: item.key.clone(), + options: item + .options + .values() + .map(|option| OptionRow { + text: entry.and_then(|e| { + e.item + .options + .iter() + .find(|o| o.id == option.letter) + .map(|o| o.text.clone()) + }), + printed: entry + .map(|e| { + printed_letters( + &forms, + &e.item, + item.item_ref.as_deref(), + &option.letter, + ) + }) + .unwrap_or_default(), + letter: option.letter.clone(), + count: option.count, + rate: option.rate, + is_key: option.is_key, + point_biserial: option.point_biserial, + nonfunctioning: !option.is_key && option.rate <= 0.05, + }) + .collect(), + flags: item.flags.iter().map(|f| f.as_str().to_string()).collect(), + notes: item.notes.clone(), + prediction_notes: item + .prediction + .as_ref() + .map(|p| p.notes.clone()) + .unwrap_or_default(), + calibrated: item.prediction.as_ref().is_some_and(|p| p.calibrated), + by_form: by_form.get(&item.number).cloned().unwrap_or_default(), + } + }) + .collect(); + + let revise: Vec = analysis + .revise_queue() + .iter() + .filter_map(|item| questions.iter().find(|q| q.number == item.number).cloned()) + .collect(); + + let mut dropped_questions: Vec = set + .all_items() + .into_iter() + .filter(|n| set.item_dropped(*n)) + .map(|number| DroppedQuestion { + number, + full_credit: set.for_item(number).iter().any(|r| r.dropped_full_credit), + }) + .collect(); + dropped_questions.sort_by_key(|d| d.number); + + // The dropped items, described from the responses rather than from the + // scoring. Dropping an item overrides its credit, so p, r, and D are + // meaningless for it and are left out. What students actually marked is + // untouched by the drop, and that spread is the evidence that justified it. + let mut dropped_detail: Vec = dropped_questions + .iter() + .map(|dropped| { + let placement = record.placement(dropped.number); + let uid = placement.map(|p| p.item.clone()); + let entry = uid.as_deref().and_then(|uid| catalog.get(uid)); + let responses = set.for_item(dropped.number); + let answered = responses + .iter() + .filter(|r| !r.chosen().is_empty()) + .count() + .max(1); + let keyed: BTreeSet = placement + .map(|p| p.key.iter().cloned().collect()) + .filter(|k: &BTreeSet| !k.is_empty()) + .or_else(|| entry.map(|e| e.item.key_letters().into_iter().collect())) + .unwrap_or_default(); + + // One row per option the item declares, so an option nobody + // marked still shows as unchosen rather than vanishing. + let options: Vec = entry + .map(|e| { + e.item + .options + .iter() + .map(|option| { + let count = responses + .iter() + .filter(|r| r.chosen().contains(&option.id)) + .count(); + OptionRow { + text: Some(option.text.clone()), + printed: printed_letters( + &forms, + &e.item, + uid.as_deref(), + &option.id, + ), + letter: option.id.clone(), + count, + rate: count as f64 / answered as f64, + is_key: keyed.contains(&option.id), + point_biserial: None, + nonfunctioning: false, + } + }) + .collect() + }) + .unwrap_or_default(); + + let targets = placement + .map(|p| p.learning_targets.clone()) + .unwrap_or_default(); + CohortQuestionRow { + number: dropped.number, + item: uid.clone(), + level: placement.and_then(|p| p.level).map(|l| l.code()), + target_texts: targets.iter().map(|id| course.text_for(id)).collect(), + targets, + dropped: true, + dropped_full_credit: dropped.full_credit, + stem: entry.map(|e| e.item.stem.clone()), + taught_in: uid + .as_deref() + .map(|uid| taught_in(catalog, uid)) + .unwrap_or_default(), + lectures: entry + .map(|e| e.item.sources.iter().map(|s| s.lecture.clone()).collect()) + .unwrap_or_default(), + difficulty_band: String::new(), + discrimination_band: String::new(), + // The keyed share before the override, which is the closest + // honest reading of how the item performed. It is not a + // p-value: it counts marks, not credit. + p_value: options + .iter() + .filter(|o| o.is_key) + .map(|o| o.rate) + .sum::() + .min(1.0), + point_biserial: None, + discrimination: None, + blank_rate: responses.iter().filter(|r| r.chosen().is_empty()).count() as f64 + / responses.len().max(1) as f64, + key: keyed.iter().cloned().collect(), + options, + flags: Vec::new(), + notes: Vec::new(), + prediction_notes: Vec::new(), + calibrated: false, + by_form: BTreeMap::new(), + } + }) + .collect(); + dropped_detail.sort_by_key(|q| q.number); + + let default_options = course.policy.options_per_item; + let triage = triage(&questions, threshold, default_options); + let predictions = prediction_summary(analysis); + let grades = grade_rows(&course.policy, &percents); + let lectures = lecture_rows(catalog, &questions, &objectives, threshold); + + CohortDiagnostic { + n_students: cohort.students.len(), + n_items: analysis.reliability.n_items, + distribution: distribution(&percents), + grades, + lectures, + dropped_questions, + triage, + predictions, + reliability: ReliabilityRow { + alpha: analysis.reliability.alpha, + sem: analysis.reliability.sem, + mean_p: analysis.reliability.mean_p, + mean_point_biserial: analysis.reliability.mean_point_biserial, + interpretation: analysis.reliability.interpretation(), + }, + levels, + objectives, + gaps, + questions, + revise, + dropped_detail, + forms: form_rows(set, cohort), + blueprint: crate::select::check_blueprint(record, course), + patterns: cohort + .archetypes + .iter() + .map(|a| PatternRow { + label: a.label.clone(), + n_students: a.members.len(), + level_means: a + .level_means + .iter() + .map(|(level, mean)| (level.code(), *mean)) + .collect(), + }) + .collect(), + warnings: analysis.warnings.clone(), + } +} + +/// Where an item was taught, as lecture titles with slide numbers. +/// +/// # Arguments +/// +/// * `catalog` - the loaded course. +/// * `uid` - the item's global id. +/// +/// # Returns +/// +/// One entry per source the item records. +fn taught_in(catalog: &Catalog, uid: &str) -> Vec { + let Some(entry) = catalog.get(uid) else { + return Vec::new(); + }; + entry + .item + .sources + .iter() + .map(|source| { + let title = catalog + .course + .lectures + .get(&source.lecture) + .map(|l| l.title.clone()) + .unwrap_or_else(|| source.lecture.clone()); + if source.slides.is_empty() { + format!("{} ({})", title, source.lecture) + } else { + let slides: Vec = source.slides.iter().map(|s| s.to_string()).collect(); + format!( + "{} ({}), slide{} {}", + title, + source.lecture, + if source.slides.len() == 1 { "" } else { "s" }, + slides.join(", ") + ) + } + }) + .collect() +} + +/// Where one bank option landed on each printed form. +/// +/// # Arguments +/// +/// * `forms` - the record's forms, in declaration order. +/// * `item` - the bank item, for its option count and lettering. +/// * `uid` - the item's global id, which salts the permutation. +/// * `letter` - the bank letter to locate. +/// +/// # Returns +/// +/// One entry per form that permutes its options. Forms printing the bank order +/// unchanged are left out, since an entry saying C was printed as C is noise on +/// every row. +fn printed_letters( + forms: &[&crate::assessment::Form], + item: &crate::item::Item, + uid: Option<&str>, + letter: &str, +) -> Vec { + let Some(uid) = uid else { + return Vec::new(); + }; + let Some(source) = item.options.iter().position(|o| o.id == letter) else { + return Vec::new(); + }; + let n = item.options.len(); + let mut out = Vec::new(); + for form in forms { + if !form.shuffle_options { + continue; + } + // The same permutation the exporter and the seal use, so these are the + // letters on the paper rather than a second guess at them. + let order = crate::select::option_order(form, uid, n); + // `order[position] == source` means the option printed in that slot is + // the one being asked about. + if let Some(position) = order.iter().position(|index| *index == source) { + out.push(PrintedLetter { + form: form.id.clone(), + letter: crate::seal::printed_letter(position), + }); + } + } + out +} + +/// The difficulty band a p-value falls in. +/// +/// Three bands rather than five. The only distinction that changes what you do +/// is whether the item had room to discriminate at all, and that is a question +/// about the middle versus the two ends. +fn difficulty_band(p: f64) -> &'static str { + if p >= 0.85 { + "too easy" + } else if p <= 0.35 { + "hard" + } else { + "moderate" + } +} + +/// The discrimination band a point-biserial falls in. +/// +/// The cut points are the conventional ones from the item-analysis literature, +/// usually attributed to Ebel: about 0.40 and above is excellent, 0.30 to 0.39 +/// good, 0.20 to 0.29 marginal, and below 0.20 poor. They are rules of thumb +/// rather than laws, and they must be read next to difficulty, because an item +/// almost everyone passes or fails has little variance left to correlate with +/// anything. +fn discrimination_band(r: Option) -> &'static str { + match r { + None => "no variance", + Some(r) if r < 0.0 => "negative", + Some(r) if r < 0.20 => "poor", + Some(r) if r < 0.30 => "marginal", + Some(r) if r < 0.40 => "good", + Some(_) => "excellent", + } +} + +/// Bins the class by the course's letter-grade scale. +/// +/// # Arguments +/// +/// * `policy` - the course policy, for its scale. +/// * `percents` - one score per student, out of 100. +/// +/// # Returns +/// +/// One row per band, highest first. Empty when the course sets no scale, which +/// is the signal for a report to fall back to ten-point bins. +fn grade_rows(policy: &crate::course::Policy, percents: &[f64]) -> Vec { + let bands = policy.bands(); + if bands.is_empty() || percents.is_empty() { + return Vec::new(); + } + + let n = percents.len() as f64; + let mut out: Vec = Vec::with_capacity(bands.len()); + let mut running = 0usize; + for (index, band) in bands.iter().enumerate() { + // The ceiling is the floor of the band above, less the smallest step a + // percentage is reported at, so the printed range reads the way a + // syllabus writes it. + let high = match index { + 0 => 100.0, + _ => bands[index - 1].min - 0.1, + }; + let count = percents + .iter() + .filter(|percent| { + **percent + 1e-9 >= band.min && (index == 0 || **percent < bands[index - 1].min) + }) + .count(); + running += count; + out.push(GradeRow { + letter: band.letter.clone(), + low: band.min, + high, + gpa: band.gpa, + attainment: band.attainment.clone(), + group: band.group_key(), + count, + share: count as f64 / n, + at_or_above: running, + }); + } + out +} + +/// Aggregates questions into per-lecture rows, worst first. +/// +/// # Arguments +/// +/// * `catalog` - the loaded course, for lecture titles and objective lectures. +/// * `questions` - the per-question rows. +/// * `objectives` - the per-objective rows, for the objective counts. +/// * `threshold` - the mastery threshold. +/// +/// # Returns +/// +/// One row per lecture that any scored item traced back to. +fn lecture_rows( + catalog: &Catalog, + questions: &[CohortQuestionRow], + objectives: &[CohortObjectiveRow], + threshold: f64, +) -> Vec { + let course = &catalog.course; + let mut items: BTreeMap> = BTreeMap::new(); + // Keyed by objective, not by the target an item was tagged with: the rows + // this is matched against are objective rows, so collecting target ids here + // left every lookup empty and every count zero. + let mut lecture_objectives: BTreeMap> = BTreeMap::new(); + + for question in questions { + // The same two routes the student report uses: the item knows which + // lecture it was written from, and the registry knows which lectures + // develop the target. + let mut lectures: BTreeSet = BTreeSet::new(); + if let Some(entry) = question.item.as_deref().and_then(|uid| catalog.get(uid)) { + for source in &entry.item.sources { + lectures.insert(source.lecture.clone()); + } + } + for target in &question.targets { + lectures.extend(course.lectures_for(target).iter().cloned()); + } + for lecture in lectures { + items.entry(lecture.clone()).or_default().push(question); + lecture_objectives.entry(lecture).or_default().extend( + question + .targets + .iter() + .map(|t| course.objective_for(t).to_string()), + ); + } + } + + let mut out: Vec = items + .into_iter() + .map(|(lecture, rows)| { + let rate = rows.iter().map(|r| r.p_value).sum::() / rows.len() as f64; + let ids = lecture_objectives + .get(&lecture) + .cloned() + .unwrap_or_default(); + let mine: Vec<&CohortObjectiveRow> = + objectives.iter().filter(|o| ids.contains(&o.id)).collect(); + CohortLectureRow { + title: course + .lectures + .get(&lecture) + .map(|l| l.title.clone()) + .unwrap_or_else(|| lecture.clone()), + n_items: rows.len(), + n_objectives: ids.len(), + n_objectives_below: mine.iter().filter(|o| o.rate < threshold).count(), + rate, + questions: rows.iter().map(|r| r.number).collect(), + // `objectives` arrives sorted worst first, so the first match is + // the weakest one under this lecture. + worst_objective: mine.first().map(|o| o.text.clone()), + lecture, + } + }) + .collect(); + + out.sort_by(|a, b| { + a.rate + .partial_cmp(&b.rate) + .unwrap_or(std::cmp::Ordering::Equal) + .then_with(|| a.lecture.cmp(&b.lecture)) + }); + out +} + +/// Summarizes how the authored expectations did. +fn prediction_summary(analysis: &Analysis) -> PredictionSummary { + let mut out = PredictionSummary::default(); + let mut signed: Vec = Vec::new(); + let mut biggest: Option<(u32, f64, f64)> = None; + + for item in &analysis.items { + let Some(prediction) = &item.prediction else { + continue; + }; + if prediction.calibrated { + out.n_calibrated += 1; + } + if let (Some(expected), Some(error)) = (prediction.expected_p, prediction.p_error) { + out.n_predicted += 1; + signed.push(error); + if prediction.p_within == Some(true) { + out.n_within += 1; + } + if biggest + .map(|(_, e, o)| (o - e).abs() < error.abs()) + .unwrap_or(true) + { + biggest = Some((item.number, expected, item.p_value)); + } + } + if prediction.expected_band.is_some() { + out.n_band += 1; + if prediction.band_hit == Some(true) { + out.n_band_hit += 1; + } + } + } + + if !signed.is_empty() { + let n = signed.len() as f64; + out.mean_signed_error = Some(signed.iter().sum::() / n); + out.mean_abs_error = Some(signed.iter().map(|e| e.abs()).sum::() / n); + } + out.biggest_surprise = biggest; + out +} + +/// Sorts every question into what to do about it. +/// +/// The order of the tests is the order of severity, and the first match wins for +/// the three item decisions. `reteach` is judged separately, because "the item +/// worked and the class missed it" is not a competing diagnosis; it is a +/// different kind of finding. +/// +/// # Arguments +/// +/// * `questions` - the per-question rows. +/// * `threshold` - the mastery threshold, which sets what counts as a content +/// gap worth reteaching. +/// * `default_options` - the course's default option count, used for the chance +/// rate when an item's own options cannot be counted. +/// +/// # Returns +/// +/// The buckets. +fn triage(questions: &[CohortQuestionRow], threshold: f64, default_options: usize) -> Triage { + let mut out = Triage::default(); + + for question in questions { + let row = |reasons: Vec, option: Option<&OptionRow>| TriageRow { + number: question.number, + item: question.item.clone(), + level: question.level, + p_value: question.p_value, + point_biserial: question.point_biserial, + discrimination: question.discrimination, + targets: question.target_texts.clone(), + taught_in: question.taught_in.clone(), + option: option.map(|o| o.letter.clone()), + option_share: option.map(|o| o.rate), + option_point_biserial: option.and_then(|o| o.point_biserial), + reasons, + }; + + let r = question.point_biserial; + let key_r = question + .options + .iter() + .filter(|o| o.is_key) + .filter_map(|o| o.point_biserial) + .fold(f64::NEG_INFINITY, f64::max); + + // Count single letters only, so a multiple-response combination row such + // as `A+D` is not mistaken for a fifth option and does not deflate the + // chance rate. + let counted = question + .options + .iter() + .filter(|o| o.letter.chars().count() == 1) + .count(); + let n_options = if counted >= 2 { + counted + } else { + default_options.max(2) + }; + let chance = 1.0 / n_options as f64; + + // The best-supported alternative: chosen by a fifth of the class or more, + // and correlating with total score at least as well as the key. The share + // matters because a defensible reading that two students found is a + // wording note, not a regrade. + let challenger = question + .options + .iter() + .filter(|o| !o.is_key && o.rate >= 0.20) + .filter(|o| o.point_biserial.unwrap_or(f64::NEG_INFINITY) > 0.0) + .filter(|o| { + !key_r.is_finite() || o.point_biserial.unwrap_or(f64::NEG_INFINITY) >= key_r + }) + .max_by(|a, b| { + a.point_biserial + .unwrap_or(f64::NEG_INFINITY) + .partial_cmp(&b.point_biserial.unwrap_or(f64::NEG_INFINITY)) + .unwrap_or(std::cmp::Ordering::Equal) + }); + + let mut placed = false; + + // 1. Discard. Negative discrimination means the students who knew the + // material did worse on it, which no amount of rewording fixes after + // the fact; scores already awarded on it are noise. + if let Some(r) = r { + if r < -0.05 { + out.discard.push(row( + vec![format!( + "students who scored well overall did worse on this one (r = {r:+.2}). \ + Whatever it measured, it was not what the rest of the exam measured." + )], + None, + )); + placed = true; + } else if r < 0.05 && question.p_value <= chance + 0.05 { + out.discard.push(row( + vec![format!( + "{:.0}% correct against {:.0}% for guessing, and no relationship to total \ + score (r = {r:+.2}). The responses are indistinguishable from random.", + question.p_value * 100.0, + chance * 100.0 + )], + None, + )); + placed = true; + } + } + + // 2. Rekey or award partial credit. + if !placed { + if let Some(option) = challenger { + let mut reasons = vec![format!( + "option {} drew {:.0}% and tracks total score at least as well as the key \ + ({:+.2} against {:+.2}).", + option.letter, + option.rate * 100.0, + option.point_biserial.unwrap_or(0.0), + if key_r.is_finite() { key_r } else { 0.0 } + )]; + if question.flags.iter().any(|f| f == "key_underperforms") { + reasons.push( + "the strongest students chose it more often than the key, which is the \ + signature of two readings rather than of a guess." + .to_string(), + ); + } + reasons.push( + "Either credit it for this administration or rewrite the stem to exclude it \ + before reuse." + .to_string(), + ); + out.rekey.push(row(reasons, Some(option))); + placed = true; + } + } + + // 3. Revise, unless the weak discrimination is explained by difficulty. + if !placed { + let mut reasons: Vec = Vec::new(); + let weak = r.map(|r| r < 0.20).unwrap_or(true); + let bounded = weak && (question.p_value >= 0.85 || question.p_value <= 0.20); + + if weak && !bounded { + reasons.push(format!( + "at {:.0}% correct the item had room to separate students and did not \ + (r = {}).", + question.p_value * 100.0, + r.map(|r| format!("{r:+.2}")) + .unwrap_or_else(|| "n/a".into()) + )); + } + let dead: Vec<&OptionRow> = question + .options + .iter() + .filter(|o| o.nonfunctioning) + .collect(); + if !dead.is_empty() { + reasons.push(format!( + "option{} {} drew almost nobody, so the item is really a {}-way choice.", + if dead.len() == 1 { "" } else { "s" }, + dead.iter() + .map(|o| o.letter.as_str()) + .collect::>() + .join(", "), + n_options.saturating_sub(dead.len()).max(2) + )); + } + if question.flags.iter().any(|f| f == "ambiguous") { + reasons.push( + "partial credit was awarded at grading time, which is a record that the item \ + admitted more than one reading." + .to_string(), + ); + } + + if bounded { + out.bounded.push(row( + vec![format!( + "{:.0}% correct leaves little variance to correlate with, so r = {} is \ + what this difficulty allows rather than a fault.", + question.p_value * 100.0, + r.map(|r| format!("{r:+.2}")) + .unwrap_or_else(|| "n/a".into()) + )], + None, + )); + placed = true; + } else if !reasons.is_empty() { + out.revise.push(row(reasons, None)); + placed = true; + } + } + + // 4. Reteach: the item did its job and the class still missed it. Judged + // independently of the three above. + let works = r.map(|r| r >= 0.20).unwrap_or(false); + if works && question.p_value < threshold { + out.reteach.push(row( + vec![format!( + "the item separated students cleanly (r = {}) and {:.0}% still missed it, so \ + this is a gap in what the class knows rather than a fault in the question.", + r.map(|r| format!("{r:+.2}")) + .unwrap_or_else(|| "n/a".into()), + (1.0 - question.p_value) * 100.0 + )], + None, + )); + } + + if !placed { + out.clean += 1; + } + } + + // Worst first inside each bucket, so the top of every list is where to start. + for bucket in [ + &mut out.discard, + &mut out.rekey, + &mut out.revise, + &mut out.bounded, + ] { + bucket.sort_by(|a, b| { + a.point_biserial + .unwrap_or(1.0) + .partial_cmp(&b.point_biserial.unwrap_or(1.0)) + .unwrap_or(std::cmp::Ordering::Equal) + .then_with(|| a.number.cmp(&b.number)) + }); + } + out.reteach.sort_by(|a, b| { + a.p_value + .partial_cmp(&b.p_value) + .unwrap_or(std::cmp::Ordering::Equal) + .then_with(|| a.number.cmp(&b.number)) + }); + out +} + +/// Per-question proportion correct, split by form. +fn per_form_p_values(set: &ResponseSet) -> BTreeMap> { + let forms: BTreeSet<&str> = set.rows.iter().filter_map(|r| r.form.as_deref()).collect(); + if forms.len() < 2 { + return BTreeMap::new(); + } + + let mut totals: BTreeMap<(u32, String), (usize, usize)> = BTreeMap::new(); + for row in set.rows.iter().filter(|r| r.counts()) { + let Some(form) = row.form.as_deref() else { + continue; + }; + let entry = totals + .entry((row.item_number, form.to_string())) + .or_insert((0, 0)); + entry.1 += 1; + if row.correct == Some(true) { + entry.0 += 1; + } + } + + let mut out: BTreeMap> = BTreeMap::new(); + for ((number, form), (correct, n)) in totals { + if n == 0 { + continue; + } + out.entry(number) + .or_default() + .insert(form, correct as f64 / n as f64); + } + out +} + +/// One row per form, when more than one was given. +fn form_rows(set: &ResponseSet, cohort: &Cohort) -> Vec { + let mut students_by_form: BTreeMap> = BTreeMap::new(); + for row in &set.rows { + if let Some(form) = row.form.as_deref() { + students_by_form + .entry(form.to_string()) + .or_default() + .insert(row.student_key.as_str()); + } + } + if students_by_form.len() < 2 { + return Vec::new(); + } + + let percent: BTreeMap<&str, f64> = cohort + .students + .iter() + .map(|s| (s.student_key.as_str(), s.percent)) + .collect(); + + students_by_form + .into_iter() + .map(|(form, students)| { + let values: Vec = students + .iter() + .filter_map(|s| percent.get(*s).copied()) + .collect(); + let n = values.len(); + let mean = if n == 0 { + 0.0 + } else { + values.iter().sum::() / n as f64 + }; + let sd = if n < 2 { + 0.0 + } else { + let variance = + values.iter().map(|v| (v - mean).powi(2)).sum::() / (n as f64 - 1.0); + variance.sqrt() + }; + FormRow { + id: form, + n_students: n, + mean, + sd, + } + }) + .collect() +} + +/// Bins a set of percentages into a distribution. +fn distribution(percents: &[f64]) -> Distribution { + let mut sorted: Vec = percents.to_vec(); + sorted.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal)); + + let n = sorted.len(); + let mean = if n == 0 { + 0.0 + } else { + sorted.iter().sum::() / n as f64 + }; + let median = match n { + 0 => 0.0, + _ if n % 2 == 1 => sorted[n / 2], + _ => (sorted[n / 2 - 1] + sorted[n / 2]) / 2.0, + }; + let sd = if n < 2 { + 0.0 + } else { + (sorted.iter().map(|v| (v - mean).powi(2)).sum::() / (n as f64 - 1.0)).sqrt() + }; + + let mut bins: Vec = (0..10) + .map(|i| Bin { + low: i * 10, + high: if i == 9 { 100 } else { i * 10 + 10 }, + count: 0, + }) + .collect(); + for value in &sorted { + let index = ((*value / 10.0).floor() as isize).clamp(0, 9) as usize; + bins[index].count += 1; + } + + Distribution { + mean, + median, + sd, + min: sorted.first().copied().unwrap_or(0.0), + max: sorted.last().copied().unwrap_or(0.0), + bins, + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn the_distribution_bins_a_hundred_into_the_last_bin() { + let d = distribution(&[100.0, 95.0, 0.0, 42.0]); + assert_eq!(d.bins[9].count, 2, "100 belongs with the nineties"); + assert_eq!(d.bins[0].count, 1); + assert_eq!(d.bins[4].count, 1); + assert_eq!(d.max, 100.0); + assert_eq!(d.min, 0.0); + } + + #[test] + fn the_median_averages_the_middle_pair() { + assert_eq!(distribution(&[10.0, 20.0, 30.0, 40.0]).median, 25.0); + assert_eq!(distribution(&[10.0, 20.0, 30.0]).median, 20.0); + } + + #[test] + fn an_empty_class_does_not_panic() { + let d = distribution(&[]); + assert_eq!(d.mean, 0.0); + assert_eq!(d.bins.len(), 10); + } + + #[test] + fn the_student_diagnostic_has_no_field_for_question_content() { + // A compile-time argument as much as a test: the struct has no stem and no + // option text, so no template can print either one. If a field is ever + // added, this serialization check is where the reason gets re-read. + let json = serde_json::to_string(&StudentDiagnostic { + student_key: "s-1".into(), + name: None, + sid: None, + email: None, + form: None, + score: Score { + points: 1.0, + points_possible: 2.0, + percent: 50.0, + bonus_points: 0.0, + correct: 1, + n_items: 2, + }, + standing: None, + levels: Vec::new(), + objectives: Vec::new(), + strengths: Vec::new(), + focus: Vec::new(), + questions: Vec::new(), + dropped_questions: Vec::new(), + review_lectures: Vec::new(), + study: Vec::new(), + }) + .unwrap(); + assert!(!json.contains("stem"), "{json}"); + assert!(!json.contains("options"), "{json}"); + } +} diff --git a/src/analysis/irt.rs b/src/analysis/irt.rs index 5a30bda..872b79d 100644 --- a/src/analysis/irt.rs +++ b/src/analysis/irt.rs @@ -444,7 +444,7 @@ pub fn fit(matrix: &Matrix, opts: &Options) -> Fit { for iteration in 0..opts.max_iterations { iterations = iteration + 1; - // ---- E step: expected counts at each quadrature point ---- + // --- E step: expected counts at each quadrature point ---- // Counts are accumulated per item rather than globally, so an item // administered to only some examinees is not charged for the others. let mut n_kj = vec![vec![0.0f64; j_count]; n_quad]; @@ -473,7 +473,7 @@ pub fn fit(matrix: &Matrix, opts: &Options) -> Fit { } } - // ---- M step: one two-parameter Newton solve per item ---- + // --- M step: one two-parameter Newton solve per item ---- let mut delta = 0.0f64; for j in 0..j_count { let counts: Vec<(f64, f64)> = (0..n_quad).map(|k| (n_kj[k][j], r_k[k][j])).collect(); @@ -529,7 +529,7 @@ pub fn fit(matrix: &Matrix, opts: &Options) -> Fit { )); } - // ---- Standard errors and per-item notes ---- + // --- Standard errors and per-item notes ---- let (grid_final, weight_final) = (grid.clone(), base_weight.clone()); let mut p_grid = vec![vec![0.0f64; j_count]; n_quad]; for k in 0..n_quad { @@ -601,7 +601,7 @@ pub fn fit(matrix: &Matrix, opts: &Options) -> Fit { }); } - // ---- Abilities, expected a posteriori ---- + // --- Abilities, expected a posteriori ---- let mut abilities = Vec::with_capacity(n); let mut log_likelihood = 0.0f64; for i in 0..n { diff --git a/src/analysis/students.rs b/src/analysis/students.rs index dd153e3..1b448ff 100644 --- a/src/analysis/students.rs +++ b/src/analysis/students.rs @@ -24,6 +24,28 @@ //! uses the interval, so it is honest). A student can be "meeting" an objective //! provisionally, and the report says so. //! +//! # Which tier gets classified +//! +//! The registry has two tiers, objectives and their targets (see +//! [`crate::course::Objective`]), and they are reported differently because the +//! evidence behind them differs in kind. An **objective** is classified: its +//! denominator is every item tagged to any of its targets, which is how an exam +//! that spends twelve questions across a topic gets to make one statement with a +//! real denominator instead of twelve statements with none. +//! +//! A **target** is not classified. It usually carries one or two items, and +//! `min_items_for_mastery` would mark almost all of them "not enough evidence", +//! which would be true but useless. So target rows report the observed rate as +//! itemized evidence for the objective's classification, and a report should +//! present them that way: not "you have not mastered this" but "here is what you +//! missed inside the objective above". +//! +//! An item tagged with two targets of the same objective counts *once* toward +//! that objective. Double counting is right across unrelated objectives, where +//! the question "how is this student doing on kinetics" should use every item +//! that measured kinetics, but within one denominator it would inflate both the +//! count and the confidence. +//! //! # Comparison to the cohort //! //! Per-level performance is reported against the class rather than in absolute @@ -33,10 +55,10 @@ use std::collections::{BTreeMap, BTreeSet}; -use crate::course::{CourseFile, Policy}; +use crate::course::CourseFile; use crate::responses::{Response, ResponseSet}; use crate::rng::Rng; -use crate::taxonomy::Level; +use crate::taxonomy::{Level, Tier}; /// How well a student has met one objective. #[derive(Debug, Clone, Copy, PartialEq, Eq)] @@ -74,14 +96,27 @@ impl Mastery { } } -/// One student's standing on one objective. +/// One student's standing on one registry entry, at either tier. #[derive(Debug, Clone)] pub struct ObjectiveMastery { - /// The objective id. - pub objective: String, + /// The registry id this row reports on, at either tier. + pub id: String, /// The objective text, for reports. pub text: String, - /// How many items on this objective the student saw. + /// Which tier this row is, since only one of them is a classification. + pub tier: Tier, + /// The objective this row sits under, for a target row. + pub objective: Option, + /// For an objective row, how many of its targets the exam reached. + /// + /// A student report can say "four of the nine things under this objective + /// were tested", which is the honest scope of the claim. Zero for a target + /// row and for an objective with no targets. + pub targets_seen: usize, + /// For an objective row, how many targets it has in the registry. + pub targets_total: usize, + /// How many items the student saw. For an objective row, items tagged to any + /// of its targets, counted once each. pub n_items: usize, /// How many they got right, counting partial credit. pub credit: f64, @@ -145,8 +180,8 @@ pub struct MissedItem { pub credit: f64, /// The level. pub level: Option, - /// The objectives involved. - pub learning_objectives: Vec, + /// The learning targets the question measured. + pub learning_targets: Vec, /// The misconception the chosen distractor was written to detect. pub misconception: Option, /// Feedback written for a student who chose that option. @@ -208,7 +243,13 @@ impl StudentSummary { pub struct Cohort { /// Per-student summaries, sorted by key. pub students: Vec, - /// Class rate per objective. + /// Class rate per target, which is the tier items are tagged at. Use it to + /// drill into an objective the class missed. + pub target_rates: BTreeMap, + /// Class rate per objective, with each item counted once. + /// + /// This is the class-level table worth acting on, and the one + /// [`Cohort::class_gaps`] is drawn from. pub objective_rates: BTreeMap, /// Class rate per level. pub level_rates: BTreeMap, @@ -218,6 +259,11 @@ pub struct Cohort { pub sd_percent: f64, /// Objectives the class as a whole did not meet, worst first. This is the /// list that should change what you reteach. + /// + /// Objectives rather than targets, because a list of forty targets below + /// threshold is a list nobody reteaches from, and because a target that + /// carried one item on this exam does not support the claim that the class + /// missed it. pub class_gaps: Vec<(String, f64)>, /// Optional grouping of students by response profile. pub archetypes: Vec, @@ -289,7 +335,8 @@ pub fn summarize( let students = set.students(); // Class rates first: every student's report is relative to these. - let objective_rates = rates_by_objective(&set.rows.iter().collect::>()); + let target_rates = rates_by_target(&set.rows.iter().collect::>()); + let objective_rates = rates_by_objective(&set.rows.iter().collect::>(), course); let level_rates = rates_by_level(&set.rows.iter().collect::>()); // Per-level spread across students, for the z comparisons. @@ -345,28 +392,62 @@ pub fn summarize( .count(); let n_items = rows.iter().filter(|r| r.counts()).count(); - // Objectives, in the course's declared order so reports read the way the - // course is taught rather than alphabetically. - let per_objective = rates_by_objective(&rows); - let counts = counts_by_objective(&rows); + // The registry in the course's declared order, so a report reads the way + // the course is taught rather than alphabetically. Each objective the + // exam reached is followed by the targets it reached, which is the order + // a report wants them in: the claim, then its evidence. + let target_counts = counts_by_target(&rows); + let objective_counts = counts_by_objective(&rows, course); let mut objectives = Vec::new(); - let mut seen: BTreeSet<&String> = BTreeSet::new(); - for id in order.iter().chain(per_objective.keys()) { - if !seen.insert(id) { + let mut seen: BTreeSet = BTreeSet::new(); + + // `order` puts each objective ahead of its own targets, so walking it + // produces the tiering. Anything the exam measured that the registry + // does not know about is appended afterwards rather than dropped. + let measured: Vec = target_counts.keys().cloned().collect(); + for id in order.iter().cloned().chain(measured) { + if !seen.insert(id.clone()) { continue; } - let Some((n, credit)) = counts.get(id).copied() else { - continue; - }; - objectives.push(objective_mastery( - id, - course, - n, - credit, - objective_rates.get(id).copied().unwrap_or(0.0), - &rows, - policy, - )); + if course.is_objective(&id) { + let Some((n, credit)) = objective_counts.get(&id).copied() else { + continue; + }; + let targets = course.targets(&id); + let reached = targets + .iter() + .filter(|target| target_counts.contains_key(**target)) + .count(); + objectives.push(objective_mastery( + &id, + course, + n, + credit, + objective_rates.get(&id).copied().unwrap_or(0.0), + &rows, + policy.min_items_for_mastery.max(1), + reached, + targets.len(), + )); + } else { + let Some((n, credit)) = target_counts.get(&id).copied() else { + continue; + }; + // One item is the normal case for a target, so it is reported + // rather than withheld. The objective row above it carries the + // classification. + objectives.push(objective_mastery( + &id, + course, + n, + credit, + target_rates.get(&id).copied().unwrap_or(0.0), + &rows, + 1, + 0, + 0, + )); + } } // Levels. @@ -400,25 +481,29 @@ pub fn summarize( // not: telling a student to review something they may already know costs // them an hour, while telling them they have mastered something they have // not costs them the next exam. + // + // Both lists are drawn from objective rows only. A focus list built from + // targets is as long as the exam and tells a student to review forty + // things, which is the same as telling them nothing; the objective list + // is short enough to act on, and the target rows underneath it say what + // to look at within each one. let strengths: Vec = objectives .iter() - .filter(|o| o.status == Mastery::Meeting && o.confident) - .map(|o| o.objective.clone()) + .filter(|o| o.tier == Tier::Objective && o.status == Mastery::Meeting && o.confident) + .map(|o| o.id.clone()) .collect(); let mut focus_pairs: Vec<(&ObjectiveMastery, f64)> = objectives .iter() + .filter(|o| o.tier == Tier::Objective) .filter(|o| matches!(o.status, Mastery::NotYet | Mastery::Developing)) .map(|o| (o, o.rate)) .collect(); focus_pairs.sort_by(|a, b| { a.1.partial_cmp(&b.1) .unwrap_or(std::cmp::Ordering::Equal) - .then_with(|| a.0.objective.cmp(&b.0.objective)) + .then_with(|| a.0.id.cmp(&b.0.id)) }); - let focus: Vec = focus_pairs - .iter() - .map(|(o, _)| o.objective.clone()) - .collect(); + let focus: Vec = focus_pairs.iter().map(|(o, _)| o.id.clone()).collect(); let missed = missed_items(&rows, catalog, course); @@ -460,6 +545,7 @@ pub fn summarize( Cohort { students: summaries, + target_rates, objective_rates, level_rates, mean_percent, @@ -469,21 +555,26 @@ pub fn summarize( } } -/// Builds one objective's mastery record. +/// Builds one objective's record, at either tier. /// /// # Arguments /// /// * `id` - the objective id. -/// * `course` - the course, for text and policy. -/// * `n` - items on this objective. +/// * `course` - the course, for text, tier, and policy. +/// * `n` - items counting toward this row. /// * `credit` - total credit earned. -/// * `cohort_rate` - the class rate. +/// * `cohort_rate` - the class rate for the same row. /// * `rows` - the student's responses, for the level list. -/// * `policy` - the course policy. +/// * `min_items` - items required before the row is classified. The policy's +/// `min_items_for_mastery` for an objective row, and 1 for a target row, which +/// is evidence rather than a classification. +/// * `targets_seen` - targets this exam reached, for an objective row. +/// * `targets_total` - targets in the registry, for an objective row. /// /// # Returns /// /// The record. +#[allow(clippy::too_many_arguments)] fn objective_mastery( id: &str, course: &CourseFile, @@ -491,12 +582,15 @@ fn objective_mastery( credit: f64, cohort_rate: f64, rows: &[&Response], - policy: &Policy, + min_items: usize, + targets_seen: usize, + targets_total: usize, ) -> ObjectiveMastery { + let policy = &course.policy; let rate = if n > 0 { credit / n as f64 } else { 0.0 }; let (lower, upper) = wilson(credit, n, 1.96); - let status = if n < policy.min_items_for_mastery.max(1) { + let status = if n < min_items.max(1) { Mastery::NotEnoughEvidence } else if rate >= policy.mastery_threshold { Mastery::Meeting @@ -506,17 +600,35 @@ fn objective_mastery( Mastery::NotYet }; + // Levels the row was assessed at. For an objective row this is every level + // any of its targets was assessed at, which is what makes "met this + // objective" a checkable claim: meeting it on three Remember items is a + // different statement from meeting it on three Analyze items. let levels: Vec = rows .iter() - .filter(|r| r.learning_objectives.iter().any(|o| o == id)) + .filter(|r| { + r.learning_targets + .iter() + .any(|t| t == id || course.objective_for(t) == id) + }) .filter_map(|r| r.level) .collect::>() .into_iter() .collect(); ObjectiveMastery { - objective: id.to_string(), - text: course.objective_text(id), + id: id.to_string(), + text: course.text_for(id), + tier: if course.is_objective(id) { + Tier::Objective + } else { + Tier::Target + }, + objective: course + .is_target(id) + .then(|| course.objective_for(id).to_string()), + targets_seen, + targets_total, n_items: n, credit, rate, @@ -563,7 +675,7 @@ fn missed_items( if let Some(entry) = cat.get(uid) { // Feedback for the specific option chosen, which is the whole // point of recording per-distractor misconceptions. - if let Some(letter) = r.selected.first() { + if let Some(letter) = r.chosen().first() { if let Some(choice) = entry.item.option(letter) { misconception = choice.misconception.clone(); feedback = choice.student_text().map(|s| s.to_string()); @@ -592,10 +704,10 @@ fn missed_items( out.push(MissedItem { number: r.item_number, item_ref: r.item_ref.clone(), - selected: r.selected.clone(), + selected: r.chosen().to_vec(), credit: r.credit, level: r.level, - learning_objectives: r.learning_objectives.clone(), + learning_targets: r.learning_targets.clone(), misconception, feedback, study, @@ -612,9 +724,9 @@ fn missed_items( /// /// # Returns /// -/// The rate for each objective mentioned. -pub fn rates_by_objective(rows: &[&Response]) -> BTreeMap { - counts_by_objective(rows) +/// The rate for each target mentioned. +pub fn rates_by_target(rows: &[&Response]) -> BTreeMap { + counts_by_target(rows) .into_iter() .map(|(id, (n, credit))| { let rate = if n > 0 { credit / n as f64 } else { 0.0 }; @@ -623,11 +735,12 @@ pub fn rates_by_objective(rows: &[&Response]) -> BTreeMap { .collect() } -/// Item counts and credit per objective. +/// Item counts and credit per target, as tagged. /// -/// An item tagged with two objectives counts toward both. That double counting is +/// An item tagged with two targets counts toward both. That double counting is /// intentional: the question "how is this student doing on kinetics" should use -/// every item that measured kinetics. +/// every item that measured kinetics. Roll-up to the objective, where the same +/// item must count once, is [`counts_by_objective`]. /// /// # Arguments /// @@ -635,14 +748,14 @@ pub fn rates_by_objective(rows: &[&Response]) -> BTreeMap { /// /// # Returns /// -/// `(item count, total credit)` per objective. -pub fn counts_by_objective(rows: &[&Response]) -> BTreeMap { +/// `(item count, total credit)` per target. +pub fn counts_by_target(rows: &[&Response]) -> BTreeMap { let mut out: BTreeMap = BTreeMap::new(); for r in rows { if !r.counts() { continue; } - for objective in &r.learning_objectives { + for objective in &r.learning_targets { let e = out.entry(objective.clone()).or_insert((0, 0.0)); e.0 += 1; e.1 += r.credit.clamp(0.0, 1.0); @@ -651,6 +764,66 @@ pub fn counts_by_objective(rows: &[&Response]) -> BTreeMap out } +/// Item counts and credit per objective. +/// +/// Each response contributes at most once to any one objective, even when it is +/// tagged with several of that objective's targets. Within a single denominator, +/// counting an item twice would inflate both the rate's weight and the +/// confidence interval's tightness, and the interval is the part of the report +/// that is supposed to stay honest. Across unrelated objectives an item still +/// counts toward each, as it does in [`counts_by_target`]. +/// +/// # Arguments +/// +/// * `rows` - the responses. +/// * `course` - the course, for the objective each tagged target belongs to. +/// +/// # Returns +/// +/// `(item count, total credit)` per objective. +pub fn counts_by_objective( + rows: &[&Response], + course: &CourseFile, +) -> BTreeMap { + let mut out: BTreeMap = BTreeMap::new(); + for r in rows { + if !r.counts() { + continue; + } + let objectives: BTreeSet<&str> = r + .learning_targets + .iter() + .map(|t| course.objective_for(t)) + .collect(); + for id in objectives { + let e = out.entry(id.to_string()).or_insert((0, 0.0)); + e.0 += 1; + e.1 += r.credit.clamp(0.0, 1.0); + } + } + out +} + +/// Credit rate per objective. +/// +/// # Arguments +/// +/// * `rows` - the responses. +/// * `course` - the course, for the objective each tagged target belongs to. +/// +/// # Returns +/// +/// The rate for each objective the responses reached. +pub fn rates_by_objective(rows: &[&Response], course: &CourseFile) -> BTreeMap { + counts_by_objective(rows, course) + .into_iter() + .map(|(id, (n, credit))| { + let rate = if n > 0 { credit / n as f64 } else { 0.0 }; + (id, rate) + }) + .collect() +} + /// Credit rate per level. /// /// # Arguments @@ -1007,21 +1180,113 @@ mod tests { } #[test] - fn objective_counts_credit_every_tagged_item() { + fn target_counts_credit_every_tagged_item() { let rows = [ make("s1", 1, 1.0, &["lo-a", "lo-b"], Some(Level::Remember)), make("s1", 2, 0.0, &["lo-a"], Some(Level::Apply)), ]; let refs: Vec<&Response> = rows.iter().collect(); - let counts = counts_by_objective(&refs); + let counts = counts_by_target(&refs); // lo-a saw both items; lo-b only the first. assert_eq!(counts["lo-a"], (2, 1.0)); assert_eq!(counts["lo-b"], (1, 1.0)); - let rates = rates_by_objective(&refs); + let rates = rates_by_target(&refs); assert_eq!(rates["lo-a"], 0.5); assert_eq!(rates["lo-b"], 1.0); } + /// A course with one objective over three targets, plus an objective with + /// no targets of its own. + fn tiered_course() -> CourseFile { + serde_yaml_ng::from_str( + r#" +course: { code: X, title: Y, term: Z } +policy: { mastery_threshold: 0.75, min_items_for_mastery: 2 } +learning_objectives: + lo-binding: { text: Quantify binding., order: 1 } + lo-standalone: { text: Untiered objective., order: 2 } +learning_targets: + t-kd: { text: Write the expression., objective: lo-binding, order: 1 } + t-plot: { text: Read a plot., objective: lo-binding, order: 2 } + t-window: { text: State the switching window., objective: lo-binding, order: 3 } +"#, + ) + .expect("course parses") + } + + #[test] + fn items_on_targets_roll_up_to_their_objective() { + let course = tiered_course(); + let rows = [ + make("s1", 1, 1.0, &["t-kd"], Some(Level::Remember)), + make("s1", 2, 0.0, &["t-plot"], Some(Level::Apply)), + make("s1", 3, 1.0, &["t-window"], Some(Level::Understand)), + make("s1", 4, 1.0, &["lo-standalone"], Some(Level::Remember)), + ]; + let refs: Vec<&Response> = rows.iter().collect(); + + let objectives = counts_by_objective(&refs, &course); + // Three items, two credited, in one denominator. + assert_eq!(objectives["lo-binding"], (3, 2.0)); + assert_eq!(objectives["lo-standalone"], (1, 1.0)); + // The targets are not themselves objective rows. + assert!(!objectives.contains_key("t-kd")); + + // As-tagged counts are still available for the drill-down. + let tagged = counts_by_target(&refs); + assert_eq!(tagged["t-kd"], (1, 1.0)); + assert_eq!(tagged.len(), 4); + + let rates = rates_by_objective(&refs, &course); + assert!((rates["lo-binding"] - 2.0 / 3.0).abs() < 1e-9); + } + + #[test] + fn one_item_counts_once_toward_its_objective() { + let course = tiered_course(); + // A single question tagged with two targets of the same objective. + let rows = [make( + "s1", + 1, + 0.0, + &["t-kd", "t-plot"], + Some(Level::Understand), + )]; + let refs: Vec<&Response> = rows.iter().collect(); + + let objectives = counts_by_objective(&refs, &course); + assert_eq!( + objectives["lo-binding"], + (1, 0.0), + "one question is one item in the objective's denominator" + ); + // Whereas as-tagged counting credits both targets, as it always has. + let tagged = counts_by_target(&refs); + assert_eq!(tagged["t-kd"], (1, 0.0)); + assert_eq!(tagged["t-plot"], (1, 0.0)); + } + + #[test] + fn an_untiered_course_rolls_up_to_itself() { + // Adopting the second tier is optional: with no targets declared, every + // entry is an objective and the rolled-up counts equal the tagged ones. + let course: CourseFile = serde_yaml_ng::from_str( + r#" +course: { code: X, title: Y, term: Z } +learning_objectives: + lo-a: { text: A } + lo-b: { text: B } +"#, + ) + .expect("course parses"); + let rows = [ + make("s1", 1, 1.0, &["lo-a", "lo-b"], Some(Level::Remember)), + make("s1", 2, 0.0, &["lo-a"], Some(Level::Apply)), + ]; + let refs: Vec<&Response> = rows.iter().collect(); + assert_eq!(counts_by_objective(&refs, &course), counts_by_target(&refs)); + } + #[test] fn level_rates_ignore_untagged_items() { let rows = [ @@ -1089,6 +1354,7 @@ mod tests { assessment_id: "a".into(), date: None, form: None, + form_position: None, student_key: student.into(), sid: None, name: None, @@ -1097,18 +1363,22 @@ mod tests { item_number: number, item_ref: None, item_version: None, + variant: None, selected: vec!["A".into()], + selected_source: vec![], eliminated: vec![], + eliminated_source: vec![], correct: Some(credit >= 0.999), credit, points_possible: 1.0, score: credit, response_time_seconds: None, level, - learning_objectives: objectives.iter().map(|s| s.to_string()).collect(), + learning_targets: objectives.iter().map(|s| s.to_string()).collect(), topics: vec![], bonus: false, dropped: false, + dropped_full_credit: false, } } diff --git a/src/assets/site/questions.css b/src/assets/site/questions.css new file mode 100644 index 0000000..77fd138 --- /dev/null +++ b/src/assets/site/questions.css @@ -0,0 +1,249 @@ +/* ============================================================================ + questions.css — worksheet items + gated solutions + Sits beside editor-notes.css: same left-border-accent language, same + small-caps auto-headers, same .quarto-dark dark-mode hook. + + Worksheet content is authored as Quarto MARKDOWN inside fenced divs, so + inline math is $ … $ and paragraphs/lists render normally. Choice letters + (A, B, C, D) are drawn by a CSS counter, so a choice is just a list item. + + Semantic color coding (information, not decoration): + indigo = a solution / answer region + green = the correct choice / accepted answer + red = a named misconception + ==========================================================================*/ + +:root { + --q-ink: #1f2328; + --q-muted: #5a5a5a; + --q-line: #e4e4e4; + --q-card-bg: #ffffff; + + --sol-accent: #4b4f9a; /* indigo: "here be answers" */ + --sol-bg: #f5f5fb; + --sol-rule: #dedcef; + + --ok-ink: #2f6249; /* correct / accepted (matches editorial green) */ + --ok-bg: #e7f2ec; + --mis-ink: #a13d2e; /* misconception (matches editorial red) */ +} + +/* ---- Question card ---- */ +.q { + border: 1px solid var(--q-line); + border-radius: 6px; + background: var(--q-card-bg); + padding: 1.1rem 1.3rem 1.25rem; + margin: 1.6rem 0; +} + +/* Header. Authored as [Question 1]{.q-num} [..]{.q-kind} [..]{.q-points}; */ +.q-head { + display: flex; + align-items: baseline; + gap: 0.75rem; + margin-bottom: 0.7rem; + padding-bottom: 0.55rem; + border-bottom: 1px solid var(--q-line); +} +.q-head > p { display: contents; margin: 0; } +.q-num { font-weight: 700; letter-spacing: 0.01em; } +.q-kind { + font-variant: small-caps; letter-spacing: 0.06em; + font-size: 0.78rem; color: var(--q-muted); +} +.q-points { + margin-left: auto; font-size: 0.8rem; color: var(--q-muted); + white-space: nowrap; +} + +.q-stem { margin: 0 0 0.9rem; } +.q-stem > p:first-child { margin-top: 0; } +.q-stem > p:last-child { margin-bottom: 0; } + +/* ---- Multiple choice (a plain ordered list; letters via counter) ---- */ +.q-choices > ol { + list-style: none; margin: 0; padding: 0; + display: grid; gap: 0.5rem; + counter-reset: choice; +} +.q-choices > ol > li { + position: relative; + padding: 0.55rem 0.7rem 0.55rem 3rem; /* room for the badge on the left */ + border: 1px solid var(--q-line); + border-radius: 5px; + counter-increment: choice; +} +.q-choices > ol > li::before { + content: counter(choice, upper-alpha); /* A, B, C, D … */ + position: absolute; + left: 0.55rem; top: 0.5rem; + display: grid; place-items: center; + width: 1.9rem; height: 1.9rem; + border: 1px solid var(--sol-accent); + border-radius: 50%; + font-weight: 700; font-size: 0.9rem; + color: var(--sol-accent); +} + +/* ---- Free-response writing space (prints with room to write) ----- */ +.q-response { + min-height: 6.5rem; + border: 1px dashed var(--q-line); + border-radius: 5px; + background: + repeating-linear-gradient( + to bottom, transparent, transparent 1.55rem, + var(--q-line) 1.55rem, var(--q-line) calc(1.55rem + 1px)); + background-position: 0 0.9rem; +} +.q-response::before { + content: "Your answer"; + display: block; + font-variant: small-caps; letter-spacing: 0.06em; + font-size: 0.72rem; color: var(--q-muted); + padding: 0.3rem 0.6rem 0; +} + +/* ---- Solution slot (filled by solutions.js on unlock) ---- */ +.qsol { + border-left: 3px solid var(--sol-accent); + border-radius: 0 4px 4px 0; + background: var(--sol-bg); + padding: 0.9rem 1.15rem; + margin-top: 1rem; + font-size: 0.95rem; line-height: 1.55; +} +.qsol::before { + content: "Solution"; + display: block; + font-variant: small-caps; letter-spacing: 0.06em; font-weight: 600; + color: var(--sol-accent); + padding-bottom: 0.35rem; margin-bottom: 0.6rem; + border-bottom: 1px solid var(--sol-rule); +} +.qsol > p:first-of-type { margin-top: 0; } +.qsol > *:last-child { margin-bottom: 0; } + +@keyframes sol-in { from { opacity: 0; transform: translateY(2px); } to { opacity: 1; } } +.qsol.is-unlocked { animation: sol-in 180ms ease-out; } +@media (prefers-reduced-motion: reduce) { .qsol.is-unlocked { animation: none; } } + +/* pieces inside a solution */ +.sol-answer { font-size: 1.02rem; } +.sol-model { + border-left: 2px solid var(--ok-ink); + background: var(--ok-bg); + padding: 0.55rem 0.8rem; border-radius: 0 4px 4px 0; margin: 0.7rem 0; +} +.sol-model > p:first-child { margin-top: 0; } +.sol-model > p:last-child { margin-bottom: 0; } +.sol-explain { margin: 0.7rem 0; } +.sol-feedback-title { + font-variant: small-caps; letter-spacing: 0.05em; font-weight: 600; + color: var(--q-muted); margin: 0.9rem 0 0.4rem; +} + +/* per-distractor feedback: badge in col 1, BOTH text spans in col 2 ---- */ +.sol-feedback { list-style: none; margin: 0; padding: 0; display: grid; gap: 0.6rem; } +.sol-feedback > li { + display: grid; + grid-template-columns: 1.9rem 1fr; /* badge | body */ + gap: 0.6rem; + align-items: start; +} +.sol-feedback .opt { + display: grid; place-items: center; width: 1.9rem; height: 1.9rem; + border-radius: 50%; font-weight: 700; font-size: 0.8rem; + color: var(--mis-ink); border: 1px solid var(--mis-ink); +} +.sol-feedback .opt-body { /* the single col-2 cell; fills 1fr */ + display: grid; gap: 0.25rem; +} +.sol-feedback .opt-mis { color: var(--mis-ink); font-style: italic; } +.sol-feedback .opt-why { color: var(--q-ink); } + +/* rubric table */ +.sol-rubric { width: 100%; border-collapse: collapse; margin: 0.7rem 0; font-size: 0.92rem; } +.sol-rubric caption { + text-align: left; font-variant: small-caps; letter-spacing: 0.05em; + font-weight: 600; color: var(--q-muted); padding-bottom: 0.35rem; +} +.sol-rubric th, .sol-rubric td { + border-top: 1px solid var(--sol-rule); padding: 0.4rem 0.55rem; text-align: left; + vertical-align: top; +} +.sol-rubric th:first-child, .sol-rubric td:first-child { + width: 2.5rem; text-align: center; font-weight: 700; color: var(--ok-ink); +} + +.sol-accepted { margin: 0.7rem 0; } +.sol-ref { margin-top: 0.7rem; font-size: 0.85rem; color: var(--q-muted); } + +/* badges */ +.badge { + display: inline-block; font-variant: small-caps; letter-spacing: 0.05em; + font-size: 0.72rem; font-weight: 700; padding: 0.05em 0.5em; + border-radius: 999px; margin-right: 0.35em; + background: var(--sol-rule); color: var(--sol-accent); +} +.badge-correct { background: var(--ok-bg); color: var(--ok-ink); } + +/* ---- The password gate --- */ +.solutions-gate { margin: 1.4rem 0; } +.gate-inner { + display: flex; flex-wrap: wrap; align-items: center; gap: 0.6rem; + padding: 0.75rem 1rem; + border: 1px solid var(--sol-rule); border-left: 3px solid var(--sol-accent); + border-radius: 0 5px 5px 0; background: var(--sol-bg); +} +.gate-lock { + width: 0.8rem; height: 0.7rem; border: 2px solid var(--sol-accent); + border-radius: 2px; position: relative; flex: none; +} +.gate-lock::before { + content: ""; position: absolute; left: 50%; top: -0.42rem; transform: translateX(-50%); + width: 0.5rem; height: 0.42rem; border: 2px solid var(--sol-accent); + border-bottom: none; border-radius: 4px 4px 0 0; +} +.gate-lock.is-open::before { left: 20%; } +.gate-label { font-variant: small-caps; letter-spacing: 0.05em; font-weight: 600; color: var(--sol-accent); } +.gate-input { + flex: 1 1 12rem; min-width: 9rem; + padding: 0.4rem 0.6rem; border: 1px solid var(--sol-rule); + border-radius: 4px; background: var(--q-card-bg); color: var(--q-ink); +} +.gate-btn { + padding: 0.42rem 1rem; border: 1px solid var(--sol-accent); border-radius: 4px; + background: var(--sol-accent); color: #fff; font-weight: 600; cursor: pointer; +} +.gate-btn:hover { filter: brightness(1.08); } +.gate-btn:disabled { opacity: 0.6; cursor: progress; } +.gate-btn-ghost { background: transparent; color: var(--sol-accent); } +.gate-status { flex-basis: 100%; margin: 0; font-size: 0.85rem; color: var(--q-muted); } +.solutions-gate.is-error .gate-input { border-color: var(--mis-ink); } +.solutions-gate.is-error .gate-status { color: var(--mis-ink); } +.gate-input:focus-visible, .gate-btn:focus-visible { outline: 2px solid var(--sol-accent); outline-offset: 2px; } + +@media print { + .solutions-gate { display: none; } + .qsol[hidden] { display: none; } + .q { break-inside: avoid; border-color: #bbb; } +} + +/* ============================ Dark mode ================================= */ +.quarto-dark { + --q-ink: #dfe2e7; + --q-muted: #a7adb6; + --q-line: #333a44; + --q-card-bg: #1b1f26; + + --sol-accent: #9aa0e6; + --sol-bg: #20222e; + --sol-rule: #343755; + + --ok-ink: #9dd3b4; + --ok-bg: #1b241f; + --mis-ink: #e0a498; +} +.quarto-dark .gate-btn { color: #14161c; } diff --git a/src/assets/site/solutions.js b/src/assets/site/solutions.js new file mode 100644 index 0000000..40c5cf4 --- /dev/null +++ b/src/assets/site/solutions.js @@ -0,0 +1,160 @@ +/* ============================================================================ + * solutions.js — per-page, password-gated solutions for the course site. + * + * How a page opts in: + * 1. Put one gate element somewhere on the page: + *
+ * `data-bundle` is resolved relative to the page URL, so each page points + * at its own bundle and therefore has its own password. Nothing else is + * shared between pages. + * 2. For every gated item, leave an empty, hidden slot where the solution + * should appear: + * + * The id in data-solution-for must match a key in the bundle's `items`. + * + * What happens on unlock: + * The typed password is run through PBKDF2 (same params as the bundle's kdf + * block) to derive an AES-256-GCM key. The first item is decrypted as a + * probe: because GCM authenticates, a wrong password throws, and we report + * "wrong password" without revealing anything. On success every slot is + * filled, un-hidden, and MathJax re-typesets the injected math. + * + * The ciphertext ships in the page, so this stops a student + * from reading answers in "View source" or the Network tab, and a wrong + * password reveals nothing. Anyone who has the password can decrypt, and the + * bundle can be brute-forced offline against a weak password. Use a real, + * non-guessable per-page password, and rotate it if a key deadline has passed. + * ==========================================================================*/ + +(() => { + "use strict"; + + const b64ToBytes = (s) => + Uint8Array.from(atob(s), (c) => c.charCodeAt(0)); + + async function deriveKey(password, kdf) { + const base = await crypto.subtle.importKey( + "raw", new TextEncoder().encode(password), "PBKDF2", false, ["deriveKey"]); + return crypto.subtle.deriveKey( + { name: "PBKDF2", salt: b64ToBytes(kdf.salt), + iterations: kdf.iterations, hash: kdf.hash }, + base, { name: "AES-GCM", length: 256 }, false, ["decrypt"]); + } + + async function decryptItem(key, item) { + const pt = await crypto.subtle.decrypt( + { name: "AES-GCM", iv: b64ToBytes(item.iv) }, key, b64ToBytes(item.ct)); + return new TextDecoder().decode(pt); + } + + function typeset(nodes) { + if (window.MathJax && typeof window.MathJax.typesetPromise === "function") { + window.MathJax.typesetPromise(nodes).catch(() => { /* leave as-is */ }); + } + } + + function buildGate(gate) { + gate.classList.add("is-locked"); + gate.innerHTML = ` +
+ + + + +

+
`; + return { + input: gate.querySelector(".gate-input"), + button: gate.querySelector(".gate-btn"), + status: gate.querySelector(".gate-status"), + }; + } + + async function unlock(gate, ui) { + const url = gate.getAttribute("data-bundle"); + const pw = ui.input.value; + if (!pw) { ui.input.focus(); return; } + + gate.classList.remove("is-error"); + ui.button.disabled = true; + ui.status.textContent = "Checking\u2026"; + + let bundle; + try { + const res = await fetch(url, { cache: "no-store" }); + if (!res.ok) throw new Error(`bundle ${res.status}`); + bundle = await res.json(); + } catch (e) { + ui.button.disabled = false; + ui.status.textContent = "Could not load the solutions file for this page."; + return; + } + + let key; + try { + key = await deriveKey(pw, bundle.kdf); + // Probe with the first item so a wrong password fails before we touch DOM. + const firstId = Object.keys(bundle.items)[0]; + await decryptItem(key, bundle.items[firstId]); + } catch (e) { + gate.classList.add("is-error"); + ui.button.disabled = false; + ui.status.textContent = "That password didn\u2019t work. Try again."; + ui.input.select(); + return; + } + + const filled = []; + for (const slot of document.querySelectorAll("[data-solution-for]")) { + const id = slot.getAttribute("data-solution-for"); + const item = bundle.items[id]; + if (!item) continue; + try { + slot.innerHTML = await decryptItem(key, item); + slot.hidden = false; + slot.classList.add("is-unlocked"); + filled.push(slot); + } catch (e) { /* skip an item that fails; others still unlock */ } + } + typeset(filled); + + gate.classList.remove("is-locked"); + gate.classList.add("is-unlocked"); + gate.innerHTML = ` +
+ + Solutions unlocked + +
`; + gate.querySelector(".gate-btn").addEventListener("click", () => { + for (const s of filled) { s.hidden = true; s.classList.remove("is-unlocked"); } + buildAndWire(gate); // relock the UI; content stays in memory only + }); + } + + function buildAndWire(gate) { + const ui = buildGate(gate); + const go = () => unlock(gate, ui); + ui.button.addEventListener("click", go); + ui.input.addEventListener("keydown", (e) => { if (e.key === "Enter") go(); }); + } + + function init() { + const gate = document.querySelector(".solutions-gate[data-bundle]"); + if (!gate) return; + if (!window.crypto || !crypto.subtle) { + gate.textContent = + "This browser can\u2019t decrypt solutions (no Web Crypto over http/file)."; + return; + } + buildAndWire(gate); + } + + if (document.readyState === "loading") { + document.addEventListener("DOMContentLoaded", init); + } else { + init(); + } +})(); \ No newline at end of file diff --git a/src/authoring/jsonschema.rs b/src/authoring/jsonschema.rs index 329ce36..cd7f360 100644 --- a/src/authoring/jsonschema.rs +++ b/src/authoring/jsonschema.rs @@ -31,6 +31,16 @@ const BASE: &str = "https://coursebank.dev/schema"; pub enum Kind { /// `course.yaml`. Course, + /// `references.yaml`. + References, + /// `lectures/*.yaml`. + Lecture, + /// `objectives/*.yaml`. + Objective, + /// `analysis/calibration.yaml`. + Calibration, + /// `analysis/administrations/*.yaml`. + Measurements, /// `banks/*.yaml`. Bank, /// `assessments/*.yaml`. @@ -38,13 +48,27 @@ pub enum Kind { } impl Kind { - /// All three kinds. - pub const ALL: [Kind; 3] = [Kind::Course, Kind::Bank, Kind::Assessment]; + /// Every kind. + pub const ALL: [Kind; 8] = [ + Kind::Course, + Kind::References, + Kind::Lecture, + Kind::Objective, + Kind::Bank, + Kind::Assessment, + Kind::Calibration, + Kind::Measurements, + ]; /// The file name a schema is written to. pub fn filename(self) -> &'static str { match self { Kind::Course => "course.schema.json", + Kind::References => "references.schema.json", + Kind::Lecture => "lecture.schema.json", + Kind::Objective => "objective.schema.json", + Kind::Calibration => "calibration.schema.json", + Kind::Measurements => "measurements.schema.json", Kind::Bank => "bank.schema.json", Kind::Assessment => "assessment.schema.json", } @@ -80,12 +104,17 @@ impl Kind { pub fn schema(kind: Kind) -> Value { match kind { Kind::Course => course_schema(), + Kind::References => references_schema(), + Kind::Lecture => lecture_fragment_schema(), + Kind::Objective => objective_fragment_schema(), + Kind::Calibration => calibration_file_schema(), + Kind::Measurements => measurements_file_schema(), Kind::Bank => bank_schema(), Kind::Assessment => assessment_schema(), } } -/// Writes all three schemas to a directory. +/// Writes every schema to a directory. /// /// # Arguments /// @@ -261,6 +290,43 @@ fn policy_schema() -> Value { "minimum": 1, "description": "Below this many items on an objective, reports say 'not enough \ evidence' rather than classifying." + }, + "grade_scale": { + "type": "array", + "description": "Letter-grade bands. Only the lower bound of each is recorded; a \ + band runs up to the next one. Set this and a class report bins \ + scores by letter rather than by ten-point interval.", + "items": { + "type": "object", + "required": ["letter", "min"], + "additionalProperties": false, + "properties": { + "letter": { + "type": "string", + "description": "The letter as it appears on a transcript." + }, + "min": { + "type": "number", + "minimum": 0, + "maximum": 100, + "description": "Lowest percentage earning this letter, inclusive." + }, + "gpa": { + "type": "number", + "minimum": 0, + "description": "Grade points the band carries." + }, + "attainment": { + "type": "string", + "description": "The attainment word attached to the band." + }, + "group": { + "type": "string", + "description": "Colour group for reports; defaults to the letter's \ + first character." + } + } + } } } }) @@ -291,7 +357,57 @@ fn lecture_schema() -> Value { "date": date("Date delivered."), "unit": { "type": "string", "description": "Unit id." }, "slides_url": { "type": "string" }, - "readings": string_array("Readings assigned with this lecture.") + "teaches": { + "type": "array", + "description": "The objectives this session develops. Each named objective gains \ + this lecture in its `lectures` list when the course is loaded, \ + so the pair is declared once, here, while planning the lecture.", + "items": { "type": "string" } + }, + "readings": { + "type": "array", + "description": "Readings assigned with this lecture, in the order you assign \ + them. A plain string is the pre-schema form and still loads.", + "items": reading_schema() + } + } + }) +} + +/// The schema for one learning target. +fn target_schema() -> Value { + json!({ + "type": "object", + "required": ["text", "objective"], + "additionalProperties": false, + "properties": { + "text": text("The target as a student would read it. Start with a verb."), + "objective": { + "type": "string", + "description": "The objective this target belongs to. Required: a target with \ + no objective would be measured and never reported." + }, + "lectures": string_array("Lecture ids that cover this."), + "order": { + "type": "integer", + "minimum": 1, + "description": "Ignored since 2.0. An objective's targets are a set of question \ + templates, not steps in a sequence — they are not taught in \ + order and an exam samples from them — so a position asserts an \ + order that does not exist. Where one target depends on another, \ + say so with `prerequisites`. `coursebank migrate order` removes \ + this." + }, + "level_ceiling": level(), + "prerequisites": string_array( + "Target or objective ids that must come first. Cycles are rejected." + ), + "tags": string_array("Free-form tags."), + "assessed": { + "type": "boolean", + "description": "Set false for something you teach but do not test. An \ + unassessed objective exempts its targets regardless." + } } }) } @@ -306,6 +422,14 @@ fn objective_schema() -> Value { "text": text("The objective as a student would read it. Start with a verb."), "unit": { "type": "string" }, "lectures": string_array("Lecture ids that cover this."), + "order": { + "type": "integer", + "minimum": 1, + "description": "Position in teaching order, low first. Derived since 2.0 from \ + the position of this objective in a lecture's `teaches` list; \ + an authored value still wins, and `coursebank migrate order` \ + removes them." + }, "level_ceiling": level(), "prerequisites": string_array( "Objective ids that must come first. Cycles are rejected." @@ -314,12 +438,119 @@ fn objective_schema() -> Value { "assessed": { "type": "boolean", "description": "Set false for an objective you teach but do not test; coverage \ - reporting will stop flagging it as a gap." + reporting will stop flagging it as a gap. This exempts its \ + targets too." } } }) } +/// The schema for one cited work. +fn reference_schema() -> Value { + json!({ + "type": "object", + "required": ["title"], + "additionalProperties": false, + "properties": { + "label": text("Short form a reading list shows, such as KKW. One work per label."), + "kind": { + "type": "string", + "enum": strings(&[ + "book", "chapter", "article", "preprint", "thesis", + "website", "software", "dataset", "video", "other" + ]), + "description": "Kind of work, following BibTeX entry types." + }, + "role": { + "type": "string", + "enum": strings(&["required", "supplemental"]), + "description": "required for a course text; supplemental for background." + }, + "title": text("Full title."), + "authors": string_array("Authors as `Family, Given`, in printed order."), + "year": { "type": "integer", "description": "Year of publication." }, + "edition": { "type": "string", "description": "Edition as printed: 7th." }, + "publisher": { "type": "string" }, + "container": { "type": "string", "description": "Journal, edited volume, or series." }, + "volume": { "type": "string" }, + "issue": { "type": "string" }, + "pages": { "type": "string", "description": "Pages of the work, not of a reading." }, + "doi": { + "type": "string", + "pattern": "^(doi:|https?://(dx\\.)?doi\\.org/)?10\\.", + "description": "Bare DOI: 10.1038/nature12373. For a manuscript this is \ + usually the only link worth storing, since a reading list \ + resolves it to doi.org." + }, + "arxiv": { "type": "string", "description": "Bare arXiv id: 2301.00001." }, + "pmcid": { + "type": "string", + "description": "PubMed Central id, which hosts the full text: PMC3084216." + }, + "pmid": { + "type": "string", + "pattern": "^[0-9]+$", + "description": "PubMed id, which hosts a record about the work: 21471563." + }, + "isbn": { "type": "string" }, + "url": { "type": "string", "description": "Canonical URL for the whole work." }, + "base_url": { + "type": "string", + "description": "Prefix a reading's `path` is joined to, so the citation key \ + appears once instead of once per reading." + }, + "note": { "type": "string", "description": "Access notes: reserve shelf, license." } + } + }) +} + +/// The schema for one reading: a location inside a reference, and what it is for. +fn reading_schema() -> Value { + json!({ + "oneOf": [ + { "type": "string", "description": "The pre-schema form: a citation, unparsed." }, + reading_mapping_schema() + ] + }) +} + +/// The mapping form of a reading. +fn reading_mapping_schema() -> Value { + json!({ + "type": "object", + "required": ["ref"], + "additionalProperties": false, + "properties": { + "ref": text("Citation key into `references`."), + "locator": text("Where inside the work: §6.1, pp. 212-219, ch. 3."), + "path": { + "type": "string", + "description": "Joined to the reference's base_url to reach this location." + }, + "url": { + "type": "string", + "description": "Full URL, when base_url does not cover the location." + }, + "role": { + "type": "string", + "enum": strings(&["assigned", "supplemental"]), + "description": "supplemental means offered but not separately assessed." + }, + "targets": string_array( + "Target ids this reading serves. A student who misses one of these is pointed \ + here, so the list is what makes study guidance specific. Cite targets rather \ + than objectives: a section of a book backs a specific performance, and a \ + reading list resolved from an objective would send a student the same six \ + sections whichever part of it they missed." + ), + "summary": text("What the section contains."), + "focus": text("What to take from it. This is the sentence a student report quotes."), + "skip": text("What to gloss, and why it is out of scope."), + "text": text("A pre-schema citation string, held unparsed.") + } + }) +} + /// The schema for one shared stimulus. fn stimulus_schema() -> Value { json!({ @@ -365,10 +596,114 @@ fn course_schema() -> Value { }, "learning_objectives": { "type": "object", - "description": "Objectives by id. Everything downstream — coverage, mastery, \ - student reports — keys off these.", + "description": "Learning objectives by id, conventionally `lo-...`. The tier a \ + syllabus lists and a report classifies as met. Objectives only: \ + the performances they are met by go in `learning_targets`.", "additionalProperties": objective_schema() }, + "learning_targets": { + "type": "object", + "description": "Learning targets by id. Each names the objective it belongs to. \ + This is the tier items are tagged to and readings are cited \ + against; results roll up to the objective. Ids must not start \ + with `lo`, so that an id in an item or a report says which tier \ + it belongs to without a lookup — `t-...` is the convention.", + "additionalProperties": target_schema() + }, + "stimuli": { + "type": "object", + "description": "Shared passages, figures, or data that several items refer to.", + "additionalProperties": stimulus_schema() + }, + "references": { + "type": "object", + "description": "Works the course cites, by citation key. Readings point in \ + here, so an edition change is one edit.", + "additionalProperties": reference_schema() + } + } + }) +} + +/// The schema for `references.yaml`. +fn references_schema() -> Value { + json!({ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": format!("{BASE}/references.schema.json"), + "title": "coursebank references file", + "description": "The works the course cites, by citation key. One fragment of the \ + course file; see course.schema.json for the whole.", + "type": "object", + "additionalProperties": false, + "properties": { + "schema_version": { + "type": ["string", "number"], + "description": format!("Format version; currently {SCHEMA_VERSION}. Declared in \ + course.yaml; fragments inherit it.") + }, + "references": { + "type": "object", + "description": "Works by citation key.", + "additionalProperties": reference_schema() + } + } + }) +} + +/// The schema for one file under `lectures/`. +fn lecture_fragment_schema() -> Value { + json!({ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": format!("{BASE}/lecture.schema.json"), + "title": "coursebank lecture file", + "description": "One session: its readings, and the objectives it develops. A fragment \ + of the course file, merged on load.", + "type": "object", + "required": ["lectures"], + "additionalProperties": false, + "properties": { + "schema_version": { + "type": ["string", "number"], + "description": "Declared in course.yaml; fragments inherit it." + }, + "lectures": { + "type": "object", + "description": "Keyed by lecture id, conventionally one entry per file.", + "additionalProperties": lecture_schema() + } + } + }) +} + +/// The schema for one file under `objectives/`. +fn objective_fragment_schema() -> Value { + json!({ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": format!("{BASE}/objective.schema.json"), + "title": "coursebank objective file", + "description": "One learning objective and the learning targets it decomposes into. A \ + fragment of the course file, merged on load.", + "type": "object", + "required": ["learning_objectives"], + "additionalProperties": false, + "properties": { + "schema_version": { + "type": ["string", "number"], + "description": "Declared in course.yaml; fragments inherit it." + }, + "learning_objectives": { + "type": "object", + "description": "Keyed by objective id, conventionally one entry per file. Its \ + `lectures` list is derived from each lecture's `teaches`, so \ + leave it out unless you prefer to declare it here.", + "additionalProperties": objective_schema() + }, + "learning_targets": { + "type": "object", + "description": "The targets of this file's objective, by id. A target with no \ + `lectures` of its own inherits its objective's.", + "additionalProperties": target_schema() + }, "stimuli": { "type": "object", "description": "Shared passages, figures, or data that several items refer to.", @@ -393,9 +728,11 @@ fn option_schema() -> Value { "properties": { "id": { "type": "string", - "pattern": "^[A-H]$", - "description": "Option letter. Identity, not print position — shuffled forms \ - relabel on the way out." + "pattern": "^(o-[a-z0-9]+(-[a-z0-9]+)*|[A-H])$", + "description": "Option id, unique within the item: `o-fourth-line`. An \ + identity, not a print position — shuffled forms relabel on the \ + way out. A single letter A-H is the pre-2.0 form; \ + `coursebank migrate options` renames it." }, "text": text("The option as a student reads it."), "correct": { "type": "boolean" }, @@ -427,7 +764,8 @@ fn option_schema() -> Value { }, "selection_rate_expected": proportion( "How often you expect this to be chosen. Compared against reality." - ) + ), + "retired": retirement_schema() } }) } @@ -466,6 +804,77 @@ fn asset_schema() -> Value { }) } +/// The worked solution, and for an open-response item how it is graded. +fn solution_schema() -> Value { + json!({ + "type": "object", + "additionalProperties": false, + "description": "The answer, the reasoning, and the rubric. Rendered in the solutions \ + document and the answer key, never in a question paper.", + "properties": { + "model_answer": { + "type": "string", + "description": "For an open-response item, the response a full-credit student \ + writes; for a choice item, an optional one-line statement of the key." + }, + "explanation": { + "type": "string", + "description": "The worked reasoning a student learns from. The body of the \ + solutions entry." + }, + "rubric": { "type": "array", "items": rubric_criterion_schema() }, + "accepted": { + "type": "array", + "items": { "type": "string" }, + "description": "Responses a short constructed answer is accepted as." + }, + "review": { + "type": "array", + "items": citation_schema(), + "description": "Where to look again after missing this item." + } + } + }) +} + +/// One rubric line for an open-response item. +fn rubric_criterion_schema() -> Value { + json!({ + "type": "object", + "required": ["description"], + "additionalProperties": false, + "properties": { + "description": text("What earns the points on this line."), + "points": { "type": "number", "minimum": 0.0 } + } + }) +} + +/// A citation into the reference registry, written as an object or a bare string. +fn citation_schema() -> Value { + json!({ + "oneOf": [ + { "type": "string", "description": "A citation, unparsed." }, + citation_mapping_schema() + ] + }) +} + +/// The object form of a citation. +fn citation_mapping_schema() -> Value { + json!({ + "type": "object", + "additionalProperties": false, + "properties": { + "ref": text("Citation key into `references`."), + "locator": { "type": "string", "description": "Where inside the work: §6.1, pp. 4-9." }, + "path": { "type": "string", "description": "Joined to the reference base_url." }, + "url": { "type": "string" }, + "text": { "type": "string" } + } + }) +} + /// The schema for authored design intent. fn design_schema() -> Value { json!({ @@ -551,6 +960,21 @@ fn calibration_schema() -> Value { "additionalProperties": option_stat_schema() }, "irt": irt_schema(), + "variants": { + "type": "array", + "description": "One record per option set ever administered. A stem shown with \ + different distractors is a different item, so a p-value pooled \ + across both would average two questions.", + "items": variant_calibration_schema() + }, + "options": { + "type": "object", + "description": "One record per option, pooled across every set it appeared in. \ + Supports one claim — this option draws nobody, anywhere — which \ + is what retires a distractor and what one administration cannot \ + show.", + "additionalProperties": option_history_schema() + }, "flags": { "type": "array", "items": { "type": "string", "enum": strings(&flags) } @@ -592,6 +1016,163 @@ fn history_schema() -> Value { }) } +/// The schema for one option set's statistics. +fn variant_calibration_schema() -> Value { + json!({ + "type": "object", + "required": ["variant"], + "additionalProperties": false, + "properties": { + "variant": { + "type": "string", + "description": "Digest of the stem, the administered options, and which was keyed." + }, + "key": string_array("The option ids keyed correct in this set."), + "distractors": string_array("The option ids offered alongside them."), + "administrations": string_array("The administrations pooled into these numbers."), + "n_examinees": { "type": "integer", "minimum": 0 }, + "p_value": proportion("Proportion correct, for this option set only."), + "point_biserial": { "type": "number", "minimum": -1.0, "maximum": 1.0 }, + "discrimination_index": { "type": "number", "minimum": -1.0, "maximum": 1.0 }, + "option_stats": { + "type": "object", + "description": "Per-option behaviour within this set, by option id.", + "additionalProperties": option_stat_schema() + }, + "irt": irt_schema(), + "flags": { + "type": "array", + "items": { + "type": "string", + "enum": strings( + &Flag::ALL.iter().map(|f| f.as_str()).collect::>() + ) + } + } + } + }) +} + +/// The schema for one option's cross-variant history. +fn option_history_schema() -> Value { + json!({ + "type": "object", + "additionalProperties": false, + "properties": { + "appearances": { "type": "integer", "minimum": 0 }, + "n_examinees": { "type": "integer", "minimum": 0 }, + "mean_selection_rate": proportion( + "Mean of the within-variant rates. For reading, not for acting on: each rate is \ + a share of a different set." + ), + "never_chosen": { + "type": "boolean", + "description": "Never chosen, anywhere. The claim that justifies retiring it." + } + } + }) +} + +/// The schema for `analysis/calibration.yaml`. +fn calibration_file_schema() -> Value { + json!({ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": format!("{BASE}/calibration.schema.json"), + "title": "coursebank calibration store", + "description": "The pooled statistics, by item id. Kept out of the banks: a bank's diff \ + should be a change of intent, not the output of a grading run. Committed \ + — every number here is a cohort aggregate, and there is no field for a \ + student.", + "type": "object", + "additionalProperties": false, + "properties": { + "schema_version": { "type": ["string", "number"] }, + "items": { + "type": "object", + "description": "Keyed by item id, which since 2.0 names the item course-wide and \ + carries no file name, so moving a question between banks does \ + not orphan its statistics.", + "additionalProperties": calibration_schema() + } + } + }) +} + +/// The schema for one file under `analysis/administrations/`. +fn measurements_file_schema() -> Value { + json!({ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": format!("{BASE}/measurements.schema.json"), + "title": "coursebank administration record", + "description": "What one administration measured. Written once and not rewritten, like \ + a seal: it records a thing that happened on a day. Cohort aggregates \ + only — no per-section or per-student breakdown, because a small cell \ + crossed with anything else stops being an aggregate.", + "type": "object", + "required": ["administration"], + "additionalProperties": false, + "properties": { + "schema_version": { "type": ["string", "number"] }, + "administration": { + "type": "object", + "required": ["id", "assessment", "n_examinees"], + "additionalProperties": false, + "properties": { + "id": { "type": "string" }, + "assessment": { "type": "string" }, + "term": { "type": "string" }, + "date": date("When it was given."), + "forms": string_array("The forms in play."), + "n_examinees": { + "type": "integer", + "minimum": 0, + "description": "The number to read before any of the others: a \ + point-biserial on twenty-seven students is a different \ + kind of claim than one on three hundred." + }, + "model": { "type": "string" }, + "generated": date("When the analysis was run."), + "coursebank": { "type": "string" } + } + }, + "items": { + "type": "array", + "items": { + "type": "object", + "required": ["item", "number", "n"], + "additionalProperties": false, + "properties": { + "item": { "type": "string" }, + "number": { "type": "integer", "minimum": 1 }, + "variant": { "type": "string" }, + "stem_digest": { "type": "string" }, + "n": { "type": "integer", "minimum": 0 }, + "p_value": proportion("Proportion correct on this administration."), + "point_biserial": { "type": "number", "minimum": -1.0, "maximum": 1.0 }, + "discrimination_index": { + "type": "number", "minimum": -1.0, "maximum": 1.0 + }, + "option_stats": { + "type": "object", + "additionalProperties": option_stat_schema() + }, + "irt": irt_schema(), + "flags": { + "type": "array", + "items": { + "type": "string", + "enum": strings( + &Flag::ALL.iter().map(|f| f.as_str()).collect::>() + ) + } + } + } + } + } + } + }) +} + /// The schema for a retirement record. fn retirement_schema() -> Value { json!({ @@ -614,9 +1195,12 @@ fn item_identity_properties() -> Value { json!({ "id": { "type": "string", - "pattern": "^q-[a-z0-9]+(-[a-z0-9]+)*-[0-9]{3}$", - "description": "Item id, e.g. q-glycolysis-014. Stable forever: assessment records \ - and stored responses refer to it." + "pattern": "^q-[a-z0-9]+(-[a-z0-9]+)*$", + "description": "Item id, e.g. q-glycolysis-rate-limiting-step. Stable forever: \ + assessment records and stored responses refer to it, so renaming one \ + is a migration rather than an edit. A trailing counter is no longer \ + expected — it recorded when the item was written, which git knows — \ + but an id that still has one stays valid." }, "version": { "type": "integer", @@ -633,9 +1217,10 @@ fn item_identity_properties() -> Value { "cognitive_process": cognitive_process(), "format": { "type": "string", - "enum": ["single_best_answer", "multiple_response", "true_false"], - "description": "single_best_answer requires exactly one keyed option; \ - multiple_response requires at least two." + "enum": ["single_best_answer", "multiple_response", "true_false", "open_response"], + "description": "single_best_answer keys exactly one option; multiple_response keys \ + two or more; open_response takes no options and is graded from its \ + solution." }, "bonus": { "type": "boolean" }, "points": { "type": "number", "exclusiveMinimum": 0.0 }, @@ -662,11 +1247,15 @@ fn item_content_properties() -> Value { "type": "array", "minItems": 2, "maxItems": 8, + "description": "Absent for an open_response item; at least two for any choice format.", "items": option_schema() }, - "learning_objectives": string_array( - "Objective ids this item measures. Reports aggregate on these, so an item with none \ - contributes to nothing." + "solution": solution_schema(), + "learning_targets": string_array( + "Target ids this item measures. Reports aggregate these onto the target's objective, \ + so an item with none contributes to nothing. Name the target rather than the \ + objective: the objective follows from it, and recording which target was asked \ + about is what lets a report explain a result instead of only stating it." ), "sources": { "type": "array", @@ -678,7 +1267,6 @@ fn item_content_properties() -> Value { "prerequisites": string_array("Objective ids a student needs before this item."), "assets": { "type": "array", "items": asset_schema() }, "design": design_schema(), - "calibration": calibration_schema(), "review": review_schema(), "history": { "type": "array", @@ -702,7 +1290,7 @@ fn item_schema() -> Value { } json!({ "type": "object", - "required": ["id", "level", "stem", "options"], + "required": ["id", "level", "stem"], "additionalProperties": false, "properties": Value::Object(properties) }) @@ -732,7 +1320,7 @@ fn bank_scope_schema() -> Value { outside it.", "properties": { "lectures": string_array("Lecture ids."), - "learning_objectives": string_array("Objective ids."), + "learning_targets": string_array("Target ids."), "units": string_array("Unit ids."), "topics": string_array("Topics.") } @@ -828,7 +1416,10 @@ fn blueprint_schema() -> Value { "type": "object", "description": "Minimum items per objective. Placed before level quotas, because \ a coverage requirement is the constraint most likely to become \ - unsatisfiable.", + unsatisfiable. A requirement is satisfied by items on any of \ + that objective's targets, so this stays short: name the dozen \ + objectives the exam is meant to cover, not the hundred targets \ + it is built from.", "additionalProperties": { "type": "integer", "minimum": 0 } }, "lectures": string_array("Restrict the draw to these lectures."), @@ -888,9 +1479,23 @@ fn placement_schema() -> Value { }, "points": { "type": "number", "minimum": 0.0 }, "bonus": { "type": "boolean" }, - "key": string_array("Keyed option letters as administered."), + "key": string_array( + "The option ids keyed correct for this administration. One for a \ + single_best_answer, chosen from the item's pool of defensible keys." + ), + "distractors": string_array( + "The option ids offered alongside the key. Resolved when the assessment is \ + assembled and written out explicitly, so a later bank edit cannot change the \ + paper. Empty means the whole pool." + ), + "variant": { + "type": "string", + "description": "Digest of the item as this administration showed it: stem, \ + administered options, and which was keyed. The key statistics \ + pool on." + }, "level": level(), - "learning_objectives": string_array("Objectives as administered."), + "learning_targets": string_array("Targets as administered."), "credit_overrides": { "type": "object", "description": "Partial credit decided after the fact, by option letter. Recording \ @@ -991,7 +1596,8 @@ mod tests { assert_eq!(props["options"]["maxItems"], 8); assert_eq!( props["options"]["items"]["properties"]["id"]["pattern"], - "^[A-H]$" + // Either form: the 2.0 name, or the letter it replaces. + "^(o-[a-z0-9]+(-[a-z0-9]+)*|[A-H])$" ); } @@ -1014,7 +1620,7 @@ mod tests { let dir = std::env::temp_dir().join(format!("cb-schema-{}", std::process::id())); std::fs::remove_dir_all(&dir).ok(); let written = write_all(&dir).unwrap(); - assert_eq!(written.len(), 3); + assert_eq!(written.len(), Kind::ALL.len()); for path in &written { assert!(path.exists()); let text = std::fs::read_to_string(path).unwrap(); diff --git a/src/authoring/lint.rs b/src/authoring/lint.rs index 040c613..70accb4 100644 --- a/src/authoring/lint.rs +++ b/src/authoring/lint.rs @@ -88,6 +88,16 @@ pub enum Rule { /// An expectation of low discrimination on a higher-level item. ContradictoryDesign, + // --- the option pool --- + /// Fewer usable distractors than a form shows. + ThinOptionPool, + /// An option that has never been administered and has not been retired. + UnusedOption, + /// An option retired without saying what it did. + UnjustifiedRetirement, + /// A distractor that has never been chosen, in any set it appeared in. + NonfunctioningDistractor, + // --- evidence --- /// Statistics describe an older version of the item. StaleCalibration, @@ -104,7 +114,7 @@ impl Rule { /// /// Used by `--list-rules`, and by the test that keeps this list in step with /// the enum. - pub const ALL: [Rule; 26] = [ + pub const ALL: [Rule; 30] = [ Rule::KeyIsLongest, Rule::UnevenOptionLength, Rule::WordRepeatCue, @@ -127,6 +137,10 @@ impl Rule { Rule::WeakFormatForLevel, Rule::ScoredBonusLevel, Rule::ContradictoryDesign, + Rule::ThinOptionPool, + Rule::UnusedOption, + Rule::UnjustifiedRetirement, + Rule::NonfunctioningDistractor, Rule::StaleCalibration, Rule::DifficultyMissed, Rule::DiscriminationMissed, @@ -152,6 +166,9 @@ impl Rule { // Statistics attached to text that has since changed are actively // misleading, which is worse than absent. R::StaleCalibration => Severity::High, + // An item that cannot fill a form is an item `assemble` will put on + // a paper short an option. + R::ThinOptionPool => Severity::High, // An unanswerable question for a screen-reader user. R::AssetWithoutAltText => Severity::High, @@ -178,6 +195,13 @@ impl Rule { | R::NoStudentFeedback | R::DifficultyMissed | R::DiscriminationMissed => Severity::Low, + + // A distractor that draws nobody across several administrations is + // evidence to act on, not a style note. + R::NonfunctioningDistractor => Severity::Medium, + // Both are tidiness: the item still works, but its pool is + // carrying something nobody has accounted for. + R::UnusedOption | R::UnjustifiedRetirement => Severity::Low, } } @@ -206,6 +230,10 @@ impl Rule { Rule::WeakFormatForLevel => "complete-format-level", Rule::ScoredBonusLevel => "complete-bonus-policy", Rule::ContradictoryDesign => "complete-design-conflict", + Rule::ThinOptionPool => "pool-thin", + Rule::UnusedOption => "pool-unused", + Rule::UnjustifiedRetirement => "pool-unjustified-retirement", + Rule::NonfunctioningDistractor => "evidence-nonfunctioning", Rule::StaleCalibration => "evidence-stale", Rule::DifficultyMissed => "evidence-difficulty", Rule::DiscriminationMissed => "evidence-discrimination", @@ -238,9 +266,11 @@ impl Rule { | Rule::WeakFormatForLevel | Rule::ScoredBonusLevel | Rule::ContradictoryDesign => "completeness", + Rule::ThinOptionPool | Rule::UnusedOption | Rule::UnjustifiedRetirement => "pool", Rule::StaleCalibration | Rule::DifficultyMissed | Rule::DiscriminationMissed + | Rule::NonfunctioningDistractor | Rule::DuplicateStem => "evidence", } } @@ -270,6 +300,10 @@ impl Rule { Rule::WeakFormatForLevel, Rule::ScoredBonusLevel, Rule::ContradictoryDesign, + Rule::ThinOptionPool, + Rule::UnusedOption, + Rule::UnjustifiedRetirement, + Rule::NonfunctioningDistractor, Rule::StaleCalibration, Rule::DifficultyMissed, Rule::DiscriminationMissed, @@ -302,6 +336,12 @@ impl Rule { Rule::WeakFormatForLevel => "true/false at an analytic level", Rule::ScoredBonusLevel => "a level the policy reserves for bonus is scored", Rule::ContradictoryDesign => "low expected discrimination on a higher-level item", + Rule::ThinOptionPool => "fewer usable distractors than a form shows", + Rule::UnusedOption => "an option has never been administered and is not retired", + Rule::UnjustifiedRetirement => "an option was retired without saying what it did", + Rule::NonfunctioningDistractor => { + "a distractor has never been chosen in any set it appeared in" + } Rule::StaleCalibration => "statistics describe an older version of the item", Rule::DifficultyMissed => "observed difficulty was far from predicted", Rule::DiscriminationMissed => "observed discrimination contradicted the prediction", @@ -421,7 +461,7 @@ pub fn lint_item(entry: &Entry, course: &CourseFile, t: &Thresholds) -> Vec = stem.split_whitespace().collect(); @@ -513,7 +553,7 @@ pub fn lint_item(entry: &Entry, course: &CourseFile, t: &Thresholds) -> Vec = it.options.iter().filter(|o| o.correct).collect(); let distractors: Vec<&crate::item::Choice> = it.options.iter().filter(|o| !o.correct).collect(); @@ -659,7 +699,7 @@ pub fn lint_item(entry: &Entry, course: &CourseFile, t: &Thresholds) -> Vec Vec { + if retirement.reason.trim().len() < 12 { + push( + Rule::UnjustifiedRetirement, + Severity::Low, + format!( + "option `{}` is retired with no real reason. The reason is the \ + finding — what it drew, or failed to draw — and it is the only \ + part of a retirement worth anything in two years", + option.id + ), + ); + } + } + None => { + // An option nobody has been shown is a draft, and a draft + // sitting in an approved item's pool will eventually be + // drawn onto a paper without ever having been reviewed + // against data. + if let Some(history) = it + .calibration + .as_ref() + .filter(|c| !c.options.is_empty()) + .and_then(|c| c.options.get(&option.id)) + { + if history.never_chosen && history.appearances > 1 { + push( + Rule::NonfunctioningDistractor, + Severity::Medium, + format!( + "option `{}` has appeared in {} option set(s) across {} \ + examinees and has never been chosen. One administration \ + would not show this; several do.", + option.id, history.appearances, history.n_examinees + ), + ); + } + } else if it + .calibration + .as_ref() + .is_some_and(|c| !c.options.is_empty()) + { + push( + Rule::UnusedOption, + Severity::Low, + format!( + "option `{}` has never been administered. Either it is waiting \ + its turn, or it was drafted and forgotten — retire it and say \ + which.", + option.id + ), + ); + } + } + } + } + let _ = keys; + } + + // --- evidence if !it.calibration_is_current() { push( Rule::StaleCalibration, @@ -868,6 +989,12 @@ fn lint_option_counts(catalog: &Catalog) -> Vec { if e.item.status == Status::Retired { continue; } + // An open-response item carries no options, so it is neither part of the + // count norm nor able to deviate from it. Leaving it out keeps a bank of + // four-option questions from reporting every essay as an odd count. + if !e.item.has_options() { + continue; + } by_bank.entry(e.bank.as_str()).or_default().push(e); } @@ -1190,7 +1317,7 @@ mod tests { fn entry(yaml: &str) -> Entry { let item: Item = serde_yaml_ng::from_str(yaml).expect("item parses"); Entry { - uid: format!("b::{}", item.id), + uid: item.id.clone(), bank: "b".into(), path: PathBuf::from("b.yaml"), index: 0, @@ -1209,6 +1336,45 @@ mod tests { c } + #[test] + fn a_pool_too_thin_to_fill_a_form_is_flagged() { + let c = codes( + r#" +id: q-a-001 +status: draft +level: 2 +stem: Which mechanism best explains the sigmoidal binding curve? +options: + - { id: o-shift, text: Ligand binding shifts the tetramer to a higher-affinity state, correct: true } + - { id: o-fixed, text: Each subunit binds with the same fixed affinity throughout } +"#, + ); + // Policy shows four options and the pool can supply two, so `assemble` + // would put this on a paper two short. + assert!(c.contains(&"pool-thin"), "{c:?}"); + } + + #[test] + fn a_retirement_with_no_finding_is_flagged() { + let c = codes( + r#" +id: q-a-001 +status: draft +level: 2 +stem: Which mechanism best explains the sigmoidal binding curve? +options: + - { id: o-shift, text: Ligand binding shifts the tetramer to a higher-affinity state, correct: true } + - { id: o-fixed, text: Each subunit binds with the same fixed affinity throughout } + - { id: o-consumed, text: "Ligand is consumed as it binds, depleting the available pool" } + - { id: o-oxidation, text: The heme iron changes oxidation state upon binding } + - { id: o-cooperative, text: Subunits bind independently of one another, retired: { 'on': 2026-09-20, reason: bad } } +"#, + ); + assert!(c.contains(&"pool-unjustified-retirement"), "{c:?}"); + // Four live distractors is enough for a four-option form. + assert!(!c.contains(&"pool-thin"), "{c:?}"); + } + #[test] fn clean_item_passes() { let c = codes( diff --git a/src/authoring/select.rs b/src/authoring/select.rs index 09fbff9..398a0b8 100644 --- a/src/authoring/select.rs +++ b/src/authoring/select.rs @@ -29,12 +29,13 @@ use std::collections::{BTreeMap, BTreeSet}; use crate::assessment::{Assessment, AssessmentFile, Blueprint, Form, Kind, Placement, Platform}; use crate::catalog::Catalog; -use crate::course::SCHEMA_VERSION; +use crate::course::{CourseFile, SCHEMA_VERSION}; use crate::date::Date; use crate::error::{Error, Result}; use crate::history::History; +use crate::item::{Choice, Item}; use crate::rng::Rng; -use crate::taxonomy::Level; +use crate::taxonomy::{Format, Level}; /// The result of a draw. #[derive(Debug, Clone)] @@ -84,7 +85,7 @@ pub fn select( let seed = blueprint.seed.unwrap_or(0); let mut notes = Vec::new(); - // ------------------------------------------------------------------ pool + // --- pool let eligible: Vec<&crate::catalog::Entry> = catalog .assemblable() .into_iter() @@ -101,15 +102,25 @@ pub fn select( let mut chosen: Vec = Vec::new(); let mut per_bank: BTreeMap = BTreeMap::new(); - // ---------------------------------------------------- objective minimums + // --- objective minimums // Placed first, because a coverage requirement is the constraint most likely // to become unsatisfiable once the level quotas are full. for (objective, needed) in &blueprint.objective_minimums { let mut have = 0; + // A requirement written against an objective is satisfied by items on + // any of its targets, which is the only way a coverage requirement stays + // writable: a blueprint that had to name each target separately would be + // as long as the registry, and would need editing every time an + // objective gained one. let mut candidates: Vec<&crate::catalog::Entry> = eligible .iter() .copied() - .filter(|e| e.item.learning_objectives.iter().any(|o| o == objective)) + .filter(|e| { + e.item + .learning_targets + .iter() + .any(|t| t == objective || catalog.course.objective_for(t) == objective) + }) .filter(|e| !e.item.bonus) .collect(); rank(&mut candidates, history, seed, "objective"); @@ -140,7 +151,7 @@ pub fn select( } } - // ------------------------------------------------------- level quotas + // --- level quotas let mut scored: Vec = Vec::new(); for (level, want) in &blueprint.level_counts { if *want == 0 { @@ -175,7 +186,7 @@ pub fn select( } } - // ------------------------------------------------------------ bonus items + // --- bonus items let mut bonus: Vec = Vec::new(); for (level, want) in &blueprint.bonus_counts { if *want == 0 { @@ -456,18 +467,29 @@ pub fn to_record( .chain(selection.bonus.iter().map(|u| (u, true))), ) { let e = catalog.require(uid)?; + let (key, distractors) = draw_options( + &e.item, + catalog.course.policy.options_per_item, + blueprint.seed.unwrap_or(0), + uid, + ); items.push(Placement { number, item: uid.clone(), - version: Some(e.item.version), + version: None, + stem_digest: Some(e.item.stem_digest()), + variant: Some(e.item.variant_digest(&key, &distractors)), fingerprint: Some(e.item.fingerprint()), points: Some(e.item.points(default_points)), bonus: is_bonus || e.item.bonus, - key: e.item.key_letters(), + distractors, + key, level: Some(e.item.level), - learning_objectives: e.item.learning_objectives.clone(), + learning_targets: e.item.learning_targets.clone(), credit_overrides: BTreeMap::new(), dropped: false, + dropped_as: None, + dropped_before_printing: false, }); } @@ -558,6 +580,85 @@ pub fn layout(record: &AssessmentFile, form: &Form) -> Vec { scored.into_iter().chain(bonus).collect() } +/// Draws the key and the distractors one placement administers. +/// +/// Resolved here, at assembly, and written into the record as explicit lists. +/// Nothing downstream samples: an export that drew its own options would print +/// a different paper every time the bank was touched. +/// +/// The draw is seeded on the blueprint and the item, so re-running `assemble` +/// with the same seed produces the same paper, and two items in one assessment +/// draw independently. +/// +/// # Arguments +/// +/// * `item` - the item, whose options are a pool. +/// * `per_item` - how many options a form shows, from course policy. +/// * `seed` - the blueprint seed. +/// * `uid` - the item id, salting the draw. +/// +/// # Returns +/// +/// The keyed ids and the distractor ids, each sorted, naming options of `item`. +/// Both empty for an item with no options, which is an open response. +pub fn draw_options( + item: &Item, + per_item: usize, + seed: u64, + uid: &str, +) -> (Vec, Vec) { + let (keys, distractors) = item.pool(); + if keys.is_empty() && distractors.is_empty() { + return (Vec::new(), Vec::new()); + } + + // Multiple response keys every correct option; anything else keys one, and + // when the pool offers several defensible keys the draw picks one so that + // the record says which. + let wanted_keys = match item.format { + Format::MultipleResponse => keys.len(), + _ => 1.min(keys.len()), + }; + let mut rng = Rng::from_label(&format!("{seed}/{uid}/options")); + + let mut key_ids = pick(&keys, wanted_keys, &mut rng); + key_ids.sort(); + + // A pool with fewer usable distractors than the policy asks for is a + // finding, not a failure: the form comes out short and `lint` says so, + // rather than `assemble` refusing to build the assessment at all. + let wanted = per_item.saturating_sub(key_ids.len()); + let mut distractor_ids = pick(&distractors, wanted.min(distractors.len()), &mut rng); + distractor_ids.sort(); + + (key_ids, distractor_ids) +} + +/// Takes `n` options, preferring the ones that were designed rather than merely +/// written. +/// +/// A distractor carrying a misconception and an error type is one you thought +/// about; one carrying neither is filler. When the pool is larger than the form, +/// the thought-about ones go on the paper. The shuffle comes first so that +/// options of equal standing are drawn by seed rather than by declaration +/// order. +fn pick(options: &[&Choice], n: usize, rng: &mut Rng) -> Vec { + if n >= options.len() { + return options.iter().map(|o| o.id.clone()).collect(); + } + let mut order: Vec = (0..options.len()).collect(); + rng.shuffle(&mut order); + order.sort_by_key(|&i| { + let o = options[i]; + u8::from(o.misconception.is_none()) + u8::from(o.error_type.is_none()) + }); + order + .into_iter() + .take(n) + .map(|i| options[i].id.clone()) + .collect() +} + /// The option order for one item on one form. /// /// # Arguments @@ -584,11 +685,14 @@ pub fn option_order(form: &Form, uid: &str, n: usize) -> Vec { /// # Arguments /// /// * `record` - the assessment record. +/// * `course` - the course, for the objective each tagged target belongs to. An +/// objective minimum is satisfied by items on any of that objective's +/// targets, the same way [`select`] fills it. /// /// # Returns /// /// One message per discrepancy, empty when the form matches the design. -pub fn check_blueprint(record: &AssessmentFile) -> Vec { +pub fn check_blueprint(record: &AssessmentFile, course: &CourseFile) -> Vec { let Some(bp) = &record.blueprint else { return vec!["the record carries no blueprint to check against".into()]; }; @@ -608,7 +712,11 @@ pub fn check_blueprint(record: &AssessmentFile) -> Vec { let got = record .items .iter() - .filter(|p| p.learning_objectives.iter().any(|o| o == objective)) + .filter(|p| { + p.learning_targets + .iter() + .any(|t| t == objective || course.objective_for(t) == objective) + }) .count(); if got < *needed { out.push(format!( @@ -619,7 +727,7 @@ pub fn check_blueprint(record: &AssessmentFile) -> Vec { out } -/// The set of objectives an assessment covers. +/// The set of learning targets an assessment covers, as tagged. /// /// # Arguments /// @@ -627,12 +735,36 @@ pub fn check_blueprint(record: &AssessmentFile) -> Vec { /// /// # Returns /// -/// The objective ids, deduplicated. -pub fn covered_objectives(record: &AssessmentFile) -> BTreeSet { +/// The target ids, deduplicated. +pub fn covered_targets(record: &AssessmentFile) -> BTreeSet { record .items .iter() - .flat_map(|p| p.learning_objectives.iter().cloned()) + .flat_map(|p| p.learning_targets.iter().cloned()) + .collect() +} + +/// The objectives an assessment covers, through the targets it measured. +/// +/// The list a coverage claim should be made from: "this exam covered eleven of +/// the course's thirty-two objectives" is a sentence about the blueprint, while +/// the same count over targets is a sentence about how finely the course happens +/// to be subdivided. +/// +/// # Arguments +/// +/// * `record` - the assessment record. +/// * `course` - the course, for the objective each tagged target belongs to. +/// +/// # Returns +/// +/// The objective ids, deduplicated. +pub fn covered_objectives(record: &AssessmentFile, course: &CourseFile) -> BTreeSet { + record + .items + .iter() + .flat_map(|p| p.learning_targets.iter()) + .map(|t| course.objective_for(t).to_string()) .collect() } @@ -649,6 +781,68 @@ mod tests { assert_eq!(form_label(27), "AB"); } + /// An item whose options are given as YAML, so the test needs no literal. + fn pool_item(options: &str) -> Item { + let src = format!( + r#"id: q-x +status: approved +level: 1 +cognitive_process: recall +stem: Which line holds the quality scores? +learning_targets: [t-x] +sources: [{{ lecture: L1 }}] +options: +{options}"# + ); + serde_yaml_ng::from_str(&src).expect("item parses") + } + + const DESIGNED: &str = r#" - { id: o-key, text: right, correct: true } + - { id: o-designed-a, text: a, misconception: mistakes the separator, error_type: recall_confusion } + - { id: o-designed-b, text: b, misconception: confuses the two, error_type: recall_confusion } + - { id: o-filler-a, text: c } + - { id: o-filler-b, text: d } +"#; + + #[test] + fn a_draw_prefers_designed_distractors_and_is_reproducible() { + let item = pool_item(DESIGNED); + + let (key, distractors) = draw_options(&item, 3, 1103, "q-x"); + assert_eq!(key, vec!["o-key".to_string()]); + assert_eq!(distractors.len(), 2); + // Thought-about distractors go on the paper before filler does. + assert!( + distractors.iter().all(|d| d.starts_with("o-designed")), + "{distractors:?}" + ); + + // Same seed, same paper. + assert_eq!(draw_options(&item, 3, 1103, "q-x"), (key, distractors)); + + // A retired option is not drawn, and the form comes out of the rest. + let retired = pool_item(&DESIGNED.replace( + "{ id: o-designed-a, text: a,", + "{ id: o-designed-a, text: a, retired: { 'on': 2026-09-20, reason: nonfunctioning },", + )); + let (_, after) = draw_options(&retired, 3, 1103, "q-x"); + assert!(!after.iter().any(|d| d == "o-designed-a"), "{after:?}"); + } + + #[test] + fn a_thin_pool_comes_out_short_rather_than_refusing_to_build() { + let item = pool_item( + " - { id: o-key, text: right, correct: true }\n - { id: o-one, text: wrong }\n", + ); + let (key, distractors) = draw_options(&item, 4, 7, "q-y"); + assert_eq!(key.len(), 1); + assert_eq!( + distractors.len(), + 1, + "one usable distractor, so one is drawn" + ); + } + #[test] fn option_order_is_a_reproducible_permutation() { let form = Form { @@ -689,12 +883,14 @@ blueprint: level_counts: { 1: 2, 3: 1 } objective_minimums: { lo-key: 2 } items: - - { number: 1, item: "b::q-1", level: 1, learning_objectives: [lo-key] } + - { number: 1, item: "b::q-1", level: 1, learning_targets: [lo-key] } - { number: 2, item: "b::q-2", level: 1 } "#, ) .unwrap(); - let issues = check_blueprint(&record); + let course: CourseFile = + serde_yaml_ng::from_str("course: { code: C, title: T, term: M }").unwrap(); + let issues = check_blueprint(&record, &course); assert!(issues.iter().any(|i| i.contains("level 3")), "{issues:?}"); assert!(issues.iter().any(|i| i.contains("lo-key")), "{issues:?}"); // Level 1 matches, so it must not be reported. @@ -702,17 +898,57 @@ items: } #[test] - fn covered_objectives_deduplicates() { + fn an_objective_minimum_is_met_by_items_on_its_targets() { + // The blueprint names the objective a report will classify; the items + // are tagged with the specific performances they measure. + let record: AssessmentFile = serde_yaml_ng::from_str( + r#" +assessment: { id: x, title: X } +blueprint: + objective_minimums: { lo-binding: 2 } +items: + - { number: 1, item: "b::q-1", level: 1, learning_targets: [t-kd] } + - { number: 2, item: "b::q-2", level: 3, learning_targets: [t-plot] } +"#, + ) + .unwrap(); + let course: CourseFile = serde_yaml_ng::from_str( + r#" +course: { code: C, title: T, term: M } +learning_objectives: + lo-binding: { text: Quantify binding. } +learning_targets: + t-kd: { text: Write the expression., objective: lo-binding } + t-plot: { text: Read a plot., objective: lo-binding } +"#, + ) + .unwrap(); + let issues = check_blueprint(&record, &course); + assert!( + !issues.iter().any(|i| i.contains("lo-binding")), + "two items on its targets satisfy it: {issues:?}" + ); + assert_eq!( + covered_objectives(&record, &course) + .into_iter() + .collect::>(), + vec!["lo-binding"], + "coverage is claimed at the tier the blueprint is written in" + ); + } + + #[test] + fn covered_targets_deduplicates() { let record: AssessmentFile = serde_yaml_ng::from_str( r#" assessment: { id: x, title: X } items: - - { number: 1, item: "b::q-1", learning_objectives: [lo-a, lo-b] } - - { number: 2, item: "b::q-2", learning_objectives: [lo-a] } + - { number: 1, item: "b::q-1", learning_targets: [lo-a, lo-b] } + - { number: 2, item: "b::q-2", learning_targets: [lo-a] } "#, ) .unwrap(); - let set = covered_objectives(&record); + let set = covered_targets(&record); assert_eq!(set.len(), 2); assert!(set.contains("lo-a")); } diff --git a/src/cli.rs b/src/cli.rs index 26003f7..ce75579 100644 --- a/src/cli.rs +++ b/src/cli.rs @@ -23,6 +23,8 @@ use clap::{Args, Parser, Subcommand, ValueEnum}; use coursebank::assessment::{Kind as AssessmentKind, Platform}; use coursebank::catalog::Severity; use coursebank::item::IrtModel; +use coursebank::lecture::Style as PageStyle; +use coursebank::references; use coursebank::store; /// Manage course item banks, assessments, and the analysis that comes back. @@ -47,6 +49,15 @@ pub(crate) struct Cli { pub(crate) enum Command { /// Create a new course directory. Init(InitArgs), + /// Inspect the course file. + #[command(subcommand)] + Course(CourseCommand), + /// One-time conversions from an older layout. + #[command(subcommand)] + Migrate(MigrateCommand), + /// Work with the bibliography. + #[command(subcommand)] + References(ReferencesCommand), /// Write JSON Schemas so your editor can validate the YAML as you type. Schema, /// Check every file for problems that must be fixed. @@ -55,6 +66,9 @@ pub(crate) enum Command { Lint(LintArgs), /// Summarize the item pool and objective coverage. Catalog(CatalogArgs), + /// Render a lecture's reading list, or check what backs each target. + #[command(subcommand)] + Lecture(LectureCommand), /// Work with item banks. #[command(subcommand)] Bank(BankCommand), @@ -83,10 +97,164 @@ pub(crate) enum Command { /// Write reports. #[command(subcommand)] Report(ReportCommand), + /// Freeze what was administered, with digests, before printing. + Seal(SealArgs), /// List what is in the response store. Data, } +/// `course`: the course file itself, which may be one file or many. +#[derive(Debug, Subcommand)] +pub(crate) enum CourseCommand { + /// List the files the course is assembled from. + Files, + /// Print the merged course, or write it to a file. + /// + /// Nothing reads what this writes. It exists so you can see what the + /// fragments add up to, and diff two revisions of a course that no longer + /// lives in one file. + Build { + /// Output path; prints to stdout when omitted. + #[arg(long)] + out: Option, + }, + /// Say which file defines an id. + Where { + /// A unit, lecture, objective, target, reference, or stimulus id. + id: String, + }, +} + +/// `migrate`: the one-time conversions, grouped so they are findable together. +#[derive(Debug, Subcommand)] +pub(crate) enum MigrateCommand { + /// Split one course.yaml into references.yaml, lectures/, and objectives/. + /// + /// The original is kept as course.yaml.bak, and the result is reassembled + /// and compared against it before the command reports success. + Split { + /// Show what would be written, and write nothing. + #[arg(long)] + dry_run: bool, + }, + /// Drop the trailing counter from every item id. + /// + /// A rename, not a normalization: `q-x-001` and `q-x` are unrelated + /// strings, so banks, records, seals, and the response store are rewritten + /// in one pass or not at all. Two ids that would collide abort it. + Counters { + /// Show the renames, and write nothing. + #[arg(long)] + dry_run: bool, + }, + /// Replace the `order:` integers with ordered declarations. + /// + /// Objective order comes from each lecture's `teaches` list, target order + /// from a `targets:` list this writes onto each objective. Run + /// `migrate split` first — without `teaches`, objective order has no + /// source. + Order { + /// Show what would change, and write nothing. + #[arg(long)] + dry_run: bool, + }, + /// Turn citations written into `note:` fields into real fields. + /// + /// Journal, volume, pages, and DOI parsed out of the prose, with whatever + /// the note still says left in it. A field the entry already declares is + /// never overwritten; a disagreement is reported instead. + References { + /// Show what would change, and write nothing. + #[arg(long)] + dry_run: bool, + }, + /// Fill in the stored `variant` column from the assessment records. + /// + /// Nothing in the tool needs it — a variant is derived from the placement + /// when a row has none. It is for pandas, DuckDB, and R, which see only + /// what is in the column and will otherwise average two option sets of one + /// stem into an item that never existed. + Variants { + /// Show what would change, and write nothing. + #[arg(long)] + dry_run: bool, + }, + /// Drop `version:` and `history:`, which 2.0 ignores. + /// + /// A stem's text is its identity: reword it and it is a new item with a new + /// id and `supersedes:` pointing back. `validate` enforces that against + /// every seal, so what a version number used to hint at is now checked. + Stems { + /// Show what would change, and write nothing. + #[arg(long)] + dry_run: bool, + }, + /// Rewrite option letters as names derived from the option text. + /// + /// A letter is a position, and a position in a field that pooled + /// statistics and `credit_overrides` join on is a bug waiting for someone + /// to reorder a YAML block. Run with --dry-run first: the names land in + /// the response store, so they are as permanent as an item id. + Options { + /// Show the derived names, and write nothing. + #[arg(long)] + dry_run: bool, + }, + /// Rewrite pre-2.0 `bank::item` ids as the item ids they name. + /// + /// Touches assessment records, seals, and the response store. Everything + /// keeps working unmigrated — an old id still resolves — but a store + /// holding both forms groups one question into two for anything reading the + /// Parquet without this tool. + Ids { + /// Show what would change, and write nothing. + #[arg(long)] + dry_run: bool, + }, +} + +/// `references`: the bibliography, and the formats other tools read it in. +#[derive(Debug, Subcommand)] +pub(crate) enum ReferencesCommand { + /// List every work, with the link a reading list would use. + /// + /// The column that matters is the last one: a work with no link is one a + /// student cannot reach from a report, which for a manuscript usually means + /// its DOI is missing. + List, + /// Write the bibliography in a citation format. + Export { + /// Which format to write. + #[arg(long, value_enum, default_value = "hayagriva")] + format: ReferenceFormat, + /// Output path; prints to stdout when omitted. + #[arg(long)] + out: Option, + }, +} + +/// The citation formats `references export` can write. +#[derive(Debug, Clone, Copy, ValueEnum)] +pub(crate) enum ReferenceFormat { + /// Hayagriva YAML, which Typst reads natively. + Hayagriva, + /// CSL-JSON, for Zotero, Pandoc, and CSL processors. + CslJson, + /// BibTeX. + Bibtex, +} + +impl ReferenceFormat { + /// The library-side format. + pub(crate) fn as_format(self) -> references::Format { + match self { + ReferenceFormat::Hayagriva => references::Format::Hayagriva, + ReferenceFormat::CslJson => references::Format::CslJson, + ReferenceFormat::Bibtex => references::Format::Bibtex, + } + } +} + #[derive(Debug, Args)] pub(crate) struct InitArgs { /// Course code, e.g. "BIOSC 1540". @@ -123,6 +291,67 @@ pub(crate) struct LintArgs { } /// CLI mirror of [`coursebank::catalog::Severity`]. +#[derive(Debug, Subcommand)] +pub(crate) enum LectureCommand { + /// Write the readings block for one lecture. + /// + /// The course file is the source of truth for what a lecture assigns and why, + /// so the list on the website is generated from it. Target numbers are + /// positional and are resolved here rather than authored. + Readings { + /// Lecture id, e.g. L1.1. + id: String, + /// Which flavour of Markdown to write. + #[arg(long, value_enum, default_value = "quarto")] + style: StyleArg, + /// Output path; prints to stdout when omitted. + #[arg(long)] + out: Option, + }, + /// Write the learning objectives block for one lecture, with each + /// objective's learning targets enumerated beneath it. + /// + /// One block rather than two: the objective and its targets are the same + /// claim at two grains, and separate sections would print every target + /// twice. The numbering comes from the same place as the `_(T 4, 7)_` lists + /// in `readings`, so generating one and hand-writing the other is what this + /// exists to prevent. + Objectives { + /// Lecture id, e.g. L1.1. + id: String, + /// Which flavour of Markdown to write. + #[arg(long, value_enum, default_value = "quarto")] + style: StyleArg, + /// Output path; prints to stdout when omitted. + #[arg(long)] + out: Option, + }, + /// Show the readings behind each learning target, and which have none. + Coverage { + /// Only this lecture's targets. + #[arg(long)] + lecture: Option, + }, +} + +#[derive(Debug, Clone, Copy, ValueEnum)] +pub(crate) enum StyleArg { + /// Pandoc definition lists, as a Quarto lecture page wants them. + Quarto, + /// Plain Markdown bullets. + Plain, +} + +impl StyleArg { + /// Converts the CLI value into the library's [`PageStyle`]. + pub(crate) fn as_style(self) -> PageStyle { + match self { + StyleArg::Quarto => PageStyle::Quarto, + StyleArg::Plain => PageStyle::Plain, + } + } +} + #[derive(Debug, Clone, Copy, ValueEnum)] pub(crate) enum SeverityArg { Low, @@ -143,7 +372,7 @@ impl SeverityArg { #[derive(Debug, Args)] pub(crate) struct CatalogArgs { - /// Show per-objective coverage and the gaps in it. + /// Show coverage per objective and per target, and the gaps in it. #[arg(long)] pub(crate) coverage: bool, /// Show topic counts. @@ -232,7 +461,8 @@ pub(crate) struct AssembleArgs { /// Bonus items per level, same syntax. #[arg(long, value_delimiter = ',')] pub(crate) bonus: Vec, - /// Minimum items per objective, e.g. --require lo-kinetics=2. + /// Minimum items per objective, e.g. --require lo-kinetics=2. Satisfied by + /// items on any of that objective's targets. #[arg(long, value_delimiter = ',')] pub(crate) require: Vec, /// Restrict the draw to these lectures. @@ -309,6 +539,11 @@ pub(crate) enum ExportCommand { /// Leave per-option feedback out of the package. #[arg(long)] no_feedback: bool, + /// Canvas attempt limit; -1 for unlimited. Overrides the assessment's + /// `attempts` field. More than one attempt also makes per-option feedback + /// show the hint rather than the misconception. + #[arg(long)] + attempts: Option, }, /// Render a printable exam, answer key, and answer sheet. /// @@ -350,6 +585,54 @@ pub(crate) enum ExportCommand { #[arg(long)] out: Option, }, + /// Write a Quarto worksheet and a matching solutions document. + /// + /// The worksheet holds the questions and nothing else; the solutions document + /// adds the key, the worked reasoning, the rubric, and the readings to revisit. + /// This is the path that does not go through Canvas, so a student can practice + /// from the `.qmd` and check themselves against the solutions. Render each with + /// `quarto render .qmd`. + Practice { + /// Assessment id. + id: String, + /// Which form's ordering to use. + #[arg(long, default_value = "A")] + form: String, + /// Which documents to write; defaults to both. Values: worksheet, solutions. + #[arg(long, value_name = "DOC")] + variant: Vec, + /// Output directory; defaults to build/. + #[arg(long)] + out: Option, + /// Do not leave written-answer space after open-response questions. + #[arg(long)] + no_answer_space: bool, + }, + /// Write a Quarto questions partial and an encrypted, password-gated + /// solutions bundle for the course website. + /// + /// Needs the `site` feature (`cargo build --features site`). Writes + /// `_questions.qmd` and `-solutions.json` into the output directory, and + /// prints a fresh password that the files do not store. + Site { + /// Assessment id. + id: String, + /// Which form's option order to print. + #[arg(long, default_value = "A")] + form: String, + /// Output directory; defaults to build/. Point it at the page's own + /// directory so the browser fetches the bundle beside the page. + #[arg(long)] + out: Option, + /// Encrypt with this password instead of a generated one. Use only to + /// re-encrypt a page with a known password. + #[arg(long)] + password: Option, + /// Also write questions.css and solutions.js into this directory, e.g. + /// the site's static assets folder. They install once, not per page. + #[arg(long, value_name = "DIR")] + assets: Option, + }, } #[derive(Debug, Subcommand)] @@ -436,12 +719,23 @@ impl FormatArg { #[derive(Debug, Subcommand)] pub(crate) enum IngestCommand { - /// Read a directory of Gradescope per-question CSV exports. + /// Read one directory of Gradescope per-question CSV exports per form. + /// + /// Write each source as `FORM=DIR`. A bare path takes `--form`. Gradescope { - /// The directory holding 1.csv .. N.csv. - dir: PathBuf, + /// The directories, e.g. A=exports/e1-a B=exports/e1-b. + #[arg(value_name = "SOURCE", required = true)] + sources: Vec, #[command(flatten)] common: IngestCommon, + /// Ingest even when a directory's graded keys do not match the form it + /// was labelled with. + #[arg(long)] + allow_mismatch: bool, + /// The export numbers questions by recorded number rather than by + /// printed position. Only for a platform that is not Gradescope. + #[arg(long)] + recorded_numbers: bool, }, /// Read a Canvas "Student Analysis" CSV. Canvas { @@ -461,6 +755,14 @@ pub(crate) enum AnalyzeCommand { /// Pool every stored administration of this assessment. #[arg(long)] pooled: bool, + /// Also write the administration record under analysis/. + /// + /// Two CSVs, one row per question and one per question-and-option, + /// written once and never rewritten. Cohort aggregates only, so unlike + /// the response data they are meant to be committed — which is what + /// keeps the history when `data/` rotates. + #[arg(long = "record")] + record: bool, }, /// Fit an IRT model. Irt { @@ -533,6 +835,60 @@ pub(crate) enum ReportCommand { /// Leave out the comparison to the class. #[arg(long)] no_comparison: bool, + /// Also write a Typst diagnostic per student. + #[arg(long)] + typst: bool, + /// Skip the Markdown reports. + #[arg(long)] + no_markdown: bool, + /// Leave the per-question map out of the Typst report. + #[arg(long)] + no_questions: bool, + /// Leave per-option feedback out of the Typst report. + #[arg(long)] + no_feedback: bool, + /// Leave out the hint written for the option the student chose. + #[arg(long)] + no_hints: bool, + /// Leave the "kinds of thinking" comparison out of the Typst report. + #[arg(long)] + no_levels: bool, + /// Leave the per-objective mastery table out of the Typst report. + #[arg(long)] + no_objectives: bool, + /// Leave the objectives called out as strengths out of the Typst report. + #[arg(long)] + no_strengths: bool, + /// Leave the objectives called out as focus areas out of the Typst report. + #[arg(long)] + no_focus: bool, + /// Leave the dropped-question notice out of the Typst report. + #[arg(long)] + no_dropped_questions: bool, + /// Leave lecture-review suggestions out of the Typst report. + #[arg(long)] + no_review_lectures: bool, + /// Leave suggested study groups and readings out of the Typst report. + #[arg(long)] + no_study: bool, + /// Name the misconception each chosen distractor was written to catch. + /// + /// Written for you rather than for them, so it reads clinically next to + /// the feedback. Off by default for that reason, not for secrecy. + #[arg(long)] + misconceptions: bool, + /// Include the worked solution for every missed question. + /// + /// This is the solutions document. Reasonable for a question you will + /// not use again, and a way to publish your bank if you reuse it. + #[arg(long)] + solutions: bool, + /// Use this template instead of the usual lookup. + #[arg(long)] + template: Option, + /// Also write each payload as JSON. + #[arg(long)] + json: bool, }, /// The instructor's item analysis. Cohort { @@ -544,9 +900,42 @@ pub(crate) enum ReportCommand { /// Output path; defaults to reports/-cohort.md. #[arg(long)] out: Option, + /// Also write a Typst class diagnostic. + #[arg(long)] + typst: bool, + /// Skip the Markdown report. + #[arg(long)] + no_markdown: bool, + /// Use this template instead of the usual lookup. + #[arg(long)] + template: Option, + /// Also write the payload as JSON. + #[arg(long)] + json: bool, }, } +#[derive(Debug, Args)] +pub(crate) struct SealArgs { + /// Assessment id. + pub(crate) id: String, + /// Check the existing seal against the course instead of writing one. + #[arg(long)] + pub(crate) check: bool, + /// Overwrite an existing seal. + #[arg(long)] + pub(crate) force: bool, + /// Only these forms; defaults to every form the record declares. + #[arg(long, value_delimiter = ',')] + pub(crate) form: Vec, + /// Store only fingerprints, not the stem and option text. + #[arg(long)] + pub(crate) no_content: bool, + /// Output path; defaults to seals/.yaml. + #[arg(long)] + pub(crate) out: Option, +} + #[cfg(test)] mod tests { use super::*; diff --git a/src/commands.rs b/src/commands.rs index e4a7f7d..9c34908 100644 --- a/src/commands.rs +++ b/src/commands.rs @@ -8,8 +8,10 @@ //! calls the matching handler. The handlers themselves live in submodules that //! follow the workflow described in the crate documentation: //! -//! - [`project`] — set up and check a course: `init`, `schema`, `validate`, -//! `lint`, `catalog`. +//! - [`project`] — set up and check a course: `init`, `course`, `schema`, +//! `validate`, `lint`, `catalog`. +//! - [`lectures`] — render a lecture's reading list and check what backs each +//! objective: `lecture`. //! - [`banks`] — manage items and build assessments: `bank`, `assessment`, //! `assemble`, `usage`. //! - [`export`] — turn an assessment into deliverables: `export`, `template`. @@ -22,6 +24,7 @@ pub(crate) mod analysis; pub(crate) mod banks; pub(crate) mod export; +pub(crate) mod lectures; pub(crate) mod project; use coursebank::error::Result; @@ -52,10 +55,14 @@ pub(crate) enum Outcome { pub(crate) fn run(cli: &Cli) -> Result { match &cli.command { Command::Init(args) => project::init(cli, args), + Command::Course(sub) => project::course(cli, sub), + Command::References(sub) => project::references(cli, sub), + Command::Migrate(sub) => project::migrate(cli, sub), Command::Schema => project::schema(cli), Command::Validate => project::validate(cli), Command::Lint(args) => project::lint(cli, args), Command::Catalog(args) => project::catalog(cli, args), + Command::Lecture(sub) => lectures::lecture(cli, sub), Command::Bank(sub) => banks::bank(cli, sub), Command::Assessment(sub) => banks::assessment(cli, sub), Command::Assemble(args) => banks::assemble(cli, args), @@ -66,6 +73,7 @@ pub(crate) fn run(cli: &Cli) -> Result { Command::Analyze(sub) => analysis::analyze(cli, sub), Command::Calibrate(args) => analysis::calibrate(cli, args), Command::Report(sub) => analysis::report(cli, sub), + Command::Seal(args) => analysis::seal(cli, args), Command::Data => analysis::data(cli), } } diff --git a/src/commands/analysis.rs b/src/commands/analysis.rs index 6c3b7c7..560251f 100644 --- a/src/commands/analysis.rs +++ b/src/commands/analysis.rs @@ -4,53 +4,240 @@ //! The responses-to-report pipeline. //! -//! Once an assessment has been given, responses come back through [`ingest`], -//! statistics come out of [`analyze`] (classical, IRT, or per-student), [`calibrate`] -//! writes those statistics back onto the items, and [`report`] produces the -//! student and cohort documents. [`data`] lists what the response store holds. +//! [`seal`] freezes what was administered before the papers are printed, which is +//! what lets everything downstream translate a student's marks back into the +//! bank's own lettering. Once an assessment has been given, responses come back +//! through [`ingest`], statistics come out of [`analyze`] (classical, IRT, or +//! per-student), [`calibrate`] writes those statistics back onto the items, and +//! [`report`] produces the student and cohort documents. [`data`] lists what the +//! response store holds. use coursebank::calibrate; use coursebank::canvas; +use coursebank::catalog::Severity; use coursebank::classical::{self, Thresholds}; -use coursebank::error::Result; -use coursebank::gradescope; +use coursebank::decode::Numbering; +use coursebank::diagnostic; +use coursebank::error::{Error, Result}; +use coursebank::intake::{self, Source}; use coursebank::irt; use coursebank::layout::Layout; use coursebank::report; +use coursebank::seal::{self, SealFile}; use coursebank::store::{self, Store}; use coursebank::students; +use coursebank::typst::diagnostic as typst_diagnostic; +use coursebank::typst::{self, Variant}; use coursebank::yaml; -use crate::cli::{AnalyzeCommand, CalibrateArgs, Cli, IngestCommand, ReportCommand}; +use crate::cli::{AnalyzeCommand, CalibrateArgs, Cli, IngestCommand, ReportCommand, SealArgs}; use crate::commands::Outcome; use crate::helpers::{context, load, load_record, read_salt, responses_for, truncate}; -/// `ingest`: read a Gradescope directory or a Canvas CSV into the response store. +/// `seal`: freeze what was administered, or check the freeze. /// -/// Enriches the parsed responses against the record, optionally pseudonymizes the -/// identifiers, and — unless `--dry-run` — writes them in the chosen format. +/// Write the seal after exporting the papers and before printing them. It records +/// the stem and options of every item, the printed-letter map for every form, and +/// digests over both, so that the question "is the bank still what the students +/// saw?" has an answer six months from now. +pub(crate) fn seal(cli: &Cli, args: &SealArgs) -> Result { + let catalog = load(cli)?; + let record = load_record(&catalog, &args.id)?; + let path = args + .out + .clone() + .unwrap_or_else(|| SealFile::path(&catalog.layout, &record.assessment.id)); + + if args.check { + let existing = SealFile::load(&path).map_err(|_| { + Error::usage(format!( + "no seal at {}. Write one with `coursebank seal {}`", + path.display(), + args.id + )) + })?; + + let drift = existing.verify(&catalog, &record); + if drift.is_empty() { + println!( + "{} matches the seal written {} ({})", + args.id, + existing.seal.sealed_on, + seal::short(&existing.seal.digest) + ); + return Ok(Outcome::Ok); + } + + println!( + "{} finding(s) against the seal written {}:\n", + drift.len(), + existing.seal.sealed_on + ); + for finding in &drift { + println!(" {:<6} {}", finding.severity.label(), finding.message); + } + if drift.iter().any(|d| d.is_blocking()) { + println!( + "\nAnalysis that pools this administration with another is comparing two \ + different questions. Per-option feedback in student reports may name the wrong \ + option." + ); + } + return Ok(Outcome::Findings); + } + + if path.exists() && !args.force { + return Err(Error::usage(format!( + "{} already exists. A seal is meant to be written once, before the exam is printed; \ + pass --force only if you are re-sealing an assessment that was never administered", + path.display() + ))); + } + + let opts = seal::Options { + content: !args.no_content, + forms: args.form.clone(), + }; + let file = seal::build(&catalog, &record, &opts)?; + file.save(&path)?; + + println!( + "wrote {} ({} item(s), {} form(s), {})", + path.display(), + file.items.len(), + file.forms.len(), + seal::short(&file.seal.digest) + ); + let mut advisories = Vec::new(); + for form in &file.forms { + let permuted = form + .questions + .iter() + .filter(|q| q.options.iter().any(|o| o.printed != o.canonical)) + .count(); + println!( + " form {}: {} question(s), {} with permuted options, {}", + form.id, + form.questions.len(), + permuted, + seal::short(&form.digest) + ); + advisories.extend(seal::balance(form).notes()); + } + + // The last moment before printing is the only cheap moment to notice that a + // shuffle produced a sequence a student will read as a mistake. + if !advisories.is_empty() { + println!(); + for note in &advisories { + println!("! {note}"); + } + println!( + "\nRe-seed a form by editing its `seed:` in the record, then re-export and re-seal \ + with --force. Nothing else has to change." + ); + } + + if !cli.quiet { + println!( + "\nCommit this file. Check it any time with: coursebank seal {} --check", + args.id + ); + } + Ok(Outcome::Ok) +} + +/// `ingest`: read grading exports into the response store. +/// +/// The Gradescope path takes one directory per form and merges them into a single +/// administration, translating each form's printed letters into the bank's +/// lettering on the way in. See [`coursebank::intake`]. pub(crate) fn ingest(cli: &Cli, sub: &IngestCommand) -> Result { let catalog = load(cli)?; let (common, mut set) = match sub { - IngestCommand::Gradescope { dir, common } => { + IngestCommand::Gradescope { + sources, + common, + allow_mismatch, + recorded_numbers, + } => { let record = load_record(&catalog, &common.assessment)?; - let ctx = context(&catalog, &record, common)?; - let import = gradescope::ingest_dir(dir, &ctx)?; + let sealed = SealFile::find(&catalog.layout, &record.assessment.id)?; + + if let Some(file) = &sealed { + let drift = file.verify(&catalog, &record); + let blocking: Vec<&seal::Drift> = + drift.iter().filter(|d| d.is_blocking()).collect(); + for finding in &drift { + println!("! {} {}", finding.severity.label(), finding.message); + } + if !blocking.is_empty() { + println!( + "\nThe seal still describes the papers the students held, so ingest will \ + use it. The bank has moved since; `coursebank seal {} --check` lists \ + what.", + common.assessment + ); + } + } else if !cli.quiet { + println!( + "! no seal for {}; the option maps will be derived from the record as it \ + stands today. Write one next time with `coursebank seal {}` before printing", + common.assessment, common.assessment + ); + } + + let mut parsed = Vec::new(); + for text in sources { + parsed.push(Source::parse(text, common.form.as_deref())?); + } + for source in &parsed { + if !intake::looks_like_export(&source.dir) { + return Err(Error::usage(format!( + "{} holds no files named `1.csv` … `N.csv`. Gradescope writes one file \ + per question; point at the directory those were unzipped into", + source.dir.display() + ))); + } + } + + let date = match &common.date { + Some(text) => Some(text.parse()?), + None => None, + }; + let opts = intake::Options { + date, + numbering: if *recorded_numbers { + Numbering::Recorded + } else { + Numbering::Printed + }, + strict: !*allow_mismatch, + }; + + let result = intake::run(&catalog, &record, sealed.as_ref(), &parsed, &opts)?; + + for form in &result.forms { + println!("{}", intake::describe(form)); + } // Grading-time partial credit is an ambiguity signal worth surfacing - // right here, while the exam is fresh. - for question in &import.questions { + // while the exam is fresh, and it is per form because a regrade + // applied to one form and not the other is its own problem. + for (form, question) in result.questions() { for (letter, value, note) in question.partial_credit() { println!( - "! q{}: option {letter} earned {value} of {} points at grading time{}", + "! form {form} q{}: option {letter} earned {value} of {} points at \ + grading time{}", question.number, question.points_possible(), note.map(|n| format!(" — {n}")).unwrap_or_default() ); } } - (common, import.responses) + + (common, result.responses) } IngestCommand::Canvas { file, common } => { let record = load_record(&catalog, &common.assessment)?; @@ -93,7 +280,7 @@ pub(crate) fn ingest(cli: &Cli, sub: &IngestCommand) -> Result { println!("wrote {}", path.display()); } println!( - "\nNext: coursebank analyze items {}\n coursebank report cohort {}", + "\nNext: coursebank analyze items {}\n coursebank report cohort {} --typst", common.assessment, common.assessment ); Ok(Outcome::Ok) @@ -107,12 +294,46 @@ pub(crate) fn analyze(cli: &Cli, sub: &AnalyzeCommand) -> Result { let store = Store::open(catalog.layout.data())?; match sub { - AnalyzeCommand::Items { id, pooled } => { + AnalyzeCommand::Items { + id, + pooled, + record: write_record, + } => { let record = load_record(&catalog, id)?; let set = responses_for(&store, &catalog, &record, *pooled)?; let analysis = classical::analyze(&set, &Thresholds::default(), Some(&record), Some(&catalog)); + if *write_record { + // The full administration id, not the assessment id: `e1` is + // given again next year, and a record named after it would + // collide with this cohort's — which `write_csv` would report + // as "already recorded" when it is a different exam entirely. + let administration = coursebank::responses::administration_id( + &catalog.course.course.code, + record + .assessment + .term + .as_deref() + .unwrap_or(&catalog.course.course.term), + &record.assessment.id, + ); + let written = calibrate::record_measurements( + &catalog.layout, + &administration, + &analysis, + &catalog, + Some(&record), + )?; + for path in &written { + println!("wrote {}", path.display()); + } + println!( + "\nCohort aggregates only, so these are committed. Review with\n git diff \ + analysis/\n" + ); + } + for w in &analysis.warnings { println!("! {w}\n"); } @@ -239,7 +460,7 @@ pub(crate) fn analyze(cli: &Cli, sub: &AnalyzeCommand) -> Result { println!( " {:>4.0}% {}", rate * 100.0, - catalog.course.objective_text(objective) + catalog.course.text_for(objective) ); } } @@ -276,21 +497,28 @@ pub(crate) fn calibrate(cli: &Cli, args: &CalibrateArgs) -> Result { } if !args.apply { println!( - "Nothing written. Re-run with --apply to write these {} change(s) into the bank \ - files, then review the git diff.", + "Nothing written. Re-run with --apply to write these {} change(s) into \ + analysis/calibration.yaml, then review the git diff.", plan.changes.len() ); return Ok(Outcome::Ok); } - for path in calibrate::apply(&plan)? { - println!("updated {}", path.display()); - } - println!("\nReview the diff before committing: git diff banks/"); + let path = calibrate::apply(&catalog.layout, &plan)?; + println!("updated {}", path.display()); + println!( + "\nReview the diff before committing: git diff analysis/\n\nThe banks are untouched. \ + Statistics are cohort aggregates with no student in\n them, which is why analysis/ is \ + committed and data/ is not." + ); Ok(Outcome::Ok) } -/// `report`: write per-student reports or the instructor's cohort item analysis. +/// `report`: write per-student diagnostics or the class diagnostic. +/// +/// Markdown is still the default for both, because it diffs and reads in a +/// terminal. `--typst` adds a document per student, or one for the class, built +/// from the templates in `templates/` and compiled with `typst compile`. pub(crate) fn report(cli: &Cli, sub: &ReportCommand) -> Result { let catalog = load(cli)?; let store = Store::open(catalog.layout.data())?; @@ -302,6 +530,22 @@ pub(crate) fn report(cli: &Cli, sub: &ReportCommand) -> Result { out, ability, no_comparison, + typst: want_typst, + no_markdown, + no_questions, + no_feedback, + no_hints, + no_levels, + no_objectives, + no_strengths, + no_focus, + no_dropped_questions, + no_review_lectures, + no_study, + misconceptions, + solutions, + template, + json, } => { let record = load_record(&catalog, id)?; let set = responses_for(&store, &catalog, &record, false)?; @@ -311,26 +555,146 @@ pub(crate) fn report(cli: &Cli, sub: &ReportCommand) -> Result { None }; let cohort = students::summarize(&set, &catalog.course, Some(&catalog), fit.as_ref()); - - let opts = report::StudentOptions { - ability: *ability, - comparison: !no_comparison, - ..report::StudentOptions::default() - }; let dir = out .clone() .unwrap_or_else(|| catalog.layout.reports().join(id)); - let written = - report::write_all_students(&dir, &cohort, &catalog.course, &record, &opts, *html)?; + + warn_about_drift(&catalog, &record, cli.quiet); + + if !no_markdown { + let opts = report::StudentOptions { + ability: *ability, + comparison: !no_comparison, + ..report::StudentOptions::default() + }; + let written = report::write_all_students( + &dir, + &cohort, + &catalog.course, + &record, + &opts, + *html, + )?; + println!( + "wrote {} Markdown file(s) for {} student(s) in {}", + written.len(), + cohort.students.len(), + dir.display() + ); + } + + if !want_typst { + return Ok(Outcome::Ok); + } + + // The item analysis supplies the class rate per question, which is + // what makes "you missed q17, and so did most of the class" possible. + let analysis = + classical::analyze(&set, &Thresholds::default(), Some(&record), Some(&catalog)); + + let mut config = typst::load_config(&catalog.layout)?.resolve(Variant::StudentReport); + // Each flag only turns a section off; leaving it unset keeps whatever + // templates/typst.yaml already resolved to, so a CLI run that doesn't + // mention a section never overrides a course's saved preference. + if *no_levels { + config.student_sections.levels = false; + } + if *no_objectives { + config.student_sections.objectives = false; + } + if *no_strengths { + config.student_sections.strengths = false; + } + if *no_focus { + config.student_sections.focus = false; + } + if *no_dropped_questions { + config.student_sections.dropped_questions = false; + } + if *no_review_lectures { + config.student_sections.review_lectures = false; + } + if *no_study { + config.student_sections.study = false; + } + let meta = typst_diagnostic::Meta::new(&catalog, &record, cohort.students.len()); + let opts = diagnostic::Options { + comparison: !no_comparison, + questions: !no_questions, + feedback: !no_feedback, + hints: !no_hints, + misconceptions: *misconceptions, + solutions: *solutions, + ability: *ability, + ..diagnostic::Options::default() + }; + + // Said once, at the moment it would matter, rather than left in the + // help text where nobody reads it twice. + if *solutions && !cli.quiet { + println!( + "! --solutions puts the worked solution for every missed question into \ + {} report(s). Reusing these items next term means reusing them against \ + students who may have seen this page", + cohort.students.len() + ); + } + + let mut written = 0usize; + let mut used_embedded = false; + for summary in &cohort.students { + let built = + diagnostic::student(summary, &cohort, &catalog, &set, Some(&analysis), &opts); + let document = typst_diagnostic::render_student( + &catalog.layout, + &meta, + &built, + &config, + template.as_deref(), + )?; + used_embedded |= document.origin == typst::Origin::Embedded; + for warning in &document.warnings { + eprintln!("warning: {warning}"); + } + + let stem = typst_diagnostic::student_stem(id, &summary.student_key); + let path = dir.join(format!("{stem}.typ")); + yaml::write_text(&path, &document.text)?; + written += 1; + + if *json { + yaml::write_json(&dir.join(format!("{stem}.json")), &built)?; + } + } + println!( - "wrote {} file(s) for {} student(s) in {}", - written.len(), - cohort.students.len(), - dir.display() + "wrote {}", + typst_diagnostic::summary(written, cohort.students.len()) ); + if !cli.quiet { + println!( + "Compile them all with:\n for f in {}/*.typ; do typst compile \"$f\"; done", + dir.display() + ); + if used_embedded { + println!( + "These used the built-in template. To take over the layout:\n \ + coursebank template dump --variant student-report" + ); + } + } Ok(Outcome::Ok) } - ReportCommand::Cohort { id, html, out } => { + + ReportCommand::Cohort { + id, + html, + out, + typst: want_typst, + no_markdown, + template, + json, + } => { let record = load_record(&catalog, id)?; let set = responses_for(&store, &catalog, &record, false)?; let analysis = @@ -338,20 +702,58 @@ pub(crate) fn report(cli: &Cli, sub: &ReportCommand) -> Result { let fit = irt::fit(&set.matrix(false), &irt::Options::default()); let cohort = students::summarize(&set, &catalog.course, Some(&catalog), Some(&fit)); - let markdown = report::cohort(&analysis, &cohort, &catalog, &record, Some(&fit)); + warn_about_drift(&catalog, &record, cli.quiet); + let path = out .clone() .unwrap_or_else(|| catalog.layout.reports().join(format!("{id}-cohort.md"))); - yaml::write_text(&path, &markdown)?; - println!("wrote {}", path.display()); - if *html { - let html_path = path.with_extension("html"); - let title = format!("{} — item analysis", record.assessment.title); - yaml::write_text(&html_path, &report::to_html(&markdown, &title))?; - println!("wrote {}", html_path.display()); + if !no_markdown { + let markdown = report::cohort(&analysis, &cohort, &catalog, &record, Some(&fit)); + yaml::write_text(&path, &markdown)?; + println!("wrote {}", path.display()); + + if *html { + let html_path = path.with_extension("html"); + let title = format!("{} — item analysis", record.assessment.title); + yaml::write_text(&html_path, &report::to_html(&markdown, &title))?; + println!("wrote {}", html_path.display()); + } } - Ok(Outcome::Ok) + + let built = diagnostic::cohort(&analysis, &cohort, &catalog, &record, &set, Some(&fit)); + + for line in typst_diagnostic::headline(&built) { + println!("{line}"); + } + + if *want_typst { + let config = typst::load_config(&catalog.layout)?.resolve(Variant::CohortReport); + let meta = typst_diagnostic::Meta::new(&catalog, &record, cohort.students.len()); + let document = typst_diagnostic::render_cohort( + &catalog.layout, + &meta, + &built, + &config, + template.as_deref(), + )?; + for warning in &document.warnings { + eprintln!("warning: {warning}"); + } + let typst_path = path.with_extension("typ"); + yaml::write_text(&typst_path, &document.text)?; + println!("wrote {} (from {})", typst_path.display(), document.origin); + + if *json { + yaml::write_json(&path.with_extension("json"), &built)?; + } + } + + Ok(if built.revise.is_empty() { + Outcome::Ok + } else { + Outcome::Findings + }) } } } @@ -384,3 +786,37 @@ pub(crate) fn data(cli: &Cli) -> Result { } Ok(Outcome::Ok) } + +/// Says so when the bank has moved since the exam was sealed. +/// +/// A report built from a drifted bank is not merely stale: the per-option +/// feedback it prints was written for options the student may never have seen. +/// Worth one line at the top of every report run. +fn warn_about_drift( + catalog: &coursebank::catalog::Catalog, + record: &coursebank::assessment::AssessmentFile, + quiet: bool, +) { + let Ok(Some(file)) = SealFile::find(&catalog.layout, &record.assessment.id) else { + return; + }; + let drift = file.verify(catalog, record); + let serious: Vec<&seal::Drift> = drift + .iter() + .filter(|d| d.severity >= Severity::Medium) + .collect(); + if serious.is_empty() { + return; + } + eprintln!( + "warning: {} item(s) have changed since this exam was sealed; feedback in these reports \ + may describe options the students did not see. Run `coursebank seal {} --check`", + serious.len(), + record.assessment.id + ); + if !quiet { + for finding in serious.iter().take(3) { + eprintln!(" {}", finding.message); + } + } +} diff --git a/src/commands/banks.rs b/src/commands/banks.rs index ee0b84c..256fe24 100644 --- a/src/commands/banks.rs +++ b/src/commands/banks.rs @@ -184,7 +184,7 @@ pub(crate) fn assemble(cli: &Cli, args: &AssembleArgs) -> Result { print_record(&catalog, &record); - let drift = select::check_blueprint(&record); + let drift = select::check_blueprint(&record, &catalog.course); if !drift.is_empty() { println!("\nBlueprint not fully satisfied:"); for d in &drift { diff --git a/src/commands/export.rs b/src/commands/export.rs index ef645b3..60450f2 100644 --- a/src/commands/export.rs +++ b/src/commands/export.rs @@ -12,6 +12,7 @@ use coursebank::assessment::Form; use coursebank::error::{Error, Result}; use coursebank::layout::Layout; +use coursebank::practice; use coursebank::qti; use coursebank::typst; use coursebank::yaml; @@ -31,18 +32,31 @@ pub(crate) fn export(cli: &Cli, sub: &ExportCommand) -> Result { form, out, no_feedback, + attempts: _, } => { let record = load_record(&catalog, id)?; let form = pick_form(&record, form)?; let opts = qti::QtiOptions { form: form.clone(), include_feedback: !no_feedback, - shuffle_in_canvas: record.assessment.shuffle.unwrap_or(false), - attempts: record.assessment.attempts.unwrap_or(1), + shuffle_in_canvas: record.assessment.shuffle.unwrap_or(true), + // No per-assessment attempts value falls back to unlimited, the + // same default `QtiOptions::default()` carries. Keeping this in + // step with the struct default avoids a record without an + // `attempts:` silently becoming single-attempt here while the + // library considers the default to be unlimited. + attempts: record.assessment.attempts.unwrap_or(-1), scoring_policy: record .assessment .scoring_policy .unwrap_or(coursebank::assessment::ScoringPolicy::KeepHighest), + // The remaining fields drive `assessment_meta.xml` (quiz type, + // results visibility, correct-answer display, one-question-at-a- + // time, timing, publish state). This command exposes no flags for + // them yet, so take the library defaults: a formative graded quiz + // that lets students review responses and the correct answer, and + // imports unpublished. + ..qti::QtiOptions::default() }; let package = qti::build(&catalog, &record, &opts)?; let path = out @@ -168,10 +182,119 @@ pub(crate) fn export(cli: &Cli, sub: &ExportCommand) -> Result { println!("wrote {}", path.display()); Ok(Outcome::Ok) } + ExportCommand::Practice { + id, + form, + variant, + out, + no_answer_space, + } => { + let record = load_record(&catalog, id)?; + let form = pick_form(&record, form)?; + let dir = out.clone().unwrap_or(build); + for v in pick_practice_variants(variant)? { + let opts = practice::Options { + form: form.clone(), + variant: v, + answer_space: !no_answer_space, + }; + let text = practice::render(&catalog, &record, &opts)?; + let path = dir.join(format!("{id}-{}{}.qmd", form.id, v.suffix())); + yaml::write_text(&path, &text)?; + println!("wrote {}", path.display()); + } + if !cli.quiet { + println!("\nRender with: quarto render .qmd"); + } + Ok(Outcome::Ok) + } + ExportCommand::Site { + id, + form, + out, + password, + assets, + } => { + { + use coursebank::site; + let record = load_record(&catalog, id)?; + let form = pick_form(&record, form)?; + let dir = out.clone().unwrap_or(build); + std::fs::create_dir_all(&dir)?; + + let rendered = site::render( + &catalog, + &record, + site::Options { + form, + password: password.clone(), + }, + )?; + + let qmd = dir.join("_questions.qmd"); + yaml::write_text(&qmd, &rendered.questions_qmd)?; + let json = dir.join(format!("{}-solutions.json", rendered.page)); + yaml::write_text(&json, &rendered.solutions_json)?; + + if let Some(asset_dir) = assets { + std::fs::create_dir_all(asset_dir)?; + for (name, body) in site::assets() { + let path = asset_dir.join(name); + yaml::write_text(&path, body)?; + if !cli.quiet { + println!("wrote {}", path.display()); + } + } + } + + // The password cannot be recovered from the files; always print it. + println!("password for {}: {}", rendered.page, rendered.password); + if !cli.quiet { + println!("wrote {}", qmd.display()); + println!("wrote {}", json.display()); + println!("\nInclude in the page with: {{{{< include _questions.qmd >}}}}"); + } + Ok(Outcome::Ok) + } + } } } -/// Resolves the `--variant` flags, defaulting to every document. +/// Resolves the `--variant` flags for `export practice`, defaulting to both. +/// +/// # Arguments +/// +/// * `names` - the raw flag values, possibly empty. +/// +/// # Returns +/// +/// The documents to write, deduplicated and in canonical order (worksheet first). +/// +/// # Errors +/// +/// Returns [`Error::Usage`] naming the valid tokens. +fn pick_practice_variants(names: &[String]) -> Result> { + if names.is_empty() { + return Ok(practice::Variant::ALL.to_vec()); + } + let mut wanted = Vec::new(); + for name in names { + let variant = practice::Variant::parse(name)?; + if !wanted.contains(&variant) { + wanted.push(variant); + } + } + Ok(practice::Variant::ALL + .into_iter() + .filter(|v| wanted.contains(v)) + .collect()) +} + +/// Resolves the `--variant` flags, defaulting to the exam set. +/// +/// The default is [`typst::Variant::EXAM`] rather than every variant: a +/// diagnostic is built from responses, not from an assessment record, so +/// `export typst` has nothing to build one out of. /// /// # Arguments /// @@ -187,7 +310,7 @@ pub(crate) fn export(cli: &Cli, sub: &ExportCommand) -> Result { /// Returns [`Error::Usage`] naming the valid tokens. fn pick_variants(names: &[String]) -> Result> { if names.is_empty() { - return Ok(typst::Variant::ALL.to_vec()); + return Ok(typst::Variant::EXAM.to_vec()); } let mut wanted = Vec::new(); for name in names { @@ -265,7 +388,15 @@ pub(crate) fn template(cli: &Cli, sub: &TemplateCommand) -> Result { force, stdout, } => { - let variants = pick_variants(variant)?; + // `export typst` defaults to the exam set, because a report is not + // built from an assessment record. Dumping is the opposite case: with + // no `--variant` it should hand over every template there is, + // including the two reports. + let variants = if variant.is_empty() { + typst::Variant::ALL.to_vec() + } else { + pick_variants(variant)? + }; if *stdout { for (index, v) in variants.iter().enumerate() { diff --git a/src/commands/handlers.rs b/src/commands/handlers.rs new file mode 100644 index 0000000..72ebe0c --- /dev/null +++ b/src/commands/handlers.rs @@ -0,0 +1,558 @@ +// SPDX-License-Identifier: Prosperity-3.0.0 +// Copyright Scientific Computing Studio +// Source: https://git.scient.ing/education/coursebank + +//! Replacement handlers for `src/commands/analysis.rs`. +//! +//! This file is not a module of its own: `seal` is new, and `ingest` and `report` +//! replace the functions of the same name in `commands/analysis.rs`. Paste them +//! in there, add the imports listed at the top, and delete this file. It is kept +//! separate here only so the diff against the existing file is obvious. +//! +//! Imports `commands/analysis.rs` needs on top of what it already has: +//! +//! ```ignore +//! use std::collections::BTreeMap; +//! +//! use coursebank::decode::Numbering; +//! use coursebank::diagnostic; +//! use coursebank::intake::{self, Source}; +//! use coursebank::seal::{self, SealFile}; +//! use coursebank::typst::{self, Variant}; +//! use coursebank::typst::diagnostic as typst_diagnostic; +//! +//! use crate::cli::SealArgs; +//! ``` + +use std::collections::BTreeMap; + +use coursebank::canvas; +use coursebank::catalog::Severity; +use coursebank::classical::{self, Thresholds}; +use coursebank::decode::Numbering; +use coursebank::diagnostic; +use coursebank::error::{Error, Result}; +use coursebank::intake::{self, Source}; +use coursebank::irt; +use coursebank::report; +use coursebank::seal::{self, SealFile}; +use coursebank::store::Store; +use coursebank::students; +use coursebank::typst::diagnostic as typst_diagnostic; +use coursebank::typst::{self, Variant}; +use coursebank::yaml; + +use crate::cli::{Cli, IngestCommand, ReportCommand, SealArgs}; +use crate::commands::Outcome; +use crate::helpers::{context, load, load_record, read_salt, responses_for}; + +/// `seal`: freeze what was administered, or check the freeze. +/// +/// Write the seal after exporting the papers and before printing them. It records +/// the stem and options of every item, the printed-letter map for every form, and +/// digests over both, so that the question "is the bank still what the students +/// saw?" has an answer six months from now. +pub(crate) fn seal(cli: &Cli, args: &SealArgs) -> Result { + let catalog = load(cli)?; + let record = load_record(&catalog, &args.id)?; + let path = args + .out + .clone() + .unwrap_or_else(|| SealFile::path(&catalog.layout, &record.assessment.id)); + + if args.check { + let existing = SealFile::load(&path).map_err(|_| { + Error::usage(format!( + "no seal at {}. Write one with `coursebank seal {}`", + path.display(), + args.id + )) + })?; + + let drift = existing.verify(&catalog, &record); + if drift.is_empty() { + println!( + "{} matches the seal written {} ({})", + args.id, + existing.seal.sealed_on, + seal::short(&existing.seal.digest) + ); + return Ok(Outcome::Ok); + } + + println!( + "{} finding(s) against the seal written {}:\n", + drift.len(), + existing.seal.sealed_on + ); + for finding in &drift { + println!(" {:<6} {}", finding.severity.label(), finding.message); + } + if drift.iter().any(|d| d.is_blocking()) { + println!( + "\nAnalysis that pools this administration with another is comparing two \ + different questions. Per-option feedback in student reports may name the wrong \ + option." + ); + } + return Ok(Outcome::Findings); + } + + if path.exists() && !args.force { + return Err(Error::usage(format!( + "{} already exists. A seal is meant to be written once, before the exam is printed; \ + pass --force only if you are re-sealing an assessment that was never administered", + path.display() + ))); + } + + let opts = seal::Options { + content: !args.no_content, + forms: args.form.clone(), + }; + let file = seal::build(&catalog, &record, &opts)?; + file.save(&path)?; + + println!( + "wrote {} ({} item(s), {} form(s), {})", + path.display(), + file.items.len(), + file.forms.len(), + seal::short(&file.seal.digest) + ); + let mut advisories = Vec::new(); + for form in &file.forms { + let permuted = form + .questions + .iter() + .filter(|q| q.options.iter().any(|o| o.printed != o.canonical)) + .count(); + println!( + " form {}: {} question(s), {} with permuted options, {}", + form.id, + form.questions.len(), + permuted, + seal::short(&form.digest) + ); + advisories.extend(seal::balance(form).notes()); + } + + // The last moment before printing is the only cheap moment to notice that a + // shuffle produced a sequence a student will read as a mistake. + if !advisories.is_empty() { + println!(); + for note in &advisories { + println!("! {note}"); + } + println!( + "\nRe-seed a form by editing its `seed:` in the record, then re-export and re-seal \ + with --force. Nothing else has to change." + ); + } + + if !cli.quiet { + println!( + "\nCommit this file. Check it any time with: coursebank seal {} --check", + args.id + ); + } + Ok(Outcome::Ok) +} + +/// `ingest`: read grading exports into the response store. +/// +/// The Gradescope path takes one directory per form and merges them into a single +/// administration, translating each form's printed letters into the bank's +/// lettering on the way in. See [`coursebank::intake`]. +pub(crate) fn ingest(cli: &Cli, sub: &IngestCommand) -> Result { + let catalog = load(cli)?; + + let (common, mut set) = match sub { + IngestCommand::Gradescope { + sources, + common, + allow_mismatch, + recorded_numbers, + } => { + let record = load_record(&catalog, &common.assessment)?; + let sealed = SealFile::find(&catalog.layout, &record.assessment.id)?; + + if let Some(file) = &sealed { + let drift = file.verify(&catalog, &record); + let blocking: Vec<&seal::Drift> = + drift.iter().filter(|d| d.is_blocking()).collect(); + for finding in &drift { + println!("! {} {}", finding.severity.label(), finding.message); + } + if !blocking.is_empty() { + println!( + "\nThe seal still describes the papers the students held, so ingest will \ + use it. The bank has moved since; `coursebank seal {} --check` lists \ + what.", + common.assessment + ); + } + } else if !cli.quiet { + println!( + "! no seal for {}; the option maps will be derived from the record as it \ + stands today. Write one next time with `coursebank seal {}` before printing", + common.assessment, common.assessment + ); + } + + let mut parsed = Vec::new(); + for text in sources { + parsed.push(Source::parse(text, common.form.as_deref())?); + } + for source in &parsed { + if !intake::looks_like_export(&source.dir) { + return Err(Error::usage(format!( + "{} holds no files named `1.csv` … `N.csv`. Gradescope writes one file \ + per question; point at the directory those were unzipped into", + source.dir.display() + ))); + } + } + + let date = match &common.date { + Some(text) => Some(text.parse()?), + None => None, + }; + let opts = intake::Options { + date, + numbering: if *recorded_numbers { + Numbering::Recorded + } else { + Numbering::Printed + }, + strict: !*allow_mismatch, + }; + + let result = intake::run(&catalog, &record, sealed.as_ref(), &parsed, &opts)?; + + for form in &result.forms { + println!("{}", intake::describe(form)); + } + + // Grading-time partial credit is an ambiguity signal worth surfacing + // while the exam is fresh, and it is per form because a regrade + // applied to one form and not the other is its own problem. + for (form, question) in result.questions() { + for (letter, value, note) in question.partial_credit() { + println!( + "! form {form} q{}: option {letter} earned {value} of {} points at \ + grading time{}", + question.number, + question.points_possible(), + note.map(|n| format!(" — {n}")).unwrap_or_default() + ); + } + } + + (common, result.responses) + } + IngestCommand::Canvas { file, common } => { + let record = load_record(&catalog, &common.assessment)?; + let ctx = context(&catalog, &record, common)?; + let set = canvas::ingest(file, &ctx, Some(&record), Some(&catalog))?; + (common, set) + } + }; + + let record = load_record(&catalog, &common.assessment)?; + set.enrich(&record, Some(&catalog)); + + if common.pseudonymize { + let salt = read_salt(common.salt_file.as_deref())?; + set.pseudonymize(&salt); + println!("identifiers replaced with keyed pseudonyms"); + } + + for warning in &set.warnings { + println!("! {warning}"); + } + + println!( + "\n{} response(s): {} student(s) x {} item(s)", + set.rows.len(), + set.students().len(), + set.all_items().len() + ); + + if common.dry_run { + println!("(dry run, nothing written)"); + return Ok(Outcome::Ok); + } + + let mut store = Store::open(catalog.layout.data())?; + if let Some(format) = common.format { + store = store.with_format(format.as_format())?; + } + for path in store.write(&set)? { + println!("wrote {}", path.display()); + } + println!( + "\nNext: coursebank analyze items {}\n coursebank report cohort {} --typst", + common.assessment, common.assessment + ); + Ok(Outcome::Ok) +} + +/// `report`: write per-student diagnostics or the class diagnostic. +/// +/// Markdown is still the default for both, because it diffs and reads in a +/// terminal. `--typst` adds a document per student, or one for the class, built +/// from the templates in `templates/` and compiled with `typst compile`. +pub(crate) fn report(cli: &Cli, sub: &ReportCommand) -> Result { + let catalog = load(cli)?; + let store = Store::open(catalog.layout.data())?; + + match sub { + ReportCommand::Students { + id, + html, + out, + ability, + no_comparison, + typst: want_typst, + no_markdown, + no_questions, + no_feedback, + template, + json, + } => { + let record = load_record(&catalog, id)?; + let set = responses_for(&store, &catalog, &record, false)?; + let fit = if *ability { + Some(irt::fit(&set.matrix(false), &irt::Options::default())) + } else { + None + }; + let cohort = students::summarize(&set, &catalog.course, Some(&catalog), fit.as_ref()); + let dir = out + .clone() + .unwrap_or_else(|| catalog.layout.reports().join(id)); + + warn_about_drift(&catalog, &record, cli.quiet); + + if !no_markdown { + let opts = report::StudentOptions { + ability: *ability, + comparison: !no_comparison, + ..report::StudentOptions::default() + }; + let written = report::write_all_students( + &dir, + &cohort, + &catalog.course, + &record, + &opts, + *html, + )?; + println!( + "wrote {} Markdown file(s) for {} student(s) in {}", + written.len(), + cohort.students.len(), + dir.display() + ); + } + + if !want_typst { + return Ok(Outcome::Ok); + } + + // The item analysis supplies the class rate per question, which is + // what makes "you missed q17, and so did most of the class" possible. + let analysis = + classical::analyze(&set, &Thresholds::default(), Some(&record), Some(&catalog)); + + let config = typst::load_config(&catalog.layout)?.resolve(Variant::StudentReport); + let meta = typst_diagnostic::Meta::new(&catalog, &record, cohort.students.len()); + let opts = diagnostic::Options { + comparison: !no_comparison, + questions: !no_questions, + feedback: !no_feedback, + ability: *ability, + ..diagnostic::Options::default() + }; + + let mut written = 0usize; + let mut used_embedded = false; + for summary in &cohort.students { + let built = diagnostic::student( + summary, + &cohort, + &catalog, + &set, + Some(&analysis), + &opts, + ); + let document = typst_diagnostic::render_student( + &catalog.layout, + &meta, + &built, + &config, + template.as_deref(), + )?; + used_embedded |= document.origin == typst::Origin::Embedded; + for warning in &document.warnings { + eprintln!("warning: {warning}"); + } + + let stem = typst_diagnostic::student_stem(id, &summary.student_key); + let path = dir.join(format!("{stem}.typ")); + yaml::write_text(&path, &document.text)?; + written += 1; + + if *json { + yaml::write_json(&dir.join(format!("{stem}.json")), &built)?; + } + } + + println!( + "wrote {}", + typst_diagnostic::summary(written, cohort.students.len()) + ); + if !cli.quiet { + println!( + "Compile them all with:\n for f in {}/*.typ; do typst compile \"$f\"; done", + dir.display() + ); + if used_embedded { + println!( + "These used the built-in template. To take over the layout:\n \ + coursebank template dump --variant student-report" + ); + } + } + Ok(Outcome::Ok) + } + + ReportCommand::Cohort { + id, + html, + out, + typst: want_typst, + no_markdown, + template, + json, + } => { + let record = load_record(&catalog, id)?; + let set = responses_for(&store, &catalog, &record, false)?; + let analysis = + classical::analyze(&set, &Thresholds::default(), Some(&record), Some(&catalog)); + let fit = irt::fit(&set.matrix(false), &irt::Options::default()); + let cohort = students::summarize(&set, &catalog.course, Some(&catalog), Some(&fit)); + + warn_about_drift(&catalog, &record, cli.quiet); + + let path = out + .clone() + .unwrap_or_else(|| catalog.layout.reports().join(format!("{id}-cohort.md"))); + + if !no_markdown { + let markdown = report::cohort(&analysis, &cohort, &catalog, &record, Some(&fit)); + yaml::write_text(&path, &markdown)?; + println!("wrote {}", path.display()); + + if *html { + let html_path = path.with_extension("html"); + let title = format!("{} — item analysis", record.assessment.title); + yaml::write_text(&html_path, &report::to_html(&markdown, &title))?; + println!("wrote {}", html_path.display()); + } + } + + let built = diagnostic::cohort( + &analysis, + &cohort, + &catalog, + &record, + &set, + Some(&fit), + ); + + for line in typst_diagnostic::headline(&built) { + println!("{line}"); + } + + if *want_typst { + let config = typst::load_config(&catalog.layout)?.resolve(Variant::CohortReport); + let meta = typst_diagnostic::Meta::new(&catalog, &record, cohort.students.len()); + let document = typst_diagnostic::render_cohort( + &catalog.layout, + &meta, + &built, + &config, + template.as_deref(), + )?; + for warning in &document.warnings { + eprintln!("warning: {warning}"); + } + let typst_path = path.with_extension("typ"); + yaml::write_text(&typst_path, &document.text)?; + println!("wrote {} (from {})", typst_path.display(), document.origin); + + if *json { + yaml::write_json(&path.with_extension("json"), &built)?; + } + } + + Ok(if built.revise.is_empty() { + Outcome::Ok + } else { + Outcome::Findings + }) + } + } +} + +/// Says so when the bank has moved since the exam was sealed. +/// +/// A report built from a drifted bank is not merely stale: the per-option +/// feedback it prints was written for options the student may never have seen. +/// Worth one line at the top of every report run. +fn warn_about_drift( + catalog: &coursebank::catalog::Catalog, + record: &coursebank::assessment::AssessmentFile, + quiet: bool, +) { + let Ok(Some(file)) = SealFile::find(&catalog.layout, &record.assessment.id) else { + return; + }; + let drift = file.verify(catalog, record); + let serious: Vec<&seal::Drift> = drift + .iter() + .filter(|d| d.severity >= Severity::Medium) + .collect(); + if serious.is_empty() { + return; + } + eprintln!( + "warning: {} item(s) have changed since this exam was sealed; feedback in these reports \ + may describe options the students did not see. Run `coursebank seal {} --check`", + serious.len(), + record.assessment.id + ); + if !quiet { + for finding in serious.iter().take(3) { + eprintln!(" {}", finding.message); + } + } +} + +/// Counts how many students each form was given to, for the ingest summary. +/// +/// Kept here rather than in the library because it exists to print a line. +#[allow(dead_code)] +fn students_per_form(set: &coursebank::responses::ResponseSet) -> BTreeMap { + let mut seen: BTreeMap> = BTreeMap::new(); + for row in &set.rows { + if let Some(form) = row.form.as_deref() { + seen.entry(form.to_string()) + .or_default() + .insert(row.student_key.as_str()); + } + } + seen.into_iter().map(|(k, v)| (k, v.len())).collect() +} diff --git a/src/commands/lectures.rs b/src/commands/lectures.rs new file mode 100644 index 0000000..9c67b05 --- /dev/null +++ b/src/commands/lectures.rs @@ -0,0 +1,110 @@ +// SPDX-License-Identifier: Prosperity-3.0.0 +// Copyright Scientific Computing Studio +// Source: https://git.scient.ing/education/coursebank + +//! Rendering lecture pages, and checking what backs each learning target. +//! +//! Both handlers here read the course file and nothing else, so neither needs a +//! bank or a single response. That is deliberate: a reading list is useful in week +//! one, before any item exists. + +use coursebank::course::CourseFile; +use coursebank::error::Result; +use coursebank::lecture::{objectives_markdown, readings_markdown}; +use coursebank::yaml; + +use crate::cli::{Cli, LectureCommand}; +use crate::commands::Outcome; + +/// `lecture`: render a reading list, or report reading coverage. +pub(crate) fn lecture(cli: &Cli, sub: &LectureCommand) -> Result { + let course = CourseFile::load_dir(&cli.course)?; + + match sub { + LectureCommand::Readings { id, style, out } => emit( + readings_markdown(&course, id, style.as_style())?, + out.as_deref(), + ), + LectureCommand::Objectives { id, style, out } => emit( + objectives_markdown(&course, id, style.as_style())?, + out.as_deref(), + ), + LectureCommand::Coverage { lecture: only } => coverage(&course, only.as_deref(), cli.quiet), + } +} + +/// Writes rendered Markdown to a file, or to stdout when no path was given. +fn emit(markdown: String, out: Option<&std::path::Path>) -> Result { + match out { + Some(path) => { + yaml::write_text(path, &markdown)?; + println!("wrote {}", path.display()); + } + None => print!("{markdown}"), + } + Ok(Outcome::Ok) +} + +/// Prints the readings behind each learning target. +/// +/// Targets rather than objectives, because that is the tier a reading is cited +/// against: a section of a book backs a specific performance. Returns +/// [`Outcome::Findings`] when an assessed target has no reading, since that is +/// the case where a student report can name what was missed but not where to go +/// and read about it. +fn coverage(course: &CourseFile, lecture: Option<&str>, quiet: bool) -> Result { + let ids: Vec = match lecture { + Some(l) => course + .lecture_targets(l) + .into_iter() + .map(str::to_string) + .collect(), + None => course.targets_in_order(), + }; + + for id in &ids { + let readings = course.readings_for_objective(id); + println!("{id}"); + if readings.is_empty() { + println!(" (no reading)"); + continue; + } + for (lecture_id, reading) in readings { + let Some(key) = reading.reference.as_deref() else { + continue; + }; + let reference = course.reference(key, lecture_id)?; + let supplemental = match reading.role { + coursebank::course::ReadingRole::Supplemental => " (supplemental)", + coursebank::course::ReadingRole::Assigned => "", + }; + println!( + " {lecture_id} {}{supplemental}", + reading.cite(key, reference) + ); + if let Some(focus) = &reading.focus { + println!( + " {}", + course.expand_objective_refs(focus, |id| { course.text_for(id) }) + ); + } + } + } + + let gaps = course.targets_without_readings(); + if gaps.is_empty() { + if !quiet { + println!("\nevery assessed target has a reading behind it"); + } + return Ok(Outcome::Ok); + } + println!( + "\n{} assessed target(s) with no reading, so a student report cannot say \ + where to go back to:", + gaps.len() + ); + for id in gaps { + println!(" - {id}"); + } + Ok(Outcome::Findings) +} diff --git a/src/commands/project.rs b/src/commands/project.rs index f7deb8b..03088f7 100644 --- a/src/commands/project.rs +++ b/src/commands/project.rs @@ -5,32 +5,45 @@ //! Setting up a course and checking it stays well-formed. //! //! These are the commands you reach for before and around authoring: create the -//! directory (`init`), write editor schemas (`schema`), and run the two kinds of -//! checking — [`validate`] for problems that must be fixed and [`lint`] for -//! item-writing guidance. [`catalog`] summarizes the pool that results. +//! directory (`init`), see and split the course file (`course`), export the +//! bibliography (`references`), write editor schemas (`schema`), and run the two +//! kinds of checking — [`validate`] for problems that must be fixed and [`lint`] +//! for item-writing guidance. [`catalog`] summarizes the pool that results. -use std::collections::BTreeMap; +use std::collections::{BTreeMap, BTreeSet}; +use std::fs; +use std::path::Path; use coursebank::assessment::AssessmentFile; use coursebank::bank::BankFile; +use coursebank::course::fragment::{self, Section}; use coursebank::course::{COURSE_FILE, CourseFile}; use coursebank::error::{Error, Result}; use coursebank::jsonschema; use coursebank::layout::Layout; use coursebank::lint::{self, Rule}; -use coursebank::taxonomy::Level; +use coursebank::migrate; +use coursebank::references; +use coursebank::taxonomy::{Level, Tier}; use coursebank::yaml; -use crate::cli::{CatalogArgs, Cli, InitArgs, LintArgs}; +use crate::cli::{ + CatalogArgs, Cli, CourseCommand, InitArgs, LintArgs, MigrateCommand, ReferencesCommand, +}; use crate::commands::Outcome; use crate::helpers::{load, truncate}; /// The `.gitignore` written by `init`. pub(crate) const GITIGNORE: &str = "\ -# Generated output: exports, rendered exams, reports. +# Generated output: exports, rendered exams, reports. A student report carries +# names, so it belongs here rather than in the repository. build/ reports/ +# Response data: every row carries a student. The statistics derived from it are +# cohort aggregates and live in analysis/, which is committed on purpose. +data/ + # Typst and PDF artifacts. *.pdf @@ -44,6 +57,9 @@ reports/ *.swp "; +/// Marks the block `init` prepends to a `.gitignore` that was already there. +const GITIGNORE_HEADER: &str = "# Added by coursebank init."; + /// `init`: create a new course directory, refusing to clobber an existing one. pub(crate) fn init(cli: &Cli, args: &InitArgs) -> Result { let layout = Layout::new(&cli.course); @@ -70,15 +86,579 @@ pub(crate) fn init(cli: &Cli, args: &InitArgs) -> Result { println!("wrote {}", bank_path.display()); } - yaml::write_text(&cli.course.join(".gitignore"), GITIGNORE)?; + write_gitignore(&cli.course.join(".gitignore"))?; println!( - "\nNext: edit {} to add your learning objectives and lectures, then\n \ - coursebank bank new unit-1 --title \"Unit 1\"\n coursebank validate", - COURSE_FILE + "\nNext: edit {COURSE_FILE} to add your learning objectives, their targets, and\n \ + your lectures, then\n coursebank bank new unit-1 --title \"Unit 1\"\n \ + coursebank validate\n\nOnce {COURSE_FILE} is more than you want to scroll, \ + `coursebank migrate split`\n moves each lecture and objective into its own file under \ + lectures/ and\n objectives/, and every command goes on reading the course as one." ); Ok(Outcome::Ok) } +/// `course`: inspect the course file, or split it into fragments. +pub(crate) fn course(cli: &Cli, sub: &CourseCommand) -> Result { + match sub { + CourseCommand::Files => course_files(cli), + CourseCommand::Build { out } => course_build(cli, out.as_deref()), + CourseCommand::Where { id } => course_where(cli, id), + } +} + +/// `migrate`: the one-time layout conversions. +pub(crate) fn migrate(cli: &Cli, sub: &MigrateCommand) -> Result { + match sub { + MigrateCommand::Split { dry_run } => course_split(cli, *dry_run), + MigrateCommand::Ids { dry_run } => migrate_ids(cli, *dry_run), + MigrateCommand::Options { dry_run } => migrate_options(cli, *dry_run), + MigrateCommand::Stems { dry_run } => migrate_stems(cli, *dry_run), + MigrateCommand::Variants { dry_run } => migrate_variants(cli, *dry_run), + MigrateCommand::References { dry_run } => migrate_references(cli, *dry_run), + MigrateCommand::Order { dry_run } => migrate_order(cli, *dry_run), + MigrateCommand::Counters { dry_run } => migrate_counters(cli, *dry_run), + } +} + +/// Drops the trailing counter from every item id. +fn migrate_counters(cli: &Cli, dry_run: bool) -> Result { + let (rename, touched) = migrate::counters(&cli.course, !dry_run)?; + if rename.is_empty() { + println!("nothing to do: no item id ends in a counter"); + return Ok(Outcome::Ok); + } + + for (old, new) in &rename { + println!(" {old} -> {new}"); + } + println!(); + for (path, n) in &touched { + println!(" {:<44} {n:>5} reference(s)", path.display()); + } + if dry_run { + println!("\nnothing written"); + return Ok(Outcome::Ok); + } + + let sealed = touched + .iter() + .filter(|(p, _)| p.starts_with("seals")) + .count(); + println!( + "\nrenamed {} item(s) across {} file(s)", + rename.len(), + touched.len() + ); + if sealed > 0 { + println!( + "\n{sealed} seal(s) were rewritten. The ids are inside the digest, so each one was \ + recomputed\n and the previous digest recorded under `superseded_digests`. A seal \ + that has been\n rewritten says so rather than looking untouched." + ); + } + println!( + "\nNext:\n coursebank validate\n coursebank seal verify\n coursebank analyze items \ + --all # the join key moved; check the data still lands" + ); + Ok(Outcome::Ok) +} + +/// Replaces the order integers with ordered declarations. +fn migrate_order(cli: &Cli, dry_run: bool) -> Result { + let touched = migrate::order(&cli.course, !dry_run)?; + if touched.is_empty() { + println!("nothing to do: no `order:` left to derive"); + return Ok(Outcome::Ok); + } + let total: usize = touched.iter().map(|(_, n)| n).sum(); + for (path, n) in &touched { + println!(" {:<44} {n:>5} order(s) dropped", path.display()); + } + if dry_run { + println!("\nnothing written"); + return Ok(Outcome::Ok); + } + println!( + "\nrewrote {} file(s), {total} integer(s) gone\n\nNext:\n coursebank validate\n \ + coursebank lecture objectives L1.2 # check the order still reads right", + touched.len() + ); + Ok(Outcome::Ok) +} + +/// Takes the citations out of the notes and puts them in fields. +fn migrate_references(cli: &Cli, dry_run: bool) -> Result { + let (touched, notes) = migrate::references(&cli.course, !dry_run)?; + + for (path, n) in &touched { + println!(" {:<44} {n:>5} note(s) taken apart", path.display()); + } + for note in ¬es { + println!(" note: {note}"); + } + if touched.is_empty() { + println!("nothing to do: no note is carrying a citation"); + return Ok(Outcome::Ok); + } + if dry_run { + println!("\nnothing written"); + return Ok(Outcome::Ok); + } + println!( + "\nrewrote {} file(s)\n\nNext:\n coursebank validate\n coursebank references list\n\n\ + An issue number is never inferred, not even from a DOI that encodes one, so add those \ + by hand.", + touched.len() + ); + Ok(Outcome::Ok) +} + +/// Fills in the stored variant column. +fn migrate_variants(cli: &Cli, dry_run: bool) -> Result { + let touched = migrate::store_variants(&cli.course, !dry_run)?; + if touched.is_empty() { + println!("nothing to fill in: every stored row already names its variant"); + return Ok(Outcome::Ok); + } + for (path, n) in &touched { + println!(" {:<44} {n:>5} row(s)", path.display()); + } + if dry_run { + println!("\nnothing written"); + return Ok(Outcome::Ok); + } + println!("\nrewrote {} data file(s)", touched.len()); + Ok(Outcome::Ok) +} + +/// Drops the version fields 2.0 ignores. +fn migrate_stems(cli: &Cli, dry_run: bool) -> Result { + let touched = migrate::stems(&cli.course, !dry_run)?; + if touched.is_empty() { + println!("nothing to migrate: no `version:` or `history:` left to drop"); + return Ok(Outcome::Ok); + } + for (path, n) in &touched { + println!(" {:<44} {n:>5} line(s) dropped", path.display()); + } + if dry_run { + println!("\nnothing written"); + return Ok(Outcome::Ok); + } + println!( + "\nrewrote {} file(s)\n\nNext:\n coursebank validate\n\nFrom here, rewording a stem \ + is an error rather than a version bump: give the new\n wording a new id and \ + `supersedes:` the old one.", + touched.len() + ); + Ok(Outcome::Ok) +} + +/// Rewrites option letters as names, showing every name before writing. +fn migrate_options(cli: &Cli, dry_run: bool) -> Result { + let (map, problems) = migrate::options_plan(&cli.course)?; + + for (item, options) in &map { + println!("{item}"); + for (letter, name) in options { + println!(" {letter} -> {name}"); + } + } + + if !problems.is_empty() { + println!("\n{} item(s) need naming by hand:", problems.len()); + for problem in &problems { + println!(" - {problem}"); + } + } + if map.is_empty() { + println!("nothing to migrate: every option is already named"); + return Ok(Outcome::Ok); + } + + let touched = migrate::apply_options(&cli.course, &map, !dry_run)?; + println!(); + for (path, n) in &touched { + println!(" {:<44} {n:>5} rename(s)", path.display()); + } + if dry_run { + println!("\nnothing written"); + return Ok(if problems.is_empty() { + Outcome::Ok + } else { + Outcome::Findings + }); + } + println!( + "\nrewrote {} file(s)\n\nSeals keep their letters on purpose; see `coursebank migrate \ + --help`.\nNext:\n coursebank validate\n coursebank lint", + touched.len() + ); + Ok(if problems.is_empty() { + Outcome::Ok + } else { + Outcome::Findings + }) +} + +/// Rewrites pre-2.0 bank-qualified item ids everywhere they are stored. +fn migrate_ids(cli: &Cli, dry_run: bool) -> Result { + let files = migrate::qualified_ids(&cli.course)?; + for (path, n, _) in &files { + println!(" {:<44} {n:>5} id(s)", path.display()); + } + if !dry_run { + migrate::apply_ids(&cli.course, &files)?; + } + + let data = migrate::store_ids(&cli.course, !dry_run)?; + for (path, n) in &data { + println!(" {:<44} {n:>5} row(s)", path.display()); + } + + if files.is_empty() && data.is_empty() { + println!("nothing to migrate: every item id already names the item course-wide"); + return Ok(Outcome::Ok); + } + if dry_run { + println!("\nnothing written"); + return Ok(Outcome::Ok); + } + println!( + "\nrewrote {} file(s) and {} data file(s)\n\nNext:\n coursebank validate\n \ + coursebank analyze items --all", + files.len(), + data.len() + ); + Ok(Outcome::Ok) +} + +/// Lists the fragments a course is assembled from, with what each defines. +fn course_files(cli: &Cli) -> Result { + let layout = Layout::new(&cli.course); + let course = CourseFile::load_dir(&cli.course)?; + + for (path, role) in fragment::files(&layout)? { + if !path.exists() { + continue; + } + let shown = path.strip_prefix(&cli.course).unwrap_or(&path); + let mut defines: Vec = Vec::new(); + for section in Section::ALL { + let n = course + .origins + .iter() + .filter(|((s, _), p)| *s == section && p.as_path() == shown) + .count(); + if n == 0 { + continue; + } + defines.push(match section { + // These are declared once for the whole course, so a count + // would always be 1 and would read as though it could be more. + Section::Course | Section::Policy => section.key().to_string(), + _ => format!("{n} {}", section.key()), + }); + } + println!( + "{:<40} {:<11} {}", + shown.display(), + role.label(), + defines.join(", ") + ); + } + Ok(Outcome::Ok) +} + +/// Prints or writes the merged course. +fn course_build(cli: &Cli, out: Option<&Path>) -> Result { + let course = CourseFile::load_dir(&cli.course)?; + match out { + Some(path) => { + course.write_resolved(path)?; + println!( + "wrote {} from {} file(s)", + path.display(), + course.fragment_paths().len() + ); + } + None => print!("{}", yaml::to_string(&course)?), + } + Ok(Outcome::Ok) +} + +/// Says which file defines an id. +fn course_where(cli: &Cli, id: &str) -> Result { + let course = CourseFile::load_dir(&cli.course)?; + match course.origin(id) { + Some((section, path)) => { + println!( + "{} defines `{id}` under `{}`", + path.display(), + section.key() + ); + Ok(Outcome::Ok) + } + None => Err(Error::Unresolved { + kind: "id", + id: id.to_string(), + context: Some(cli.course.display().to_string()), + }), + } +} + +/// Splits `course.yaml` into fragments, then checks the result reassembles. +fn course_split(cli: &Cli, dry_run: bool) -> Result { + let plan = migrate::split_plan(&cli.course)?; + + println!("{} file(s):", plan.files.len()); + for (path, lines) in plan.lines() { + let summary = plan + .files + .iter() + .find(|f| f.path == path) + .map(|f| f.summary.clone()) + .unwrap_or_default(); + println!(" {:<44} {lines:>5} lines {summary}", path.display()); + } + for note in &plan.notes { + println!("\nnote: {note}"); + } + + if dry_run { + println!("\nnothing written"); + return Ok(Outcome::Ok); + } + + // Loaded before anything is written, since it is the thing the result is + // checked against. + let before = CourseFile::load(&Layout::new(&cli.course).course_file())?; + let written = migrate::apply(&cli.course, &plan)?; + println!("\nwrote {} file(s)", written.len()); + + let after = CourseFile::load_dir(&cli.course)?; + let diffs = migrate::differences(&before, &after)?; + if diffs.is_empty() { + println!( + "reassembled and compared against {COURSE_FILE}.bak: identical\n\nNext:\n \ + coursebank validate\n coursebank schema\n git add -A && git diff --cached --stat" + ); + return Ok(Outcome::Ok); + } + + println!( + "\n{} difference(s) between the original and the reassembled course:", + diffs.len() + ); + for diff in &diffs { + println!(" - {diff}"); + } + println!( + "\nThe original is at {COURSE_FILE}.bak. Restore it with\n mv {COURSE_FILE}.bak \ + {COURSE_FILE} && rm -r lectures objectives {}", + fragment::REFERENCES_FILE + ); + Ok(Outcome::Findings) +} + +/// `references`: list the bibliography, or export it in a citation format. +pub(crate) fn references(cli: &Cli, sub: &ReferencesCommand) -> Result { + let course = CourseFile::load_dir(&cli.course)?; + + match sub { + ReferencesCommand::List => { + let mut unreachable = 0; + for (key, reference) in &course.references { + let link = match reference.href(None, None) { + Some(url) => url, + None => { + unreachable += 1; + "(no link)".to_string() + } + }; + println!( + "{:<28} {:<10} {:<9} {:<6} {link}", + truncate(key, 27), + format!("{:?}", reference.kind).to_lowercase(), + format!("{:?}", reference.role).to_lowercase(), + reference.label_or("-"), + ); + } + if unreachable > 0 && !cli.quiet { + println!( + "\n{unreachable} work(s) with no link. A student report can name one but \ + cannot send anyone to it; for a manuscript, add its `doi`." + ); + } + Ok(Outcome::Ok) + } + ReferencesCommand::Export { format, out } => { + let format = format.as_format(); + let text = references::render(&course, format)?; + match out { + Some(path) => { + yaml::write_text(path, &text)?; + println!( + "wrote {} ({} work(s) as {})", + path.display(), + course.references.len(), + format.label() + ); + } + None => print!("{text}"), + } + Ok(Outcome::Ok) + } + } +} + +/// What reconciling [`GITIGNORE`] against a file already on disk would do. +struct GitignoreMerge { + /// The file to write. Identical to the input when nothing was missing. + text: String, + /// The patterns that were missing, in the order [`GITIGNORE`] lists them. + added: Vec, + /// Patterns the file un-ignores with a `!` rule, which are left out. Git + /// applies the last matching rule, so a line added at the top would lose. + negated: Vec, +} + +/// The comparison key for one `.gitignore` line, or `None` for a blank or comment. +/// +/// `build`, `build/`, and `/build/` are one pattern spelled three ways, so the key +/// drops the slashes. A leading `!` stays, because `!build` is the opposite of +/// `build` rather than a restatement of it. +fn pattern_key(line: &str) -> Option { + let trimmed = line.trim(); + if trimmed.is_empty() || trimmed.starts_with('#') { + return None; + } + let (bang, rest) = match trimmed.strip_prefix('!') { + Some(rest) => ("!", rest.trim_start()), + None => ("", trimmed), + }; + let rest = rest.trim_start_matches('/').trim_end_matches('/'); + if rest.is_empty() { + return None; + } + Some(format!("{bang}{rest}")) +} + +/// Reconciles [`GITIGNORE`] against a `.gitignore` that is already on disk. +/// +/// Patterns the file already has are skipped, and a block of [`GITIGNORE`] left +/// with no patterns loses its comment too, so nobody ends up with a heading over +/// nothing. What survives goes above the existing content, which is copied +/// through unchanged, including its line endings. +fn merge_gitignore(existing: &str) -> GitignoreMerge { + let keys: BTreeSet = existing.lines().filter_map(pattern_key).collect(); + let crlf = existing.contains("\r\n"); + let newline = if crlf { "\r\n" } else { "\n" }; + + let mut block: Vec<&str> = Vec::new(); + let mut pending: Vec<&str> = Vec::new(); + let mut added: Vec = Vec::new(); + let mut negated: Vec = Vec::new(); + let mut kept_in_block = false; + + for line in GITIGNORE.lines() { + let trimmed = line.trim(); + if trimmed.is_empty() { + // A blank line starts a new block, so any comment still waiting for a + // pattern belonged to a block that was dropped entirely. + pending.clear(); + kept_in_block = false; + continue; + } + if trimmed.starts_with('#') { + pending.push(trimmed); + continue; + } + let Some(key) = pattern_key(trimmed) else { + continue; + }; + if keys.contains(&key) { + continue; + } + if keys.contains(&format!("!{key}")) { + negated.push(trimmed.to_string()); + continue; + } + if !kept_in_block && !block.is_empty() { + block.push(""); + } + block.append(&mut pending); + kept_in_block = true; + block.push(trimmed); + added.push(trimmed.to_string()); + } + + let mut text = String::new(); + if !block.is_empty() { + for line in std::iter::once(GITIGNORE_HEADER).chain(block).chain([""]) { + text.push_str(line); + text.push_str(newline); + } + } + text.push_str(existing); + + GitignoreMerge { + text, + added, + negated, + } +} + +/// Writes the `.gitignore`, merging into one that is already there. +/// +/// `init` is often run in a repository that already has a `.gitignore`, and the +/// first version of this overwrote it. An existing file now keeps everything it +/// had and gains only the patterns it was missing, at the top where they are easy +/// to see in the diff. +fn write_gitignore(path: &Path) -> Result<()> { + let existing = match fs::read_to_string(path) { + Ok(text) => text, + // Nothing to merge with. Stay quiet about it, the way this always has. + Err(e) if e.kind() == std::io::ErrorKind::NotFound => { + return yaml::write_text(path, GITIGNORE); + } + // A file that is there but unreadable, or not UTF-8, is not one to + // replace on a guess. + Err(e) => return Err(Error::io(path, e)), + }; + + let merge = merge_gitignore(&existing); + if merge.added.is_empty() { + println!( + "{} already has every pattern init would add", + path.display() + ); + } else { + yaml::write_text(path, &merge.text)?; + println!( + "added {} pattern(s) to the top of {}: {}", + merge.added.len(), + path.display(), + merge.added.join(" ") + ); + } + + for pattern in &merge.negated { + println!( + "note: {} un-ignores `{pattern}`, and the last matching rule wins, \ + so init did not add it", + path.display() + ); + if pattern.contains("salt") { + println!( + " that rule will commit the pseudonymization salt; \ + remove it before you push" + ); + } + } + Ok(()) +} + /// `schema`: (re)write the JSON Schemas an editor uses to validate the YAML. pub(crate) fn schema(cli: &Cli) -> Result { let layout = Layout::new(&cli.course); @@ -108,6 +688,13 @@ pub(crate) fn validate(cli: &Cli) -> Result { } } + let seals = coursebank::seal::SealFile::load_all(&catalog.layout.seals())?; + all.extend(catalog.validate_seals(&seals)); + + // The statistics are kept in a different file from the questions they + // describe, so the link between them is worth checking rather than assuming. + all.extend(catalog.calibration.validate(&catalog)); + if all.is_empty() { if !cli.quiet { println!( @@ -235,15 +822,29 @@ pub(crate) fn catalog(cli: &Cli, args: &CatalogArgs) -> Result { let coverage = catalog.coverage(); println!("\nObjective coverage:"); println!( - " {:<40} {:>6} {:>6} MAX LEVEL", - "OBJECTIVE", "ITEMS", "READY" + " {:<44} {:>6} {:>6} {:>7} MAX LEVEL", + "OBJECTIVE / TARGET", "ITEMS", "READY", "TARGETS" ); for row in &coverage.rows { + // Objectives carry the totals and are the tier a blueprint is + // written at, so they are the rows to scan; the indented target rows + // say where inside each one the items sit. + let label = if row.tier == Tier::Objective { + truncate(&row.id, 44) + } else { + truncate(&format!(" - {}", row.id), 44) + }; + let targets = if row.targets > 0 { + format!("{}/{}", row.targets_covered, row.targets) + } else { + "-".to_string() + }; println!( - " {:<40} {:>6} {:>6} {}", - truncate(&row.objective, 40), + " {:<44} {:>6} {:>6} {:>7} {}", + label, row.total, row.assemblable, + targets, row.max_level .map(|l| l.code().to_string()) .unwrap_or_else(|| "-".into()) @@ -271,4 +872,74 @@ mod tests { assert!(GITIGNORE.contains(".coursebank-salt")); assert!(GITIGNORE.contains("build/")); } + + #[test] + fn merging_keeps_the_existing_file_and_adds_only_what_was_missing() { + let existing = "# rules I wrote\ntarget/\nbuild/\n*.pdf\n"; + let merge = merge_gitignore(existing); + + assert!( + merge.text.ends_with(existing), + "the existing file must survive byte for byte:\n{}", + merge.text + ); + assert!(merge.text.starts_with(GITIGNORE_HEADER)); + assert!(!merge.added.iter().any(|p| p == "build/" || p == "*.pdf")); + assert!(merge.added.iter().any(|p| p == "reports/")); + assert!(merge.added.iter().any(|p| p == ".coursebank-salt")); + } + + #[test] + fn a_block_with_nothing_left_to_add_loses_its_comment() { + let merge = merge_gitignore("build/\nreports/\n"); + assert!(!merge.text.contains("# Generated output")); + assert!(merge.text.contains("# Typst and PDF artifacts.")); + } + + #[test] + fn slashes_do_not_make_a_pattern_look_new() { + let merge = merge_gitignore("/build\nreports\n/data/\n"); + assert!( + !merge.added.iter().any(|p| p.contains("build")), + "`/build` already covers `build/`, so it must not be added again" + ); + assert!(!merge.added.iter().any(|p| p.contains("reports"))); + } + + #[test] + fn a_complete_file_is_left_exactly_as_it_was() { + let merge = merge_gitignore(GITIGNORE); + assert!(merge.added.is_empty()); + assert_eq!(merge.text, GITIGNORE); + } + + #[test] + fn a_negated_pattern_is_reported_instead_of_reinserted() { + let merge = merge_gitignore("*.pdf\n!*.salt\n"); + assert_eq!(merge.negated, vec!["*.salt".to_string()]); + assert!(!merge.added.iter().any(|p| p == "*.salt")); + // The un-ignore covers one spelling of the salt, not the other. + assert!(merge.added.iter().any(|p| p == ".coursebank-salt")); + } + + #[test] + fn line_endings_follow_the_file_being_merged_into() { + let merge = merge_gitignore("target/\r\n"); + assert_eq!( + merge.text.matches('\n').count(), + merge.text.matches("\r\n").count(), + "a CRLF file must not gain bare LF lines:\n{:?}", + merge.text + ); + } + + #[test] + fn comments_and_blank_lines_are_not_patterns() { + assert_eq!(pattern_key(" build/ ").as_deref(), Some("build")); + assert_eq!(pattern_key("/build/").as_deref(), Some("build")); + assert_eq!(pattern_key("!build").as_deref(), Some("!build")); + assert_eq!(pattern_key("# build/"), None); + assert_eq!(pattern_key(" "), None); + assert_eq!(pattern_key("/"), None); + } } diff --git a/src/data.rs b/src/data.rs index db5b7f2..f36152c 100644 --- a/src/data.rs +++ b/src/data.rs @@ -26,8 +26,9 @@ //! `--no-default-features` a one-file change rather than a refactor. pub mod canvas; +pub mod decode; pub mod gradescope; +pub mod intake; pub mod responses; pub mod store; -#[cfg(feature = "parquet")] pub mod store_parquet; diff --git a/src/data/canvas.rs b/src/data/canvas.rs index 16f9982..a6ce18a 100644 --- a/src/data/canvas.rs +++ b/src/data/canvas.rs @@ -325,6 +325,10 @@ pub fn ingest( assessment_id: ctx.assessment_id.clone(), date: ctx.date, form: ctx.form.clone(), + // Canvas numbers questions as the record does, and its exports + // carry no printed order, so there is no printed position to + // record and no letter map to decode against. + form_position: None, student_key: student_key.clone(), sid: sid.clone(), name: name.clone(), @@ -333,18 +337,22 @@ pub fn ingest( item_number: number, item_ref, item_version: None, + variant: None, selected, + selected_source: Vec::new(), eliminated: Vec::new(), + eliminated_source: Vec::new(), correct, credit, points_possible: points, score, response_time_seconds: None, level: None, - learning_objectives: Vec::new(), + learning_targets: Vec::new(), topics: Vec::new(), bonus: false, dropped: false, + dropped_full_credit: false, }); } } diff --git a/src/data/decode.rs b/src/data/decode.rs new file mode 100644 index 0000000..447ae8f --- /dev/null +++ b/src/data/decode.rs @@ -0,0 +1,848 @@ +// SPDX-License-Identifier: Prosperity-3.0.0 +// Copyright Scientific Computing Studio +// Source: https://git.scient.ing/education/coursebank + +//! Turning what a student marked into what a student chose. +//! +//! A grading export speaks in positions and printed letters. Question 14 is the +//! fourteenth thing on the page; option C is the third bubble. An item bank speaks +//! in ids and its own lettering. When forms shuffle, those two vocabularies +//! disagree, and every analysis downstream of the disagreement is wrong in a way +//! that looks right: +//! +//! * Pooled distractor statistics add form A's option C to form B's option C, +//! which are different sentences. The resulting table is noise with the shape of +//! data. +//! * A student report looks up the misconception recorded on option C and shows it +//! to a student who chose a different option. The feedback is confident, +//! specific, and about the wrong thing. +//! * Any `credit_overrides` written in the record's lettering are applied to +//! whoever happened to mark that letter on their form. +//! +//! None of these fail loudly. That is the argument for doing the translation once, +//! at ingest, and storing both sides of it. +//! +//! # What a decoder knows +//! +//! For one form: which recorded question number sits at each printed position, +//! which bank letter each printed letter stands for, and which printed letters are +//! keyed. It is built from a [`SealFile`] when one exists, and derived from the +//! record and the form seed when one does not. Sealed is better, and not only +//! because it is faster: a derived decoder describes the form the bank *would* +//! print today, while a sealed one describes the form that was actually printed. +//! +//! # Catching a swapped directory +//! +//! Gradescope's point-value row reveals which printed letter earned full credit on +//! every question. A decoder knows what that letter should be. Comparing them +//! across a whole directory is close to a proof of which form the directory holds: +//! agreement is near total for the right form and near chance for the wrong one. +//! [`identify_form`] uses that to refuse an ingest that names form A over a +//! directory of form B papers, which is otherwise a mistake nobody catches until +//! the item statistics look strange three weeks later. + +use std::collections::{BTreeMap, BTreeSet}; + +use crate::assessment::{AssessmentFile, Form}; +use crate::catalog::Catalog; +use crate::error::{Error, Result}; +use crate::gradescope::Question; +use crate::responses::ResponseSet; +use crate::seal::{SealFile, printed_letter}; +use crate::select; + +/// Where a decoder's mapping came from. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Provenance { + /// Read from a seal written before administration. Authoritative. + Seal, + /// Derived from the assessment record and the form seed, as of now. + Derived, +} + +impl Provenance { + /// A short label for output. + pub fn label(self) -> &'static str { + match self { + Provenance::Seal => "seal", + Provenance::Derived => "derived from the record", + } + } +} + +/// How a grading export numbers its questions. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Numbering { + /// The export counts printed positions, which is what Gradescope's `N.csv` + /// file names mean. Positions are translated to recorded numbers. + Printed, + /// The export already carries recorded question numbers, so numbers pass + /// through untouched. + Recorded, +} + +/// The form marker put on a row that could not be translated, so it can be +/// removed after the borrow on `set.rows` ends. No real form id can collide +/// with it: form ids come from the record and are short labels like `A`. +const UNMAPPED: &str = "\u{1f}unmapped"; + +/// One question's mapping on one form. +#[derive(Debug, Clone)] +pub struct QuestionMap { + /// Printed position on the page, counting from 1. + pub position: u32, + /// The recorded question number, the join key to the record and the store. + pub number: u32, + /// The item's global id. + pub item: String, + /// Keyed letters as printed on this form. + pub printed_key: Vec, + /// Keyed letters in the bank's own lettering. + pub canonical_key: Vec, + /// Printed letter to bank letter. + pub to_canonical: BTreeMap, + /// Bank letter to printed letter. + pub to_printed: BTreeMap, +} + +impl QuestionMap { + /// The bank letter a printed letter stands for. + /// + /// # Arguments + /// + /// * `printed` - the letter as the student saw it. + /// + /// # Returns + /// + /// The bank letter, or `None` when the printed letter is not one of this + /// question's options. + pub fn canonical(&self, printed: &str) -> Option<&str> { + self.to_canonical + .get(&printed.trim().to_ascii_uppercase()) + .map(|s| s.as_str()) + } + + /// How many options this question has. + pub fn n_options(&self) -> usize { + self.to_canonical.len() + } + + /// Whether this question's options were actually permuted. + pub fn is_permuted(&self) -> bool { + self.to_canonical.iter().any(|(k, v)| k != v) + } +} + +/// One form's full mapping. +#[derive(Debug, Clone)] +pub struct FormDecoder { + /// The form id. + pub form: String, + /// Where the mapping came from. + pub provenance: Provenance, + /// Questions by printed position. + by_position: BTreeMap, + /// Questions by recorded number. + by_number: BTreeMap, +} + +impl FormDecoder { + /// Builds a decoder from the question maps. + fn assemble(form: String, provenance: Provenance, maps: Vec) -> FormDecoder { + let by_position = maps.iter().map(|m| (m.position, m.clone())).collect(); + let by_number = maps.into_iter().map(|m| (m.number, m)).collect(); + FormDecoder { + form, + provenance, + by_position, + by_number, + } + } + + /// Reads one form's mapping out of a seal. + /// + /// # Arguments + /// + /// * `seal` - the seal. + /// * `form_id` - the form to decode, matched case-insensitively. + /// + /// # Returns + /// + /// The decoder. + /// + /// # Errors + /// + /// Returns [`Error::Usage`] when the seal does not cover that form. + pub fn from_seal(seal: &SealFile, form_id: &str) -> Result { + let form = seal.form(form_id).ok_or_else(|| { + Error::usage(format!( + "the seal for `{}` does not cover form `{form_id}`; it covers {}", + seal.seal.assessment, + seal.forms + .iter() + .map(|f| f.id.as_str()) + .collect::>() + .join(", ") + )) + })?; + + let maps = form + .questions + .iter() + .map(|q| { + let mut to_canonical = BTreeMap::new(); + let mut to_printed = BTreeMap::new(); + for map in &q.options { + to_canonical.insert(map.printed.clone(), map.canonical.clone()); + to_printed.insert(map.canonical.clone(), map.printed.clone()); + } + let canonical_key: Vec = q + .printed_key + .iter() + .filter_map(|p| to_canonical.get(p).cloned()) + .collect(); + QuestionMap { + position: q.position, + number: q.number, + item: q.item.clone(), + printed_key: q.printed_key.clone(), + canonical_key, + to_canonical, + to_printed, + } + }) + .collect(); + + Ok(FormDecoder::assemble( + form.id.clone(), + Provenance::Seal, + maps, + )) + } + + /// Derives one form's mapping from the record and the bank. + /// + /// Uses the same two functions every export calls, so a derived decoder and a + /// freshly exported paper agree by construction. + /// + /// # Arguments + /// + /// * `catalog` - the loaded course. + /// * `record` - the assessment record. + /// * `form` - the form. + /// + /// # Returns + /// + /// The decoder. + /// + /// # Errors + /// + /// Returns [`Error::Unresolved`] when a placement references a missing item. + pub fn derive(catalog: &Catalog, record: &AssessmentFile, form: &Form) -> Result { + let printed: Vec<_> = select::layout(record, form) + .into_iter() + .filter(|p| !p.dropped) + .collect(); + + let mut maps = Vec::with_capacity(printed.len()); + for (index, placement) in printed.iter().enumerate() { + let entry = catalog.require(&placement.item)?; + let item = &entry.item; + let shown = item.administered(&placement.key, &placement.distractors); + let order = select::option_order(form, &placement.item, shown.len()); + + let canonical_key: BTreeSet = if placement.key.is_empty() { + item.key_letters().into_iter().collect() + } else { + placement.key.iter().cloned().collect() + }; + + let mut to_canonical = BTreeMap::new(); + let mut to_printed = BTreeMap::new(); + let mut printed_key = Vec::new(); + for (position, source_index) in order.iter().enumerate() { + let canonical = shown + .get(*source_index) + .map(|c| c.id.clone()) + .unwrap_or_else(|| printed_letter(*source_index)); + let label = printed_letter(position); + if canonical_key.contains(&canonical) { + printed_key.push(label.clone()); + } + to_canonical.insert(label.clone(), canonical.clone()); + to_printed.insert(canonical, label); + } + + maps.push(QuestionMap { + position: index as u32 + 1, + number: placement.number, + item: placement.item.clone(), + printed_key, + canonical_key: canonical_key.into_iter().collect(), + to_canonical, + to_printed, + }); + } + + Ok(FormDecoder::assemble( + form.id.clone(), + Provenance::Derived, + maps, + )) + } + + /// Builds a decoder, preferring the seal. + /// + /// # Arguments + /// + /// * `seal` - the seal, when one has been written. + /// * `catalog` - the loaded course. + /// * `record` - the assessment record. + /// * `form` - the form. + /// + /// # Returns + /// + /// The decoder. + /// + /// # Errors + /// + /// As [`FormDecoder::from_seal`] and [`FormDecoder::derive`]. A seal that does + /// not cover the requested form falls back to deriving rather than failing, + /// since a form added after sealing is a real situation. + pub fn resolve( + seal: Option<&SealFile>, + catalog: &Catalog, + record: &AssessmentFile, + form: &Form, + ) -> Result { + if let Some(seal) = seal { + if seal.form(&form.id).is_some() { + return FormDecoder::from_seal(seal, &form.id); + } + } + FormDecoder::derive(catalog, record, form) + } + + /// The question at a printed position. + /// + /// # Arguments + /// + /// * `position` - the printed position, counting from 1. + pub fn at_position(&self, position: u32) -> Option<&QuestionMap> { + self.by_position.get(&position) + } + + /// The question with a recorded number. + /// + /// # Arguments + /// + /// * `number` - the recorded number. + pub fn at_number(&self, number: u32) -> Option<&QuestionMap> { + self.by_number.get(&number) + } + + /// The question an export's numbering refers to. + /// + /// # Arguments + /// + /// * `n` - the number as the export gives it. + /// * `numbering` - how the export numbers questions. + pub fn lookup(&self, n: u32, numbering: Numbering) -> Option<&QuestionMap> { + match numbering { + Numbering::Printed => self.at_position(n), + Numbering::Recorded => self.at_number(n), + } + } + + /// How many questions this form prints. + pub fn len(&self) -> usize { + self.by_position.len() + } + + /// Every recorded question number this form carries. + pub fn numbers(&self) -> impl Iterator + '_ { + self.by_position.values().map(|q| q.number) + } + + /// Whether the form prints nothing, which means the record is empty. + pub fn is_empty(&self) -> bool { + self.by_position.is_empty() + } + + /// Whether any question on this form has permuted options. + /// + /// Used to decide whether to say anything about translation at all: on an + /// unshuffled form the whole mechanism is an identity map and mentioning it + /// is noise. + pub fn is_permuted(&self) -> bool { + self.by_position.values().any(|q| q.is_permuted()) + } + + /// Whether printed positions and recorded numbers disagree anywhere. + /// + /// True when items were shuffled, and also when a bonus item sits mid-record, + /// since the layout moves bonus items to the end of the paper. + pub fn is_renumbered(&self) -> bool { + self.by_position.values().any(|q| q.position != q.number) + } +} + +/// What a directory of graded questions says about which form it holds. +#[derive(Debug, Clone)] +pub struct FormFit { + /// The form id. + pub form: String, + /// Questions whose graded key matched this form's printed key. + pub matched: usize, + /// Questions that could be compared at all. + pub compared: usize, + /// Question positions where the graded key disagreed. + pub mismatches: Vec, +} + +impl FormFit { + /// The share of comparable questions that agreed. + pub fn rate(&self) -> f64 { + if self.compared == 0 { + 0.0 + } else { + self.matched as f64 / self.compared as f64 + } + } + + /// Whether the fit is good enough to proceed without a warning. + /// + /// The threshold is high on purpose. A correctly matched directory agrees on + /// every question; anything less than total agreement is either a regrade that + /// moved a key or the wrong directory, and both are worth a sentence. + pub fn is_convincing(&self) -> bool { + self.compared > 0 && self.matched == self.compared + } +} + +/// Compares a parsed Gradescope directory against one form's expected keys. +/// +/// # Arguments +/// +/// * `questions` - the parsed question files. +/// * `decoder` - the form to test against. +/// * `numbering` - how the export numbers questions. +/// +/// # Returns +/// +/// The fit. +pub fn fit_form(questions: &[Question], decoder: &FormDecoder, numbering: Numbering) -> FormFit { + let mut matched = 0usize; + let mut compared = 0usize; + let mut mismatches = Vec::new(); + + for question in questions { + let Some(map) = decoder.lookup(question.number, numbering) else { + continue; + }; + let graded: BTreeSet = question.keyed().into_iter().collect(); + if graded.is_empty() { + continue; + } + let expected: BTreeSet = map.printed_key.iter().cloned().collect(); + if expected.is_empty() { + continue; + } + compared += 1; + if graded == expected { + matched += 1; + } else { + mismatches.push(question.number); + } + } + + FormFit { + form: decoder.form.clone(), + matched, + compared, + mismatches, + } +} + +/// Ranks every candidate form against a directory. +/// +/// # Arguments +/// +/// * `questions` - the parsed question files. +/// * `decoders` - one decoder per declared form. +/// * `numbering` - how the export numbers questions. +/// +/// # Returns +/// +/// The fits, best first. +pub fn identify_form( + questions: &[Question], + decoders: &[FormDecoder], + numbering: Numbering, +) -> Vec { + let mut fits: Vec = decoders + .iter() + .map(|d| fit_form(questions, d, numbering)) + .collect(); + fits.sort_by(|a, b| { + b.rate() + .partial_cmp(&a.rate()) + .unwrap_or(std::cmp::Ordering::Equal) + .then_with(|| a.form.cmp(&b.form)) + }); + fits +} + +/// Explains a fit in a sentence, or says nothing when the fit is perfect. +/// +/// # Arguments +/// +/// * `claimed` - the form the directory was ingested as. +/// * `fits` - every form's fit, best first. +/// +/// # Returns +/// +/// A warning, or `None`. +pub fn form_warning(claimed: &str, fits: &[FormFit]) -> Option { + let mine = fits.iter().find(|f| f.form.eq_ignore_ascii_case(claimed))?; + if mine.is_convincing() { + return None; + } + if mine.compared == 0 { + return Some(format!( + "form {claimed}: the export carries no point values, so the graded keys could not be \ + checked against the form. Nothing verified this directory is form {claimed}" + )); + } + + let better = fits + .iter() + .find(|f| !f.form.eq_ignore_ascii_case(claimed) && f.rate() > mine.rate()); + + let head = format!( + "form {claimed}: the graded key matches this form on {} of {} question(s)", + mine.matched, mine.compared + ); + let where_ = if mine.mismatches.is_empty() { + String::new() + } else { + let list: Vec = mine + .mismatches + .iter() + .take(8) + .map(|n| n.to_string()) + .collect(); + format!( + " (q{}{})", + list.join(", q"), + if mine.mismatches.len() > 8 { + ", …" + } else { + "" + } + ) + }; + match better { + Some(other) => Some(format!( + "{head}{where_}, but matches form {} on {} of {}. This directory is almost certainly \ + form {}, not form {claimed}", + other.form, other.matched, other.compared, other.form + )), + None => Some(format!( + "{head}{where_}. Either those questions were regraded after printing, or the form is \ + not the one named" + )), + } +} + +/// Translates a form's responses into the bank's vocabulary. +/// +/// Rewrites, for every row whose `form` matches this decoder: +/// +/// * `item_number`, from printed position to recorded number, when the export +/// numbers by position; +/// * `form_position`, recording where the question sat on the page; +/// * `selected_source` and `eliminated_source`, the bank letters for what was +/// marked. The printed letters stay in `selected` and `eliminated`, because what +/// a student physically marked is the fact and the translation is the +/// interpretation. +/// +/// Correctness and credit are untouched. Both come from the grading platform, +/// which scored the paper the student actually held, and are already right. +/// +/// # Arguments +/// +/// * `set` - the responses to translate, in place. +/// * `decoder` - the form's mapping. +/// * `numbering` - how the export numbered questions. +/// +/// # Returns +/// +/// Warnings for anything that could not be translated. +pub fn apply(set: &mut ResponseSet, decoder: &FormDecoder, numbering: Numbering) -> Vec { + let mut warnings = Vec::new(); + let mut unmapped_positions: BTreeSet = BTreeSet::new(); + let mut colliding_positions: BTreeSet = BTreeSet::new(); + let mut unmapped_letters: BTreeSet = BTreeSet::new(); + let mut translated = 0usize; + + // Numbers this form really uses. An untranslated row whose raw number is one + // of these would silently masquerade as that question, and two rows would + // then share a number: one the student's answer to it, one an answer to + // something else entirely. Nothing downstream can tell them apart, so the + // collision has to be caught here. + let recorded: BTreeSet = decoder.numbers().collect(); + + for row in &mut set.rows { + let belongs = row + .form + .as_deref() + .map(|f| f.eq_ignore_ascii_case(&decoder.form)) + .unwrap_or(false); + if !belongs { + continue; + } + + let Some(map) = decoder.lookup(row.item_number, numbering) else { + unmapped_positions.insert(row.item_number); + if recorded.contains(&row.item_number) { + colliding_positions.insert(row.item_number); + // Marked so the row can be discarded below. Attributing it to + // the question that legitimately holds this number would corrupt + // that question's statistics. + row.form = Some(UNMAPPED.to_string()); + } + continue; + }; + + row.form_position = Some(map.position); + row.item_number = map.number; + + let mut convert = |letters: &[String]| -> Vec { + let mut out = Vec::with_capacity(letters.len()); + for letter in letters { + match map.canonical(letter) { + Some(canonical) => out.push(canonical.to_string()), + None => { + unmapped_letters.insert(format!("q{} {}", map.number, letter)); + } + } + } + out.sort(); + out + }; + + row.selected_source = convert(&row.selected); + row.eliminated_source = convert(&row.eliminated); + translated += 1; + } + + if !unmapped_positions.is_empty() { + let list: Vec = unmapped_positions.iter().map(|n| n.to_string()).collect(); + warnings.push(format!( + "form {}: question(s) {} are in the export but not on this form ({} printed). The \ + export may have been taken before a question was dropped, or from a different \ + form.", + decoder.form, + list.join(", "), + decoder.len() + )); + } + if !colliding_positions.is_empty() { + let list: Vec = colliding_positions.iter().map(|n| n.to_string()).collect(); + let discarded = set.rows.len(); + set.rows.retain(|row| row.form.as_deref() != Some(UNMAPPED)); + warnings.push(format!( + "form {}: {} response(s) at position(s) {} could not be translated, and their raw \ + numbers are numbers this form does use. Keeping them would have given those \ + questions two different answers each, so they were discarded. This is the shape of \ + an export made before a question was dropped: re-export the responses from the \ + administration you sealed, or re-run with --recorded-numbers if the export already \ + carries recorded numbers.", + decoder.form, + discarded - set.rows.len(), + list.join(", ") + )); + } + if !unmapped_letters.is_empty() { + let list: Vec = unmapped_letters.iter().take(10).cloned().collect(); + warnings.push(format!( + "form {}: {} marked option(s) are not options on the printed form ({}), which usually \ + means a rubric column was added by hand in Gradescope", + decoder.form, + unmapped_letters.len(), + list.join(", ") + )); + } + if translated == 0 { + warnings.push(format!( + "form {}: no response rows carried this form id, so nothing was translated", + decoder.form + )); + } + + warnings +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::responses::{Response, administration_id}; + + fn map(position: u32, number: u32, pairs: &[(&str, &str)], key: &str) -> QuestionMap { + let mut to_canonical = BTreeMap::new(); + let mut to_printed = BTreeMap::new(); + for (printed, canonical) in pairs { + to_canonical.insert(printed.to_string(), canonical.to_string()); + to_printed.insert(canonical.to_string(), printed.to_string()); + } + let printed_key = vec![to_printed.get(key).cloned().unwrap_or_default()]; + QuestionMap { + position, + number, + item: format!("b::q-{number}"), + printed_key, + canonical_key: vec![key.to_string()], + to_canonical, + to_printed, + } + } + + fn decoder() -> FormDecoder { + FormDecoder::assemble( + "B".to_string(), + Provenance::Seal, + vec![ + // Printed A..D show bank C, A, D, B. The key is bank A, printed B. + map(1, 1, &[("A", "C"), ("B", "A"), ("C", "D"), ("D", "B")], "A"), + // A bonus item recorded as 9 but printed last, at position 2. + map(2, 9, &[("A", "B"), ("B", "A")], "B"), + ], + ) + } + + fn row(number: u32, selected: &str) -> Response { + Response { + administration_id: administration_id("C", "2026f", "e1"), + course: "C".into(), + term: "2026f".into(), + assessment_id: "e1".into(), + date: None, + form: Some("B".into()), + student_key: "s1".into(), + sid: None, + name: None, + email: None, + section: None, + item_number: number, + form_position: None, + item_ref: None, + item_version: None, + variant: None, + selected: vec![selected.into()], + eliminated: Vec::new(), + selected_source: Vec::new(), + eliminated_source: Vec::new(), + correct: Some(false), + credit: 0.0, + points_possible: 1.0, + score: 0.0, + response_time_seconds: None, + level: None, + learning_targets: Vec::new(), + topics: Vec::new(), + bonus: false, + dropped: false, + dropped_full_credit: false, + } + } + + #[test] + fn printed_letters_become_bank_letters() { + let decoder = decoder(); + let q = decoder.at_position(1).unwrap(); + assert_eq!(q.canonical("A"), Some("C")); + assert_eq!(q.canonical("D"), Some("B")); + assert_eq!(q.canonical("E"), None); + assert_eq!(q.printed_key, vec!["B".to_string()]); + } + + #[test] + fn positions_become_recorded_numbers() { + let mut set = ResponseSet::new(); + set.rows.push(row(2, "A")); + let warnings = apply(&mut set, &decoder(), Numbering::Printed); + assert!(warnings.is_empty(), "{warnings:?}"); + assert_eq!(set.rows[0].item_number, 9, "position 2 is recorded as 9"); + assert_eq!(set.rows[0].form_position, Some(2)); + assert_eq!(set.rows[0].selected, vec!["A".to_string()], "printed kept"); + assert_eq!(set.rows[0].selected_source, vec!["B".to_string()]); + } + + #[test] + fn rows_from_another_form_are_left_alone() { + let mut set = ResponseSet::new(); + let mut other = row(1, "A"); + other.form = Some("A".into()); + set.rows.push(other); + apply(&mut set, &decoder(), Numbering::Printed); + assert!(set.rows[0].selected_source.is_empty()); + assert_eq!(set.rows[0].item_number, 1); + } + + #[test] + fn an_unknown_position_is_reported_not_guessed() { + let mut set = ResponseSet::new(); + set.rows.push(row(7, "A")); + let warnings = apply(&mut set, &decoder(), Numbering::Printed); + assert!( + warnings.iter().any(|w| w.contains("not on this form")), + "{warnings:?}" + ); + assert_eq!(set.rows[0].item_number, 7, "left as found"); + } + + #[test] + fn a_perfect_fit_says_nothing() { + let fits = vec![FormFit { + form: "A".into(), + matched: 30, + compared: 30, + mismatches: Vec::new(), + }]; + assert!(form_warning("A", &fits).is_none()); + } + + #[test] + fn a_swapped_directory_names_the_form_it_really_is() { + let fits = vec![ + FormFit { + form: "B".into(), + matched: 30, + compared: 30, + mismatches: Vec::new(), + }, + FormFit { + form: "A".into(), + matched: 8, + compared: 30, + mismatches: (1..=22).collect(), + }, + ]; + let warning = form_warning("A", &fits).expect("a mismatch this large must warn"); + assert!(warning.contains("almost certainly form B"), "{warning}"); + } + + #[test] + fn a_single_regraded_key_warns_without_accusing_the_wrong_form() { + let fits = vec![FormFit { + form: "A".into(), + matched: 29, + compared: 30, + mismatches: vec![14], + }]; + let warning = form_warning("A", &fits).unwrap(); + assert!(warning.contains("q14"), "{warning}"); + assert!(warning.contains("regraded"), "{warning}"); + } +} diff --git a/src/data/gradescope.rs b/src/data/gradescope.rs index 050739c..57f5914 100644 --- a/src/data/gradescope.rs +++ b/src/data/gradescope.rs @@ -633,6 +633,10 @@ pub fn to_responses(questions: &[Question], ctx: &Context) -> Import { assessment_id: ctx.assessment_id.clone(), date: ctx.date, form: ctx.form.clone(), + // The printed position and the bank's lettering are written + // later, by `decode::apply`, which is the only place that knows + // which form this directory holds. + form_position: None, student_key, sid: row.sid.clone(), name: row.name.clone(), @@ -641,18 +645,22 @@ pub fn to_responses(questions: &[Question], ctx: &Context) -> Import { item_number: q.number, item_ref: None, item_version: None, + variant: None, selected, + selected_source: Vec::new(), eliminated, + eliminated_source: Vec::new(), correct, credit, points_possible: points, score: row.score, response_time_seconds: None, level: None, - learning_objectives: Vec::new(), + learning_targets: Vec::new(), topics: Vec::new(), bonus, dropped: false, + dropped_full_credit: false, }); } } diff --git a/src/data/intake.rs b/src/data/intake.rs new file mode 100644 index 0000000..385b93b --- /dev/null +++ b/src/data/intake.rs @@ -0,0 +1,565 @@ +// SPDX-License-Identifier: Prosperity-3.0.0 +// Copyright Scientific Computing Studio +// Source: https://git.scient.ing/education/coursebank + +//! Reading several forms of one exam back in at once. +//! +//! A two-form exam is two Gradescope assignments, each exporting its own +//! directory of `1.csv` through `N.csv`, each numbered against its own paper. They +//! are one administration: one set of students, one item pool, one set of +//! statistics. Ingesting them one command at a time does not work, because the +//! store keys on the administration and the second write replaces the first. +//! +//! So this module takes the whole set: +//! +//! ```text +//! coursebank ingest gradescope A=exports/e1-a B=exports/e1-b --assessment e1 +//! ``` +//! +//! and does four things the single-directory path cannot: +//! +//! *Checks each directory is the form it claims to be.* Every export carries the +//! graded key in its point-value row; every form knows what its printed key should +//! be. Comparing them catches a swapped pair of directories immediately rather +//! than three weeks later, when the distractor table looks strange. See +//! [`crate::decode::identify_form`]. +//! +//! *Translates each form into the bank's vocabulary before merging.* Otherwise +//! form A's option C and form B's option C land in the same column of the same +//! table while meaning different things. +//! +//! *Merges into one response set*, written once, so `analyze` and `report` see the +//! whole class. +//! +//! *Reports what the merge revealed*: a student who appears on two forms, a form +//! that is missing a question the other has, a form that ran materially harder +//! than the other. + +use std::collections::{BTreeMap, BTreeSet}; +use std::path::{Path, PathBuf}; + +use crate::assessment::{AssessmentFile, Form}; +use crate::catalog::Catalog; +use crate::date::Date; +use crate::decode::{self, FormDecoder, FormFit, Numbering}; +use crate::error::{Error, Result}; +use crate::gradescope::{self, Context, Question}; +use crate::responses::ResponseSet; +use crate::seal::SealFile; + +/// One directory of graded questions, and the form it holds. +#[derive(Debug, Clone)] +pub struct Source { + /// The form id. + pub form: String, + /// The directory of per-question CSV exports. + pub dir: PathBuf, +} + +impl Source { + /// Parses a `FORM=DIR` argument. + /// + /// A bare path is accepted and takes the fallback form, so the single-form + /// case stays as short as it was. + /// + /// # Arguments + /// + /// * `text` - the argument, e.g. `A=exports/e1-a` or `exports/e1`. + /// * `fallback` - the form to use when the argument names none. + /// + /// # Returns + /// + /// The source. + /// + /// # Errors + /// + /// Returns [`Error::Usage`] when the argument names no form and no fallback + /// was given, or when the form label is empty. + pub fn parse(text: &str, fallback: Option<&str>) -> Result { + // Split on the first `=` only: a directory name may contain one, a form + // label may not. + if let Some((form, dir)) = text.split_once('=') { + let form = form.trim(); + if form.is_empty() { + return Err(Error::usage(format!( + "`{text}` has an empty form label; write it as FORM=DIR, e.g. A=exports/e1-a" + ))); + } + if !dir.trim().is_empty() { + return Ok(Source { + form: form.to_string(), + dir: PathBuf::from(dir.trim()), + }); + } + } + match fallback { + Some(form) => Ok(Source { + form: form.to_string(), + dir: PathBuf::from(text.trim()), + }), + None => Err(Error::usage(format!( + "`{text}` does not say which form it holds; write it as FORM=DIR (e.g. \ + A=exports/e1-a) or pass --form" + ))), + } + } +} + +/// How to run an intake. +#[derive(Debug, Clone)] +pub struct Options { + /// The administration date, overriding the record's. + pub date: Option, + /// How the exports number their questions. + pub numbering: Numbering, + /// Whether a form that fails its key check stops the ingest. + /// + /// On by default. A directory that does not match the form it was named as is + /// the one ingest error that produces confident, wrong analysis rather than an + /// obvious failure, so the default is to refuse and say so. + pub strict: bool, +} + +impl Default for Options { + fn default() -> Options { + Options { + date: None, + numbering: Numbering::Printed, + strict: true, + } + } +} + +/// What one form's directory contributed. +#[derive(Debug, Clone)] +pub struct FormIntake { + /// The form id. + pub form: String, + /// The directory it came from. + pub dir: PathBuf, + /// Where its mapping came from. + pub provenance: decode::Provenance, + /// How many students it held. + pub students: usize, + /// How many questions it held. + pub questions: usize, + /// How the graded keys compared to every declared form, best first. + pub fits: Vec, + /// The parsed question files, kept so grading-time decisions stay available. + pub parsed: Vec, +} + +impl FormIntake { + /// This form's own fit. + pub fn own_fit(&self) -> Option<&FormFit> { + self.fits + .iter() + .find(|f| f.form.eq_ignore_ascii_case(&self.form)) + } +} + +/// The result of reading every form of one administration. +#[derive(Debug, Clone)] +pub struct Intake { + /// The merged, translated responses. + pub responses: ResponseSet, + /// Per-form detail. + pub forms: Vec, + /// Problems that did not stop the ingest. + pub warnings: Vec, +} + +impl Intake { + /// Every parsed question file, across forms. + /// + /// # Returns + /// + /// Pairs of form id and question. + pub fn questions(&self) -> Vec<(&str, &Question)> { + self.forms + .iter() + .flat_map(|f| f.parsed.iter().map(move |q| (f.form.as_str(), q))) + .collect() + } +} + +/// Reads every source into one response set. +/// +/// # Arguments +/// +/// * `catalog` - the loaded course. +/// * `record` - the assessment record. +/// * `seal` - the seal, when one was written. Strongly preferred: it describes the +/// paper that was printed rather than the paper the bank would print today. +/// * `sources` - the directories and the forms they hold. +/// * `opts` - how to run. +/// +/// # Returns +/// +/// The merged intake. +/// +/// # Errors +/// +/// Returns [`Error::Usage`] when a source names a form the record does not +/// declare, when two sources name the same form, or when a key check fails under +/// `strict`. Propagates parse errors from the exports themselves. +pub fn run( + catalog: &Catalog, + record: &AssessmentFile, + seal: Option<&SealFile>, + sources: &[Source], + opts: &Options, +) -> Result { + if sources.is_empty() { + return Err(Error::usage( + "no directories to ingest; pass one per form, e.g. A=exports/e1-a B=exports/e1-b" + .to_string(), + )); + } + + let declared = declared_forms(record); + let mut seen: BTreeSet = BTreeSet::new(); + for source in sources { + if !seen.insert(source.form.to_ascii_uppercase()) { + return Err(Error::usage(format!( + "form {} was given twice; each form is one directory", + source.form + ))); + } + if !declared + .iter() + .any(|f| f.id.eq_ignore_ascii_case(&source.form)) + { + return Err(Error::usage(format!( + "the record for `{}` declares no form `{}`; it declares {}", + record.assessment.id, + source.form, + declared + .iter() + .map(|f| f.id.as_str()) + .collect::>() + .join(", ") + ))); + } + } + + // One decoder per declared form, not just per ingested form: identifying a + // swapped directory means testing it against the forms it might be. + let mut decoders: Vec = Vec::new(); + for form in &declared { + decoders.push(FormDecoder::resolve(seal, catalog, record, form)?); + } + + let mut merged = ResponseSet::new(); + let mut forms = Vec::new(); + let mut warnings = Vec::new(); + let mut blocking = Vec::new(); + + for source in sources { + let decoder = decoders + .iter() + .find(|d| d.form.eq_ignore_ascii_case(&source.form)) + .expect("every source's form was checked against the declared list"); + + let ctx = Context { + course: catalog.course.course.code.clone(), + term: record + .assessment + .term + .clone() + .unwrap_or_else(|| catalog.course.course.term.clone()), + assessment_id: record.assessment.id.clone(), + date: opts.date.or(record.assessment.date), + form: Some(decoder.form.clone()), + }; + + let import = gradescope::ingest_dir(&source.dir, &ctx)?; + let mut set = import.responses; + + let fits = decode::identify_form(&import.questions, &decoders, opts.numbering); + if let Some(problem) = decode::form_warning(&decoder.form, &fits) { + let message = format!("{} [{}]", problem, source.dir.display()); + let convincing_alternative = fits + .iter() + .any(|f| !f.form.eq_ignore_ascii_case(&decoder.form) && f.is_convincing()); + if opts.strict && convincing_alternative { + blocking.push(message); + } else { + warnings.push(message); + } + } + + warnings.extend(decode::apply(&mut set, decoder, opts.numbering)); + + forms.push(FormIntake { + form: decoder.form.clone(), + dir: source.dir.clone(), + provenance: decoder.provenance, + students: set.students().len(), + questions: set.all_items().len(), + fits, + parsed: import.questions, + }); + + merged.absorb(set); + } + + if !blocking.is_empty() { + blocking.push( + "Nothing was written. Fix the form labels, or pass --allow-mismatch if the keys really \ + did change after printing." + .to_string(), + ); + return Err(Error::Invalid(blocking)); + } + + warnings.extend(cross_form_checks(&merged, &forms, record)); + merged.warnings.extend(warnings.clone()); + + Ok(Intake { + responses: merged, + forms, + warnings, + }) +} + +/// The forms a record declares, with the implicit single form for a record that +/// declares none. +fn declared_forms(record: &AssessmentFile) -> Vec
{ + if record.forms.is_empty() { + vec![Form { + id: "A".to_string(), + seed: 0, + shuffle_items: false, + shuffle_options: false, + }] + } else { + record.forms.clone() + } +} + +/// Checks that only merging several forms can make. +fn cross_form_checks( + merged: &ResponseSet, + forms: &[FormIntake], + record: &AssessmentFile, +) -> Vec { + let mut out = Vec::new(); + if forms.len() < 2 { + return out; + } + + // A student on two forms sat one exam and was graded twice, or two people + // share an identifier. Either way the response set now double counts them. + let mut by_student: BTreeMap<&str, BTreeSet<&str>> = BTreeMap::new(); + for row in &merged.rows { + if let Some(form) = row.form.as_deref() { + by_student + .entry(row.student_key.as_str()) + .or_default() + .insert(form); + } + } + let doubled: Vec<&str> = by_student + .iter() + .filter(|(_, forms)| forms.len() > 1) + .map(|(student, _)| *student) + .collect(); + if !doubled.is_empty() { + out.push(format!( + "{} student(s) appear on more than one form ({}); each is counted twice in every \ + total until one submission is removed", + doubled.len(), + doubled + .iter() + .take(5) + .copied() + .collect::>() + .join(", ") + )); + } + + // Every form is the same items in a different order, so a question present on + // one and absent from another means a directory is short a file. + let expected: BTreeSet = record + .items + .iter() + .filter(|p| !p.dropped) + .map(|p| p.number) + .collect(); + for form in forms { + let present: BTreeSet = merged + .rows + .iter() + .filter(|r| { + r.form + .as_deref() + .map(|f| f.eq_ignore_ascii_case(&form.form)) + .unwrap_or(false) + }) + .map(|r| r.item_number) + .collect(); + let missing: Vec = expected + .difference(&present) + .map(|n| n.to_string()) + .collect(); + if !missing.is_empty() { + out.push(format!( + "form {}: no responses for question(s) {}; the export directory is missing those \ + files", + form.form, + missing.join(", ") + )); + } + } + + // Forms are meant to be the same test. A large gap between their means is + // either a permutation that made one form easier or an uneven split of the + // class, and both are worth knowing before any grade is released. + let means: Vec<(String, f64, usize)> = forms + .iter() + .map(|f| { + let students: Vec<&str> = merged + .rows + .iter() + .filter(|r| { + r.form + .as_deref() + .map(|x| x.eq_ignore_ascii_case(&f.form)) + .unwrap_or(false) + }) + .map(|r| r.student_key.as_str()) + .collect::>() + .into_iter() + .collect(); + let possible = merged.points_available(); + let percents: Vec = students + .iter() + .map(|s| { + if possible > 0.0 { + 100.0 * merged.scored_total(s) / possible + } else { + 0.0 + } + }) + .collect(); + let mean = if percents.is_empty() { + 0.0 + } else { + percents.iter().sum::() / percents.len() as f64 + }; + (f.form.clone(), mean, percents.len()) + }) + .collect(); + + if let (Some(low), Some(high)) = ( + means + .iter() + .min_by(|a, b| a.1.partial_cmp(&b.1).unwrap_or(std::cmp::Ordering::Equal)), + means + .iter() + .max_by(|a, b| a.1.partial_cmp(&b.1).unwrap_or(std::cmp::Ordering::Equal)), + ) { + let gap = high.1 - low.1; + if gap >= 5.0 && low.2 >= 5 && high.2 >= 5 { + out.push(format!( + "form {} averaged {:.0}% and form {} averaged {:.0}%, a {:.0}-point gap. With {} \ + and {} students that may be the split rather than the forms, but it is worth \ + looking at before releasing grades", + high.0, high.1, low.0, low.1, gap, high.2, low.2 + )); + } + } + + out +} + +/// A one-line summary of where a directory came from and what it held. +/// +/// # Arguments +/// +/// * `intake` - one form's intake. +/// +/// # Returns +/// +/// The line, without a trailing newline. +pub fn describe(intake: &FormIntake) -> String { + let fit = match intake.own_fit() { + Some(fit) if fit.compared > 0 => { + format!(", keys matched {}/{}", fit.matched, fit.compared) + } + _ => String::new(), + }; + format!( + "form {}: {} student(s) x {} question(s) from {} (mapping {}{fit})", + intake.form, + intake.students, + intake.questions, + intake.dir.display(), + intake.provenance.label(), + ) +} + +/// Whether a path looks like a Gradescope per-question export directory. +/// +/// Used to give a better error than "no numbered CSV files" when someone points +/// at the zip they downloaded, or at the directory above the one they meant. +/// +/// # Arguments +/// +/// * `dir` - the candidate directory. +/// +/// # Returns +/// +/// `true` when it holds at least one file named like `1.csv`. +pub fn looks_like_export(dir: &Path) -> bool { + let Ok(entries) = std::fs::read_dir(dir) else { + return false; + }; + entries.filter_map(|e| e.ok()).any(|entry| { + let path = entry.path(); + path.extension().and_then(|e| e.to_str()) == Some("csv") + && path + .file_stem() + .and_then(|s| s.to_str()) + .map(|s| s.chars().all(|c| c.is_ascii_digit())) + .unwrap_or(false) + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_labelled_source_parses() { + let source = Source::parse("B=exports/e1-b", None).unwrap(); + assert_eq!(source.form, "B"); + assert_eq!(source.dir, PathBuf::from("exports/e1-b")); + } + + #[test] + fn a_bare_path_needs_a_fallback_form() { + let source = Source::parse("exports/e1", Some("A")).unwrap(); + assert_eq!(source.form, "A"); + assert_eq!(source.dir, PathBuf::from("exports/e1")); + + let err = Source::parse("exports/e1", None).unwrap_err(); + assert!(err.to_string().contains("which form"), "{err}"); + } + + #[test] + fn a_windows_style_path_is_not_mistaken_for_a_label() { + // The split is on the first `=`, and a drive letter has none, so this is + // only a hazard for a path that genuinely contains one. + let source = Source::parse("A=C:/exports/e1-a", None).unwrap(); + assert_eq!(source.form, "A"); + assert_eq!(source.dir, PathBuf::from("C:/exports/e1-a")); + } + + #[test] + fn an_empty_label_is_refused() { + let err = Source::parse("=exports/e1", None).unwrap_err(); + assert!(err.to_string().contains("empty form label"), "{err}"); + } +} diff --git a/src/data/responses.rs b/src/data/responses.rs index 12f4a0e..9a92440 100644 --- a/src/data/responses.rs +++ b/src/data/responses.rs @@ -52,6 +52,9 @@ pub struct Response { pub date: Option, /// Which form the student took, when forms were used. pub form: Option, + /// Where this question sat on the student's own form, when that differs from + /// the recorded number. Provenance, not a join key. + pub form_position: Option, /// The identifier analysis groups by. A pseudonym when pseudonymizing. pub student_key: String, @@ -72,10 +75,26 @@ pub struct Response { /// The item version as administered. pub item_version: Option, + /// The variant administered: which option set this row's student saw. + /// + /// The grouping key for pooled statistics. Set at ingest from the + /// assessment record, and stored so that anything reading the Parquet + /// without this tool can group the same way — a store that only has + /// `item_ref` cannot tell two option sets of one stem apart, and will + /// average them. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub variant: Option, + /// Option letters the student chose. pub selected: Vec, + /// The selected options in the bank's own lettering, written at ingest by + /// [`crate::decode::apply`]. Empty when the form was never decoded, which is + /// the case for data ingested before seals existed. + pub selected_source: Vec, /// Option letters the student eliminated, for elimination-scored items. pub eliminated: Vec, + /// The eliminated options in the bank's lettering. + pub eliminated_source: Vec, /// Whether the response earned full credit. `None` when it cannot be /// determined, e.g. a blank response on an item with no recorded key. pub correct: Option, @@ -92,17 +111,33 @@ pub struct Response { /// The item's level, denormalized so analysis need not carry the catalog. pub level: Option, /// The item's learning objectives, denormalized for per-objective mastery. - pub learning_objectives: Vec, + pub learning_targets: Vec, /// The item's topics, denormalized. pub topics: Vec, /// Whether the item was bonus, and so excluded from the scored total. pub bonus: bool, /// Whether the item was dropped after the fact. pub dropped: bool, + /// Whether that drop was applied by crediting every option, so the item is + /// still part of the points of record even though it is out of the + /// statistics. + /// + /// Written by [`ResponseSet::enrich`] from the placement's `dropped_as`. + #[serde(default)] + pub dropped_full_credit: bool, } impl Response { - /// Whether this row should count toward scored totals and item statistics. + /// Whether this row counts as evidence. + /// + /// Evidence means item statistics, objective mastery, level rates, and the + /// IRT fit. A dropped item is never evidence, however the drop was applied: + /// an item everyone was given has no variance to contribute and would only + /// flatter the objective it was written against. + /// + /// This is deliberately not the same question as [`Response::scores`]. The + /// two were one predicate until dropping a question stopped always meaning + /// removing it. /// /// # Returns /// @@ -111,6 +146,21 @@ impl Response { !self.bonus && !self.dropped } + /// Whether this row counts toward the points of record. + /// + /// A question dropped by crediting every option still sits in the student's + /// total on the platform, and a report that disagreed with the platform + /// about a student's percentage would be worse than no report. So a + /// full-credit drop stays in both the numerator and the denominator here, + /// while a removed drop leaves both. + /// + /// # Returns + /// + /// `true` when the row belongs in the score. + pub fn scores(&self) -> bool { + !self.bonus && (!self.dropped || self.dropped_full_credit) + } + /// The response coded for a dichotomous model. /// /// Partial credit is rounded toward the majority: a half-credit response is @@ -132,6 +182,19 @@ impl Response { pub fn selected_joined(&self) -> String { self.selected.join(",") } + + /// The choices, in the bank's lettering when it is known. + /// + /// `selected` is what the student marked on the page they held; this is what + /// they chose. On an unshuffled form the two agree, which is why reading the + /// wrong one is a bug that only appears once you add a second form. + pub fn chosen(&self) -> &[String] { + if self.selected_source.is_empty() { + &self.selected + } else { + &self.selected_source + } + } } /// A set of responses plus anything worth telling the user about the ingest. @@ -166,7 +229,11 @@ impl ResponseSet { set.into_iter().map(|s| s.to_string()).collect() } - /// The distinct item numbers that count toward the scored total, sorted. + /// The distinct item numbers that count as evidence, sorted. + /// + /// Named for the scored total it once described; it is the analysis matrix's + /// item list, so it uses [`Response::counts`] and excludes every dropped + /// item. [`ResponseSet::points_available`] is the scoring denominator. pub fn scored_items(&self) -> Vec { let set: BTreeSet = self .rows @@ -225,11 +292,12 @@ impl ResponseSet { /// /// # Returns /// - /// The sum of `score` over scored, undropped items. + /// The sum of `score` over the items that count toward the score, which + /// includes a question dropped by crediting every option. pub fn scored_total(&self, key: &str) -> f64 { self.rows .iter() - .filter(|r| r.student_key == key && r.counts()) + .filter(|r| r.student_key == key && r.scores()) .map(|r| r.score) .sum() } @@ -247,7 +315,7 @@ impl ResponseSet { /// for each item so a student who skipped an item still has a denominator. pub fn points_available(&self) -> f64 { let mut per_item: BTreeMap = BTreeMap::new(); - for r in self.rows.iter().filter(|r| r.counts()) { + for r in self.rows.iter().filter(|r| r.scores()) { let e = per_item.entry(r.item_number).or_insert(0.0); if r.points_possible > *e { *e = r.points_possible; @@ -256,6 +324,49 @@ impl ResponseSet { per_item.values().sum() } + /// How many rows each option set of each item has. + /// + /// What a calibration report needs to be honest about sample size: an item + /// administered three times with three different option sets has three + /// cells, not one, and reporting "n = 72" of it would be wrong three ways. + /// + /// # Returns + /// + /// Row counts keyed by item id and variant, with an empty variant for rows + /// that carry none. + pub fn variants(&self) -> BTreeMap<(String, String), usize> { + let mut out: BTreeMap<(String, String), usize> = BTreeMap::new(); + for row in &self.rows { + let Some(item) = &row.item_ref else { continue }; + let key = (item.clone(), row.variant.clone().unwrap_or_default()); + *out.entry(key).or_insert(0) += 1; + } + out + } + + /// The rows for one option set of one item. + /// + /// # Arguments + /// + /// * `item_ref` - the item id. + /// * `variant` - the variant digest, or `None` for rows carrying none. + /// + /// # Returns + /// + /// A set holding only those rows, keeping the warnings of the original. + pub fn for_variant(&self, item_ref: &str, variant: Option<&str>) -> ResponseSet { + ResponseSet { + rows: self + .rows + .iter() + .filter(|r| r.item_ref.as_deref() == Some(item_ref)) + .filter(|r| r.variant.as_deref() == variant) + .cloned() + .collect(), + warnings: self.warnings.clone(), + } + } + /// Builds the response matrix for psychometrics. /// /// # Arguments @@ -358,8 +469,12 @@ impl ResponseSet { }; r.item_ref = Some(p.item.clone()); r.item_version = p.version; + // Recorded when the record says so; derived from the option set + // otherwise, which is the case for every administration before 2.0. + r.variant = p.variant.clone(); r.bonus = r.bonus || p.bonus; r.dropped = r.dropped || p.dropped; + r.dropped_full_credit = r.dropped_full_credit || p.dropped_with_credit(); if let Some(points) = p.points { // The record is authoritative for points as administered; the // export sometimes carries a stale maximum. @@ -375,11 +490,14 @@ impl ResponseSet { if let Some(cat) = catalog { if let Some(entry) = cat.get(&p.item) { + if r.variant.is_none() { + r.variant = Some(p.variant_of(&entry.item)); + } r.level = Some(entry.item.level); - r.learning_objectives = if p.learning_objectives.is_empty() { - entry.item.learning_objectives.clone() + r.learning_targets = if p.learning_targets.is_empty() { + entry.item.learning_targets.clone() } else { - p.learning_objectives.clone() + p.learning_targets.clone() }; r.topics = entry.item.topics.clone(); if r.points_possible == 0.0 && !p.bonus { @@ -388,17 +506,28 @@ impl ResponseSet { } } else { r.level = p.level; - r.learning_objectives = p.learning_objectives.clone(); + r.learning_targets = p.learning_targets.clone(); } // Apply the record's credit overrides, which is how a decision to // award partial credit after the fact becomes visible in analysis. - if !p.credit_overrides.is_empty() && r.selected.len() == 1 { - if let Some(over) = p.credit_overrides.get(&r.selected[0]) { - if (*over - r.credit).abs() > 1e-9 { - r.credit = *over; - r.score = *over * r.points_possible; - r.correct = Some(*over >= 0.999); + // + // The override map is keyed in the bank's letters, so the lookup has + // to use `chosen()` rather than `selected`. Cloning the letter first + // ends the borrow of `r` before the row is written to. + let chosen = if r.chosen().len() == 1 { + r.chosen().first().cloned() + } else { + None + }; + if !p.credit_overrides.is_empty() { + if let Some(letter) = chosen { + if let Some(over) = p.credit_overrides.get(&letter) { + if (*over - r.credit).abs() > 1e-9 { + r.credit = *over; + r.score = *over * r.points_possible; + r.correct = Some(*over >= 0.999); + } } } } @@ -566,6 +695,8 @@ pub struct FlatResponse { pub item_ref: String, /// The item version, 0 when unknown. pub item_version: u32, + /// The administered variant, empty when unknown. + pub variant: String, /// Comma-joined selected letters. pub selected: String, /// Comma-joined eliminated letters. @@ -582,14 +713,35 @@ pub struct FlatResponse { pub response_time_seconds: String, /// The level code 1..5, 0 when unknown. pub level: u8, - /// Comma-joined objective ids. - pub learning_objectives: String, + /// Comma-joined target ids. + /// + /// The column keeps its original name. Renaming a column in a store that + /// already holds collected administrations would make last term's responses + /// unreadable, and no vocabulary improvement is worth that. + #[serde(rename = "learning_objectives")] + pub learning_targets: String, /// Comma-joined topics. pub topics: String, /// Whether the item was bonus. pub bonus: bool, /// Whether the item was dropped. pub dropped: bool, + + // Appended rather than interleaved: the columns above are the order every + // CSV written so far uses, and `#[serde(default)]` is what keeps one written + // last term readable now that three more exist. + /// The printed position, 0 when unknown. + #[serde(default)] + pub form_position: u32, + /// Comma-joined selected letters in the bank's lettering. + #[serde(default)] + pub selected_source: String, + /// Comma-joined eliminated letters in the bank's lettering. + #[serde(default)] + pub eliminated_source: String, + /// Whether a dropped item was dropped by crediting every option. + #[serde(default)] + pub dropped_full_credit: bool, } impl FlatResponse { @@ -610,6 +762,8 @@ impl FlatResponse { assessment_id: r.assessment_id.clone(), date: r.date.map(|d| d.to_string()).unwrap_or_default(), form: r.form.clone().unwrap_or_default(), + dropped_full_credit: r.dropped_full_credit, + form_position: r.form_position.unwrap_or(0), student_key: r.student_key.clone(), sid: r.sid.clone().unwrap_or_default(), email: r.email.clone().unwrap_or_default(), @@ -617,8 +771,11 @@ impl FlatResponse { item_number: r.item_number, item_ref: r.item_ref.clone().unwrap_or_default(), item_version: r.item_version.unwrap_or(0), + variant: r.variant.clone().unwrap_or_default(), selected: r.selected.join(","), + selected_source: r.selected_source.join(","), eliminated: r.eliminated.join(","), + eliminated_source: r.eliminated_source.join(","), correct: match r.correct { Some(true) => "1".to_string(), Some(false) => "0".to_string(), @@ -632,7 +789,7 @@ impl FlatResponse { .map(|s| format!("{s:.1}")) .unwrap_or_default(), level: r.level.map(|l| l.code()).unwrap_or(0), - learning_objectives: r.learning_objectives.join(","), + learning_targets: r.learning_targets.join(","), topics: r.topics.join(","), bonus: r.bonus, dropped: r.dropped, @@ -660,20 +817,32 @@ impl FlatResponse { assessment_id: self.assessment_id.clone(), date: self.date.parse().ok(), form: none_if_empty(&self.form), + dropped_full_credit: self.dropped_full_credit, + form_position: if self.form_position == 0 { + None + } else { + Some(self.form_position) + }, student_key: self.student_key.clone(), sid: none_if_empty(&self.sid), name: None, email: none_if_empty(&self.email), section: none_if_empty(&self.section), item_number: self.item_number, - item_ref: none_if_empty(&self.item_ref), + // Canonicalized on read, so a term ingested before 2.0 pools with + // one ingested after it instead of splitting into two items. + item_ref: none_if_empty(&self.item_ref) + .map(|id| crate::item::canonical_id(&id).to_string()), + variant: none_if_empty(&self.variant), item_version: if self.item_version == 0 { None } else { Some(self.item_version) }, selected: split(&self.selected), + selected_source: split(&self.selected_source), eliminated: split(&self.eliminated), + eliminated_source: split(&self.eliminated_source), correct: match self.correct.as_str() { "1" | "true" => Some(true), "0" | "false" => Some(false), @@ -684,7 +853,7 @@ impl FlatResponse { score: self.score, response_time_seconds: self.response_time_seconds.parse().ok(), level: Level::from_code(self.level), - learning_objectives: split(&self.learning_objectives), + learning_targets: split(&self.learning_targets), topics: split(&self.topics), bonus: self.bonus, dropped: self.dropped, @@ -713,6 +882,7 @@ mod tests { assessment_id: "e1".into(), date: None, form: None, + form_position: None, student_key: student.into(), sid: Some(format!("sid-{student}")), name: None, @@ -721,21 +891,50 @@ mod tests { item_number: number, item_ref: None, item_version: None, + variant: None, selected: vec!["A".into()], + selected_source: vec![], eliminated: vec![], + eliminated_source: vec![], correct: Some(credit >= 0.999), credit, points_possible: 2.0, score: credit * 2.0, response_time_seconds: None, level: None, - learning_objectives: vec![], + learning_targets: vec![], topics: vec![], bonus: false, dropped: false, + dropped_full_credit: false, } } + #[test] + fn rows_group_by_the_option_set_they_administered() { + let mut set = ResponseSet::new(); + for (student, variant) in [("s1", "v1"), ("s2", "v1"), ("s3", "v2")] { + let mut r = row(student, 1, 1.0); + r.item_ref = Some("q-x".into()); + r.variant = Some(variant.into()); + set.rows.push(r); + } + // A row from before the column existed. + let mut old = row("s4", 1, 1.0); + old.item_ref = Some("q-x".into()); + set.rows.push(old); + + let counts = set.variants(); + assert_eq!(counts[&("q-x".to_string(), "v1".to_string())], 2); + assert_eq!(counts[&("q-x".to_string(), "v2".to_string())], 1); + // Unknown groups on its own rather than joining either set. + assert_eq!(counts[&("q-x".to_string(), String::new())], 1); + + assert_eq!(set.for_variant("q-x", Some("v1")).rows.len(), 2); + assert_eq!(set.for_variant("q-x", None).rows.len(), 1); + assert_eq!(set.for_variant("q-other", Some("v1")).rows.len(), 0); + } + #[test] fn matrix_is_students_by_items() { let mut set = ResponseSet::new(); diff --git a/src/data/store.rs b/src/data/store.rs index f843193..55e8cae 100644 --- a/src/data/store.rs +++ b/src/data/store.rs @@ -49,21 +49,14 @@ impl Format { /// /// Parquet when the `parquet` feature is on, CSV otherwise. pub fn preferred() -> Format { - #[cfg(feature = "parquet")] - { - Format::Parquet - } - #[cfg(not(feature = "parquet"))] - { - Format::Csv - } + Format::Parquet } /// Whether this format can be written by the current build. pub fn is_available(self) -> bool { match self { Format::Csv => true, - Format::Parquet => cfg!(feature = "parquet"), + Format::Parquet => true, } } @@ -323,6 +316,70 @@ pub fn read_path(path: &Path) -> Result { } } +/// Reads a response file without turning its rows into [`Response`]s. +/// +/// What a migration wants: the rows exactly as they sit on disk, so rewriting +/// one column cannot disturb another through a round trip. +/// +/// # Arguments +/// +/// * `path` - the file to read. +/// +/// # Returns +/// +/// The rows. +/// +/// # Errors +/// +/// Returns [`Error::Other`] for an unrecognized extension, [`Error::Csv`] or +/// [`Error::Other`] on a parse failure, and [`Error::FeatureDisabled`] for +/// Parquet without the feature. +pub fn read_flat(path: &Path) -> Result> { + match Format::from_path(path) { + Some(Format::Csv) => { + let mut r = csv::Reader::from_path(path).map_err(|e| Error::Csv { + path: path.to_path_buf(), + source: e, + })?; + let mut out = Vec::new(); + for rec in r.deserialize::() { + out.push(rec.map_err(|e| Error::Csv { + path: path.to_path_buf(), + source: e, + })?); + } + Ok(out) + } + Some(Format::Parquet) => crate::store_parquet::read(path), + None => Err(Error::Other(format!( + "{} is not a response file; expected a .parquet or .csv", + path.display() + ))), + } +} + +/// Writes flat responses back to the file they came from. +/// +/// # Arguments +/// +/// * `path` - the destination, whose extension picks the format. +/// * `rows` - the rows. +/// +/// # Errors +/// +/// Returns [`Error::Other`] for an unrecognized extension and +/// [`Error::FeatureDisabled`] for Parquet without the feature. +pub fn write_flat(path: &Path, rows: &[FlatResponse]) -> Result<()> { + match Format::from_path(path) { + Some(Format::Csv) => write_csv(path, rows), + Some(Format::Parquet) => write_parquet(path, rows), + None => Err(Error::Other(format!( + "{} is not a response file; expected a .parquet or .csv", + path.display() + ))), + } +} + /// Writes flat responses as CSV. /// /// # Arguments @@ -386,21 +443,10 @@ fn read_csv(path: &Path) -> Result { /// /// * `path` - the destination. /// * `rows` - the rows. -/// -/// # Errors -/// -/// Returns [`Error::FeatureDisabled`] when the feature is off. -#[cfg(feature = "parquet")] fn write_parquet(path: &Path, rows: &[FlatResponse]) -> Result<()> { crate::store_parquet::write(path, rows) } -/// Stub for builds without Parquet support. -#[cfg(not(feature = "parquet"))] -fn write_parquet(_path: &Path, _rows: &[FlatResponse]) -> Result<()> { - Err(Error::FeatureDisabled("Parquet", "parquet")) -} - /// Reads flat responses from Parquet. /// /// # Arguments @@ -414,7 +460,6 @@ fn write_parquet(_path: &Path, _rows: &[FlatResponse]) -> Result<()> { /// # Errors /// /// Returns [`Error::FeatureDisabled`] when the feature is off. -#[cfg(feature = "parquet")] fn read_parquet(path: &Path) -> Result { let rows = crate::store_parquet::read(path)?; let mut set = ResponseSet::new(); @@ -422,16 +467,6 @@ fn read_parquet(path: &Path) -> Result { Ok(set) } -/// Stub for builds without Parquet support. -#[cfg(not(feature = "parquet"))] -fn read_parquet(path: &Path) -> Result { - Err(Error::Other(format!( - "{} is a Parquet file, but this build has Parquet support compiled out; \ - rebuild with `--features parquet`, or re-ingest with `--format csv`", - path.display() - ))) -} - /// Makes an administration id usable as a file name. /// /// # Arguments @@ -570,6 +605,7 @@ mod tests { assessment_id: "exam-4".into(), date: None, form: None, + form_position: None, student_key: student.into(), sid: None, name: None, @@ -578,18 +614,22 @@ mod tests { item_number: number, item_ref: Some("bank::q-x-001".into()), item_version: Some(2), + variant: None, selected: vec!["C".into()], + selected_source: vec![], eliminated: vec![], + eliminated_source: vec![], correct: Some(true), credit: 1.0, points_possible: 1.5, score: 1.5, response_time_seconds: None, level: None, - learning_objectives: vec!["lo-a".into()], + learning_targets: vec!["lo-a".into()], topics: vec![], bonus: false, dropped: false, + dropped_full_credit: false, } } @@ -616,9 +656,11 @@ mod tests { let back = store.read("BIOSC1540/2026s/exam-4").unwrap(); assert_eq!(back.rows.len(), 2); - assert_eq!(back.rows[0].item_ref.as_deref(), Some("bank::q-x-001")); + // Canonicalized on the way in: the row was written with a pre-2.0 + // `bank::item` key, and reading it yields the item it names. + assert_eq!(back.rows[0].item_ref.as_deref(), Some("q-x-001")); assert_eq!(back.rows[0].selected, vec!["C".to_string()]); - assert_eq!(back.rows[0].learning_objectives, vec!["lo-a".to_string()]); + assert_eq!(back.rows[0].learning_targets, vec!["lo-a".to_string()]); let all = store.read_all().unwrap(); assert_eq!(all.rows.len(), 2); diff --git a/src/data/store_parquet.rs b/src/data/store_parquet.rs index 32143fc..d5a7abf 100644 --- a/src/data/store_parquet.rs +++ b/src/data/store_parquet.rs @@ -49,6 +49,7 @@ pub fn schema() -> Schema { Field::new("item_number", DataType::UInt32, false), Field::new("item_ref", DataType::Utf8, false), Field::new("item_version", DataType::UInt32, false), + Field::new("variant", DataType::Utf8, false), Field::new("selected", DataType::Utf8, false), Field::new("eliminated", DataType::Utf8, false), Field::new("correct", DataType::Utf8, false), @@ -57,10 +58,16 @@ pub fn schema() -> Schema { Field::new("score", DataType::Float64, false), Field::new("response_time_seconds", DataType::Utf8, false), Field::new("level", DataType::UInt32, false), + // The stored column name predates the objective/target rename and is left + // alone: an existing parquet file has to keep loading. Field::new("learning_objectives", DataType::Utf8, false), Field::new("topics", DataType::Utf8, false), Field::new("bonus", DataType::Boolean, false), Field::new("dropped", DataType::Boolean, false), + Field::new("form_position", DataType::UInt32, false), + Field::new("dropped_full_credit", DataType::Boolean, false), + Field::new("selected_source", DataType::Utf8, false), + Field::new("eliminated_source", DataType::Utf8, false), ]) } @@ -108,6 +115,7 @@ fn to_batch(rows: &[FlatResponse]) -> Result { u32c(|r| r.item_number), s(|r| &r.item_ref), u32c(|r| r.item_version), + s(|r| &r.variant), s(|r| &r.selected), s(|r| &r.eliminated), s(|r| &r.correct), @@ -116,10 +124,14 @@ fn to_batch(rows: &[FlatResponse]) -> Result { f64c(|r| r.score), s(|r| &r.response_time_seconds), u32c(|r| r.level as u32), - s(|r| &r.learning_objectives), + s(|r| &r.learning_targets), s(|r| &r.topics), boolc(|r| r.bonus), boolc(|r| r.dropped), + u32c(|r| r.form_position), + boolc(|r| r.dropped_full_credit), + s(|r| &r.selected_source), + s(|r| &r.eliminated_source), ]; RecordBatch::try_new(Arc::new(schema()), columns).map_err(|e| { @@ -237,6 +249,25 @@ fn from_batch(batch: &RecordBatch, path: &Path) -> Result> { .ok_or_else(|| column_error(name, "boolean", path)) }; + // The three decode columns are read optionally rather than required, so a + // file written before seals existed still loads. A missing column is not a + // foreign file; it is last term's data. + let optional_strings = |name: &str| -> Option<&StringArray> { + batch + .column_by_name(name) + .and_then(|c| c.as_any().downcast_ref::()) + }; + let optional_uints = |name: &str| -> Option<&UInt32Array> { + batch + .column_by_name(name) + .and_then(|c| c.as_any().downcast_ref::()) + }; + let optional_bools = |name: &str| -> Option<&BooleanArray> { + batch + .column_by_name(name) + .and_then(|c| c.as_any().downcast_ref::()) + }; + let administration_id = strings("administration_id")?; let course = strings("course")?; let term = strings("term")?; @@ -250,6 +281,9 @@ fn from_batch(batch: &RecordBatch, path: &Path) -> Result> { let item_number = uints("item_number")?; let item_ref = strings("item_ref")?; let item_version = uints("item_version")?; + // Added after the first stores were written, so absent rather than fatal in + // a file from before 2.0; `coursebank migrate variants` fills it in. + let variant = optional_strings("variant"); let selected = strings("selected")?; let eliminated = strings("eliminated")?; let correct = strings("correct")?; @@ -258,11 +292,16 @@ fn from_batch(batch: &RecordBatch, path: &Path) -> Result> { let score = floats("score")?; let response_time_seconds = strings("response_time_seconds")?; let level = uints("level")?; - let learning_objectives = strings("learning_objectives")?; + let learning_targets = strings("learning_objectives")?; let topics = strings("topics")?; let bonus = bools("bonus")?; let dropped = bools("dropped")?; + let form_position = optional_uints("form_position"); + let dropped_full_credit = optional_bools("dropped_full_credit"); + let selected_source = optional_strings("selected_source"); + let eliminated_source = optional_strings("eliminated_source"); + let mut out = Vec::with_capacity(batch.num_rows()); for i in 0..batch.num_rows() { out.push(FlatResponse { @@ -272,6 +311,7 @@ fn from_batch(batch: &RecordBatch, path: &Path) -> Result> { assessment_id: assessment_id.value(i).to_string(), date: date.value(i).to_string(), form: form.value(i).to_string(), + form_position: form_position.map(|a| a.value(i)).unwrap_or(0), student_key: student_key.value(i).to_string(), sid: sid.value(i).to_string(), email: email.value(i).to_string(), @@ -279,18 +319,26 @@ fn from_batch(batch: &RecordBatch, path: &Path) -> Result> { item_number: item_number.value(i), item_ref: item_ref.value(i).to_string(), item_version: item_version.value(i), + variant: variant.map(|c| c.value(i).to_string()).unwrap_or_default(), selected: selected.value(i).to_string(), + selected_source: selected_source + .map(|a| a.value(i).to_string()) + .unwrap_or_default(), eliminated: eliminated.value(i).to_string(), + eliminated_source: eliminated_source + .map(|a| a.value(i).to_string()) + .unwrap_or_default(), correct: correct.value(i).to_string(), credit: credit.value(i), points_possible: points_possible.value(i), score: score.value(i), response_time_seconds: response_time_seconds.value(i).to_string(), level: level.value(i) as u8, - learning_objectives: learning_objectives.value(i).to_string(), + learning_targets: learning_targets.value(i).to_string(), topics: topics.value(i).to_string(), bonus: bonus.value(i), dropped: dropped.value(i), + dropped_full_credit: dropped_full_credit.map(|a| a.value(i)).unwrap_or(false), }); } Ok(out) @@ -324,6 +372,7 @@ mod tests { item_number: number, item_ref: "bank::q-a-001".into(), item_version: 3, + variant: "4c81fa".into(), selected: "C".into(), eliminated: String::new(), correct: "1".into(), @@ -332,10 +381,14 @@ mod tests { score: 1.5, response_time_seconds: "42.0".into(), level: 3, - learning_objectives: "lo-a,lo-b".into(), + learning_targets: "lo-a,lo-b".into(), topics: "kinetics".into(), bonus: false, dropped: false, + dropped_full_credit: false, + form_position: 0, + selected_source: String::new(), + eliminated_source: String::new(), } } @@ -360,7 +413,7 @@ mod tests { assert_eq!(back.len(), 2); assert_eq!(back[0].student_key, "s1"); assert_eq!(back[1].item_number, 2); - assert_eq!(back[0].learning_objectives, "lo-a,lo-b"); + assert_eq!(back[0].learning_targets, "lo-a,lo-b"); assert_eq!(back[0].credit, 1.0); assert_eq!(back[0].level, 3); assert!(!back[0].bonus); diff --git a/src/export.rs b/src/export.rs index 2a02566..c00dd60 100644 --- a/src/export.rs +++ b/src/export.rs @@ -8,7 +8,11 @@ //! |:--|:--|:--| //! | [`qti`] | a QTI 1.2 zip | importing into Canvas | //! | [`typst`] | `.typ` source | a printed exam, answer key, and bubble sheet | +//! | [`practice`] | Quarto Markdown | a worksheet and a solutions document, off Canvas | +//! | [`site`] | a Quarto partial and an encrypted bundle | a course page with password-gated solutions | //! | [`report`] | Markdown and HTML | students, and yourself | +//! | [`lecture`] | Markdown | the reading list on the course website | +//! | [`references`] | Hayagriva, CSL-JSON, BibTeX | Typst, Zotero, LaTeX | //! //! [`qti`] and [`typst`] share one rule that is easy to get wrong: a form's answer //! key must be generated from the same permutation that produced its question @@ -20,6 +24,10 @@ //! answers, other students' data, and any numeric rank. //! The instructor report answers "what should I fix?" and holds the item statistics. +pub mod lecture; +pub mod practice; pub mod qti; +pub mod references; pub mod report; +pub mod site; pub mod typst; diff --git a/src/export/lecture.rs b/src/export/lecture.rs new file mode 100644 index 0000000..b5b84a9 --- /dev/null +++ b/src/export/lecture.rs @@ -0,0 +1,645 @@ +// SPDX-License-Identifier: Prosperity-3.0.0 +// Copyright Scientific Computing Studio +// Source: https://git.scient.ing/education/coursebank + +//! Rendering a lecture's objectives and readings as Markdown. +//! +//! The course file is the source of truth for what a lecture assigns and why, so +//! the reading list on the course website is generated rather than kept in step by +//! hand. Two copies of the same prose drift within a term; one copy and a build +//! step do not. +//! +//! Two pages come out of here: the lecture's learning objectives with their +//! targets enumerated under each, and its readings. +//! +//! [`Style::Quarto`] reproduces the definition-list shape a Quarto page wants, +//! with `_(T 4, 7)_` numbering resolved from [`numbered_targets`]. Those numbers +//! are positional and so cannot be authored: inserting a target renumbers +//! everything after it. They are computed here and never stored. +//! +//! What this module does not do is invent prose. Everything printed comes from +//! `summary`, `focus`, and `skip` on the reading, in that order, and a reading with +//! none of the three renders as a bare citation. + +use crate::course::{CourseFile, Reading, ReadingRole, Reference}; +use crate::error::{Error, Result}; +use crate::taxonomy::Level; + +/// Which flavour of Markdown to emit. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub enum Style { + /// Pandoc definition lists with `
` before the objective line, which is + /// what a Quarto lecture page uses. + #[default] + Quarto, + /// Plain Markdown bullets, for a report or a README. + Plain, +} + +/// How a lecture's targets group under its objectives. +/// +/// One function feeds every page, so the printed order and the numbering cannot +/// come apart. +/// +/// # Arguments +/// +/// * `course` - the loaded course file. +/// * `lecture` - the lecture id. +/// +/// # Returns +/// +/// The groups as `(objective id, its targets in this lecture)`, and the +/// leftovers: targets whose objective this lecture does not teach, and +/// objectives that have no targets and so stand as their own. +fn target_layout<'a>( + course: &'a CourseFile, + lecture: &str, +) -> (Vec<(&'a str, Vec<&'a str>)>, Vec<&'a str>) { + let targets = course.lecture_targets(lecture); + let mut groups: Vec<(&str, Vec<&str>)> = Vec::new(); + for objective in course.lecture_objectives(lecture) { + let mine: Vec<&str> = course + .targets(objective) + .into_iter() + .filter(|t| targets.contains(t)) + .collect(); + if !mine.is_empty() { + groups.push((objective, mine)); + } + } + let grouped: Vec<&str> = groups + .iter() + .flat_map(|(_, ts)| ts.iter().copied()) + .collect(); + let leftovers: Vec<&str> = targets + .into_iter() + .filter(|t| !grouped.contains(t)) + .collect(); + (groups, leftovers) +} + +/// Whether a lecture's registry entries use the second tier. +/// +/// # Arguments +/// +/// * `course` - the loaded course file. +/// * `lecture` - the lecture id. +/// +/// # Returns +/// +/// `true` when any entry the lecture names is a target. +fn is_tiered(course: &CourseFile, lecture: &str) -> bool { + course + .lecture_entries(lecture) + .iter() + .any(|id| course.is_target(id)) +} + +/// The entries a lecture's pages number, in the order they print. +/// +/// Numbering is a property of the *targets* page, because that is the tier a +/// reading serves: a section of a textbook backs a specific performance, not a +/// whole objective. An untiered lecture numbers its objectives instead, since +/// there each objective is its own target. +/// +/// # Arguments +/// +/// * `course` - the loaded course file. +/// * `lecture` - the lecture id. +/// +/// # Returns +/// +/// Registry ids in printed order. +pub fn numbered_targets(course: &CourseFile, lecture: &str) -> Vec { + if !is_tiered(course, lecture) { + // Untiered page: level groups, in the order they print, which is the + // numbering this page has always had. + return by_level(course, &course.lecture_entries(lecture)) + .into_iter() + .flat_map(|(_, members)| members) + .map(str::to_string) + .collect(); + } + let (groups, leftovers) = target_layout(course, lecture); + groups + .into_iter() + .flat_map(|(_, targets)| targets) + .chain(leftovers) + .map(str::to_string) + .collect() +} + +/// Groups ids by the level they are assessed up to, in taxonomy order. +/// +/// # Arguments +/// +/// * `course` - the loaded course file. +/// * `ids` - the ids to group. +/// +/// # Returns +/// +/// Non-empty groups, with whatever declares no ceiling last under `None`. +fn by_level<'a>(course: &CourseFile, ids: &[&'a str]) -> Vec<(Option, Vec<&'a str>)> { + let mut groups: Vec<(Option, Vec<&str>)> = + Level::ALL.iter().map(|l| (Some(*l), Vec::new())).collect(); + groups.push((None, Vec::new())); + for id in ids { + let ceiling = course.effective_level_ceiling(id); + if let Some(slot) = groups.iter_mut().find(|(level, _)| *level == ceiling) { + slot.1.push(id); + } + } + groups.retain(|(_, members)| !members.is_empty()); + groups +} + +/// Renders the learning objectives for one lecture, with each objective's +/// targets enumerated beneath it. +/// +/// One page rather than two. The objective and its targets are the same claim +/// at two grains, so printing them in separate sections would list every target +/// twice and leave a reader matching them up by hand. +/// +/// The objective is set in bold rather than as a heading, and the targets are +/// numbered: the numbers are what a reading's `_(T 4, 7)_` points at, and they +/// run continuously down the page rather than restarting under each objective, +/// because a reading cites a target without caring which objective it serves. +/// Pandoc's `(@)` example lists continue numbering across the paragraphs +/// between them, which is why the objective lines can sit in the middle of the +/// sequence without breaking it. +/// +/// An objective with no targets of its own is printed as a numbered line +/// instead, since it stands as its own target and a reading may cite it. +/// Bolding it and then repeating it as its only target is the duplication this +/// layout exists to avoid. +/// +/// On a lecture whose entries are all objectives, the output is what it has +/// always been, grouped by level. +/// +/// # Arguments +/// +/// * `course` - the loaded course file. +/// * `lecture` - the lecture id. +/// * `style` - which flavour to emit. +/// +/// # Returns +/// +/// The Markdown, ending in a newline. On an untiered lecture, objectives with +/// no `level_ceiling` are grouped last under no heading. +/// +/// # Errors +/// +/// Returns [`Error::Unresolved`] when the lecture id is not registered. +pub fn objectives_markdown(course: &CourseFile, lecture: &str, style: Style) -> Result { + course.lecture(lecture, "lecture page")?; + + let mut out = String::from("## Learning objectives\n\n"); + out.push_str("After this lecture, you should be able to do the following.\n\n"); + + if !is_tiered(course, lecture) { + // Levels in taxonomy order, then whatever declares no ceiling. + for (level, members) in by_level(course, &course.lecture_entries(lecture)) { + if let Some(level) = level { + out.push_str(&format!("### {}\n\n", level.name())); + } + for id in members { + out.push_str(&bullet(&course.text_for(id), style)); + } + out.push('\n'); + } + return Ok(out); + } + + let (groups, leftovers) = target_layout(course, lecture); + for (objective, targets) in groups { + out.push_str(&format!("**{}**\n\n", course.text_for(objective))); + for target in targets { + out.push_str(&bullet(&course.text_for(target), style)); + } + out.push('\n'); + } + // Targets whose objective this lecture does not teach, and objectives with + // no targets, in the order `numbered_targets` expects them. + if !leftovers.is_empty() { + for id in leftovers { + out.push_str(&bullet(&course.text_for(id), style)); + } + out.push('\n'); + } + Ok(out) +} + +/// One list item, numbered so a reading can point at it. +/// +/// # Arguments +/// +/// * `text` - the objective or target text. +/// * `style` - which flavour to emit. +/// +/// # Returns +/// +/// The line, ending in a newline. +fn bullet(text: &str, style: Style) -> String { + match style { + Style::Quarto => format!("(@) {text}\n"), + Style::Plain => format!("1. {text}\n"), + } +} + +/// Renders the readings for one lecture. +/// +/// # Arguments +/// +/// * `course` - the loaded course file. +/// * `lecture` - the lecture id, such as `L1.1`. +/// * `style` - which flavour to emit. +/// +/// # Returns +/// +/// The Markdown, ending in a newline. Supplemental readings follow the assigned +/// ones under their own subheading, and are omitted entirely when there are none. +/// +/// # Errors +/// +/// Returns [`Error::Unresolved`] when the lecture id or a cited reference is not +/// registered. +pub fn readings_markdown(course: &CourseFile, lecture: &str, style: Style) -> Result { + let lec = course.lecture(lecture, "lecture page")?; + + // Positional numbers for this page, so `{lo-id}` in a note and the trailing + // `_(T ...)_` agree with the enumerated targets printed above them. + let order = numbered_targets(course, lecture); + let number = |id: &str| order.iter().position(|o| o == id).map(|i| i + 1); + + let mut out = String::from("## Readings\n\n"); + for role in [ReadingRole::Assigned, ReadingRole::Supplemental] { + let group: Vec<&Reading> = lec.readings.iter().filter(|r| r.role == role).collect(); + if group.is_empty() { + continue; + } + if role == ReadingRole::Supplemental { + out.push_str("### Supplemental\n\n"); + } + for reading in group { + out.push_str(&entry(course, reading, style, &number)?); + } + } + Ok(out) +} + +/// Renders one reading. +/// +/// # Arguments +/// +/// * `course` - the course, for resolving references and placeholders. +/// * `reading` - the reading. +/// * `style` - which flavour to emit. +/// * `number` - the position of an objective on this page, if it has one. +/// +/// # Returns +/// +/// The entry, followed by a blank line. +/// +/// # Errors +/// +/// Returns [`Error::Unresolved`] when the cited reference is not registered. +fn entry( + course: &CourseFile, + reading: &Reading, + style: Style, + number: &impl Fn(&str) -> Option, +) -> Result { + // A reading carried over from the old string form has nothing to resolve. + if let (None, Some(text)) = (&reading.reference, &reading.text) { + return Ok(match style { + Style::Quarto => format!("{text}\n\n"), + Style::Plain => format!("- {text}\n"), + }); + } + let key = reading + .reference + .as_deref() + .ok_or_else(|| Error::other("reading has neither a reference nor text"))?; + let reference = course.reference(key, "lecture page")?; + + let mut out = String::new(); + out.push_str(&heading(reading, key, reference, style)); + + // The three prose fields in the order a reader wants them: what it is, what to + // take from it, what to leave. + let body: Vec = [&reading.summary, &reading.focus, &reading.skip] + .into_iter() + .flatten() + .map(|prose| { + course.expand_objective_refs(prose, |id| match number(id) { + Some(n) => format!("T {n}"), + None => course.text_for(id), + }) + }) + .collect(); + + match style { + Style::Quarto => { + if !body.is_empty() { + out.push_str(&format!(": {}\n", body.join("\n"))); + } + let mut numbers: Vec = + reading.targets.iter().filter_map(|o| number(o)).collect(); + numbers.sort_unstable(); + if !numbers.is_empty() { + let list: Vec = numbers.iter().map(|n| n.to_string()).collect(); + out.push_str(&format!("
\n_(T {})_\n", list.join(", "))); + } + out.push('\n'); + } + Style::Plain => { + if !body.is_empty() { + out.push_str(&format!(" {}\n", body.join(" "))); + } + } + } + Ok(out) +} + +/// The citation line that opens an entry. +/// +/// # Arguments +/// +/// * `reading` - the reading. +/// * `key` - its citation key. +/// * `reference` - the cited work. +/// * `style` - which flavour to emit. +/// +/// # Returns +/// +/// A linked citation when the location has a URL, and a plain one when it does not. +fn heading(reading: &Reading, key: &str, reference: &Reference, style: Style) -> String { + let label = reference.label_or(key); + let locator = reading.locator.as_deref().unwrap_or(""); + let linked = match reading.resolve_url(reference) { + Some(url) if !locator.is_empty() => format!("[{locator}]({url})"), + Some(url) => format!("[{}]({url})", reference.title), + None if !locator.is_empty() => locator.to_string(), + None => reference.title.clone(), + }; + match style { + Style::Quarto => format!("`{label}` {linked}\n"), + Style::Plain => format!("- **{label}** {linked}\n"), + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::course::{Lecture, Objective, ReferenceRole, Target}; + + /// A course with one lecture, two objectives, and one reference. + fn course() -> CourseFile { + let mut c = CourseFile::skeleton("BIOSC 1000", "Biochemistry", "2026f"); + c.lectures.clear(); + c.learning_objectives.clear(); + c.learning_targets.clear(); + + c.references.insert( + "kuriyan2013molecules".into(), + Reference { + label: Some("KKW".into()), + role: ReferenceRole::Required, + title: "The molecules of life".into(), + base_url: Some("https://example.org/kkw/".into()), + ..Reference::default() + }, + ); + for (id, order) in [("lo-second", 2), ("lo-first", 1)] { + c.learning_objectives.insert( + id.into(), + Objective { + text: format!("objective {id}"), + lectures: vec!["L1.1".into()], + order: Some(order), + ..objective_defaults() + }, + ); + } + c.lectures.insert( + "L1.1".into(), + Lecture { + title: "Enthalpy".into(), + date: None, + unit: None, + slides_url: None, + teaches: Vec::new(), + readings: vec![ + Reading { + reference: Some("kuriyan2013molecules".into()), + locator: Some("§6.1".into()), + path: Some("6/A/#1".into()), + targets: vec!["lo-second".into(), "lo-first".into()], + summary: Some("What a system is.".into()), + focus: Some("A worked instance of {lo-first}.".into()), + ..Reading::default() + }, + Reading { + reference: Some("kuriyan2013molecules".into()), + locator: Some("§1.9".into()), + path: Some("1/B/#9".into()), + role: ReadingRole::Supplemental, + targets: vec!["lo-second".into()], + summary: Some("Background.".into()), + ..Reading::default() + }, + ], + }, + ); + c + } + + /// The non-defaulted half of an objective, so the fixtures stay short. + fn objective_defaults() -> Objective { + Objective { + text: String::new(), + unit: None, + lectures: Vec::new(), + order: None, + level_ceiling: None, + prerequisites: Vec::new(), + tags: Vec::new(), + assessed: true, + } + } + + #[test] + fn the_quarto_form_matches_the_page_it_replaces() { + let md = readings_markdown(&course(), "L1.1", Style::Quarto).expect("renders"); + let expected = "\ +## Readings + +`KKW` [§6.1](https://example.org/kkw/6/A/#1) +: What a system is. +A worked instance of T 1. +
+_(T 1, 2)_ + +### Supplemental + +`KKW` [§1.9](https://example.org/kkw/1/B/#9) +: Background. +
+_(T 2)_ + +"; + assert_eq!(md, expected); + } + + #[test] + fn target_numbers_follow_teaching_order_not_id_order() { + // `lo-second` sorts first alphabetically and second by `order`. + let md = readings_markdown(&course(), "L1.1", Style::Quarto).expect("renders"); + assert!(md.contains("_(T 1, 2)_")); + assert!(md.contains("A worked instance of T 1.")); + } + + #[test] + fn objectives_group_by_level_in_taxonomy_order() { + let mut c = course(); + c.learning_objectives + .get_mut("lo-first") + .expect("fixture") + .level_ceiling = Some(Level::Remember); + c.learning_objectives + .get_mut("lo-second") + .expect("fixture") + .level_ceiling = Some(Level::Apply); + let md = objectives_markdown(&c, "L1.1", Style::Quarto).expect("renders"); + let expected = "\ +## Learning objectives + +After this lecture, you should be able to do the following. + +### Remember + +(@) objective lo-first + +### Apply + +(@) objective lo-second + +"; + assert_eq!(md, expected); + } + + /// The fixture course with its two entries moved into the target registry + /// under a new objective, which is the shape a migrated course has. + fn tiered_course() -> CourseFile { + let mut c = course(); + c.learning_objectives.insert( + "lo-binding".into(), + Objective { + text: "Quantify binding.".into(), + lectures: vec!["L1.1".into()], + order: Some(0), + ..objective_defaults() + }, + ); + // The readings cite `lo-second` and `lo-first`, so the ids are kept and + // only the registry they live in changes. A real migration renames them + // to the `t-` spelling and rewrites the citations with them. + for (id, order) in [("lo-first", 1), ("lo-second", 2)] { + let objective = c.learning_objectives.remove(id).expect("fixture"); + c.learning_targets.insert( + id.into(), + Target { + text: objective.text, + objective: "lo-binding".into(), + lectures: objective.lectures, + order: Some(order), + level_ceiling: None, + prerequisites: Vec::new(), + tags: Vec::new(), + assessed: true, + }, + ); + } + c + } + + #[test] + fn the_page_bolds_each_objective_and_enumerates_its_targets() { + let md = objectives_markdown(&tiered_course(), "L1.1", Style::Quarto).expect("renders"); + let expected = "\ +## Learning objectives + +After this lecture, you should be able to do the following. + +**Quantify binding.** + +(@) objective lo-first +(@) objective lo-second + +"; + assert_eq!(md, expected); + // Each target appears once on the page, which is the point of merging + // the two sections. + assert_eq!(md.matches("objective lo-first").count(), 1); + } + + #[test] + fn reading_numbers_point_at_the_enumerated_targets() { + // The numbers on the page and the ones a reading cites come from the + // same function, so they cannot drift apart. + let readings = readings_markdown(&tiered_course(), "L1.1", Style::Quarto).expect("renders"); + assert!(readings.contains("_(T 1, 2)_"), "{readings}"); + assert!(readings.contains("A worked instance of T 1.")); + } + + #[test] + fn an_untiered_lecture_is_grouped_by_level_as_before() { + // Where no targets are declared, each objective is its own target and + // there is nothing to nest, so the page keeps its level headings. + let md = objectives_markdown(&course(), "L1.1", Style::Quarto).expect("renders"); + assert!(md.contains("objective lo-first"), "{md}"); + assert!(md.contains("objective lo-second"), "{md}"); + assert!( + !md.contains("**"), + "nothing to bold on an untiered page: {md}" + ); + } + + #[test] + fn an_objective_with_no_level_still_appears() { + // Ungrouped, at the end, rather than silently dropped. + let md = objectives_markdown(&course(), "L1.1", Style::Quarto).expect("renders"); + assert!(md.contains("(@) objective lo-first")); + assert!(md.contains("(@) objective lo-second")); + assert!(!md.contains("###")); + } + + #[test] + fn a_supplemental_heading_appears_only_when_something_is_under_it() { + let mut c = course(); + c.lectures.get_mut("L1.1").expect("lecture").readings.pop(); + let md = readings_markdown(&c, "L1.1", Style::Quarto).expect("renders"); + assert!(!md.contains("Supplemental")); + } + + #[test] + fn a_legacy_string_reading_still_renders() { + let mut c = course(); + let readings = &mut c.lectures.get_mut("L1.1").expect("lecture").readings; + readings.clear(); + readings.push(Reading { + text: Some("KKW §6.1: system and surroundings. https://example.org".into()), + ..Reading::default() + }); + let md = readings_markdown(&c, "L1.1", Style::Quarto).expect("renders"); + assert!(md.contains("system and surroundings")); + } + + #[test] + fn an_unknown_reference_is_an_error_rather_than_a_blank() { + let mut c = course(); + c.references.clear(); + let err = readings_markdown(&c, "L1.1", Style::Quarto); + assert!(err.is_err()); + } +} diff --git a/src/export/practice.rs b/src/export/practice.rs new file mode 100644 index 0000000..d9a5b13 --- /dev/null +++ b/src/export/practice.rs @@ -0,0 +1,673 @@ +// SPDX-License-Identifier: Prosperity-3.0.0 +// Copyright Scientific Computing Studio +// Source: https://git.scient.ing/education/coursebank + +//! Rendering an assessment as a Quarto worksheet a student can work through, and +//! a matching solutions document they can learn from. +//! +//! This is the path that does not go through Canvas. You assemble a homework, +//! quiz, or practice set the same way you assemble an exam, then render it as two +//! `.qmd` files: [`Variant::Worksheet`] holds the questions and nothing else, and +//! [`Variant::Solutions`] holds the same questions with the key marked, the worked +//! reasoning, the rubric for anything open-ended, and where to read again. A +//! student with neither the Canvas quiz nor the printed exam can still practice +//! from the worksheet and check themselves against the solutions. +//! +//! A worksheet never contains the answer. It is built only from stems and +//! options, and the option letters are the printed positions, so the document has +//! nothing in it to leak: not a `correct` flag, not a solution, not a rationale. +//! [`Variant::Solutions`] is a separate render from the same input. +//! +//! Option order comes from the form's seed. When a form shuffles, both +//! documents relabel to the printed order through +//! [`select::option_order`], so a worksheet handed to +//! a student who saw form B agrees with the form B solutions. +//! +//! Everything a solution shows is authored: the model answer, the explanation, the +//! per-option notes, the rubric, and the review citations. Nothing is invented +//! here. A question with an empty [`crate::item::Solution`] renders its key and +//! stops, which is a visible cue to go finish writing it. + +use crate::assessment::{AssessmentFile, Form, Placement}; +use crate::catalog::Catalog; +use crate::course::CourseFile; +use crate::error::Result; +use crate::item::{Choice, Citation, Item}; +use crate::markup; +use crate::select; + +/// Which of the two documents to render. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub enum Variant { + /// Questions only, for a student to work through. + #[default] + Worksheet, + /// Questions with the key, worked solutions, rubric, and readings. + Solutions, +} + +impl Variant { + /// Both documents, in the order they are usually written. + pub const ALL: [Variant; 2] = [Variant::Worksheet, Variant::Solutions]; + + /// The token used on the command line and in a file name. + pub fn as_str(self) -> &'static str { + match self { + Variant::Worksheet => "worksheet", + Variant::Solutions => "solutions", + } + } + + /// The suffix a generated file name carries, e.g. `-solutions`. + pub fn suffix(self) -> &'static str { + match self { + Variant::Worksheet => "", + Variant::Solutions => "-solutions", + } + } + + /// The word for this document in a title. + fn title_word(self) -> &'static str { + match self { + Variant::Worksheet => "Questions", + Variant::Solutions => "Solutions", + } + } + + /// Parses a `--variant` value. + /// + /// # Arguments + /// + /// * `name` - the token, case insensitive; `questions` is accepted for the + /// worksheet and `key` for the solutions, since those are what people type. + /// + /// # Returns + /// + /// The variant. + /// + /// # Errors + /// + /// Returns [`crate::error::Error::Usage`] naming the valid tokens. + pub fn parse(name: &str) -> Result { + match name.trim().to_ascii_lowercase().as_str() { + "worksheet" | "questions" | "q" => Ok(Variant::Worksheet), + "solutions" | "solution" | "key" => Ok(Variant::Solutions), + other => Err(crate::error::Error::usage(format!( + "unknown practice document `{other}`; use worksheet or solutions" + ))), + } + } +} + +/// What to render. +#[derive(Debug, Clone)] +pub struct Options { + /// Which form's ordering to use. Defaults to an unshuffled form. + pub form: Form, + /// Which document. + pub variant: Variant, + /// Leave vertical space after each question on the worksheet for a written + /// answer. Ignored for the solutions document. + pub answer_space: bool, +} + +impl Default for Options { + fn default() -> Options { + Options { + form: Form { + id: "A".to_string(), + seed: 0, + shuffle_items: false, + shuffle_options: false, + }, + variant: Variant::Worksheet, + answer_space: true, + } + } +} + +impl Options { + /// Options for one variant on one form. + /// + /// # Arguments + /// + /// * `variant` - which document. + /// * `form` - the form whose ordering to use. + /// + /// # Returns + /// + /// The options, with the answer space on. + pub fn new(variant: Variant, form: Form) -> Options { + Options { + form, + variant, + answer_space: true, + } + } +} + +/// Renders the questions-only worksheet. +/// +/// # Arguments +/// +/// * `catalog` - the loaded course. +/// * `record` - the assessment record. +/// * `form` - the form whose ordering to use. +/// +/// # Returns +/// +/// The Quarto Markdown, ending in a newline. +/// +/// # Errors +/// +/// Returns [`crate::error::Error::Unresolved`] when a placement references a +/// missing item. +pub fn worksheet(catalog: &Catalog, record: &AssessmentFile, form: &Form) -> Result { + render( + catalog, + record, + &Options::new(Variant::Worksheet, form.clone()), + ) +} + +/// Renders the solutions document. +/// +/// # Arguments +/// +/// * `catalog` - the loaded course. +/// * `record` - the assessment record. +/// * `form` - the form whose ordering to use. +/// +/// # Returns +/// +/// The Quarto Markdown, ending in a newline. +/// +/// # Errors +/// +/// As [`worksheet`]. +pub fn solutions(catalog: &Catalog, record: &AssessmentFile, form: &Form) -> Result { + render( + catalog, + record, + &Options::new(Variant::Solutions, form.clone()), + ) +} + +/// Renders one document. +/// +/// # Arguments +/// +/// * `catalog` - the loaded course. +/// * `record` - the assessment record. +/// * `opts` - what to render. +/// +/// # Returns +/// +/// The Quarto Markdown, ending in a newline. +/// +/// # Errors +/// +/// Returns [`crate::error::Error::Unresolved`] when a placement references a +/// missing item. +pub fn render(catalog: &Catalog, record: &AssessmentFile, opts: &Options) -> Result { + let course = &catalog.course; + let mut out = front_matter(course, record, opts.variant); + + if let Some(instructions) = &record.assessment.instructions { + out.push_str(&markup::to_markdown(instructions)); + out.push_str("\n\n"); + } + + // One shared stimulus is printed once, above the first question that uses it, + // so a testlet reads as a block rather than repeating the vignette per item. + let mut printed_stimulus: Option = None; + + let layout = select::layout(record, &opts.form); + for (position, placement) in layout.iter().filter(|p| p.was_printed()).enumerate() { + let entry = catalog.require(&placement.item)?; + let item = &entry.item; + let number = position + 1; + + if let Some(stimulus_id) = &item.stimulus { + if printed_stimulus.as_deref() != Some(stimulus_id.as_str()) { + if let Some(stimulus) = course.stimuli.get(stimulus_id) { + out.push_str("::: {.stimulus}\n\n"); + out.push_str(&markup::to_markdown(&stimulus.body)); + out.push_str("\n\n:::\n\n"); + } + printed_stimulus = Some(stimulus_id.clone()); + } + } + + match opts.variant { + Variant::Worksheet => worksheet_question( + &mut out, + number, + placement, + item, + &opts.form, + opts.answer_space, + ), + Variant::Solutions => { + solution_question(&mut out, number, placement, item, &opts.form, course) + } + } + } + + Ok(out) +} + +/// The Quarto YAML front matter. +fn front_matter(course: &CourseFile, record: &AssessmentFile, variant: Variant) -> String { + // Both documents name themselves. Two PDFs called "Homework 1" are + // indistinguishable in a downloads folder, which is how a solutions copy + // gets posted in place of the worksheet. + let title = format!("{}: {}", record.assessment.title, variant.title_word()); + let subtitle = format!("{} · {}", course.course.code, course.course.title); + let mut out = String::from("---\n"); + out.push_str(&format!("title: \"{}\"\n", yaml_quote(&title))); + out.push_str(&format!("subtitle: \"{}\"\n", yaml_quote(&subtitle))); + if let Some(date) = record.assessment.date { + out.push_str(&format!("date: \"{date}\"\n")); + } + out.push_str("format:\n html:\n toc: false\n number-sections: false\n"); + out.push_str("---\n\n"); + out +} + +/// One question on the worksheet: stem, options in printed order, no answer. +fn worksheet_question( + out: &mut String, + number: usize, + placement: &Placement, + item: &Item, + form: &Form, + answer_space: bool, +) { + out.push_str(&heading(number, placement)); + out.push_str(&markup::to_markdown(&item.stem)); + out.push_str("\n\n"); + + if item.has_options() { + let ordered = ordered_options(item, placement, form); + for (position, source) in ordered.iter().enumerate() { + out.push_str(&format!( + "{}. {}\n", + letter(position), + markup::to_markdown(&source.text) + )); + } + out.push('\n'); + } else if answer_space { + // A place to write, sized by the theme, present only when asked for. + out.push_str("::: {.answer-space}\n:::\n\n"); + } +} + +/// One question in the solutions document: stem, key, worked reasoning, rubric, +/// and where to look again. +fn solution_question( + out: &mut String, + number: usize, + placement: &Placement, + item: &Item, + form: &Form, + course: &CourseFile, +) { + out.push_str(&heading(number, placement)); + out.push_str(&meta_line(placement, item)); + out.push_str(&markup::to_markdown(&item.stem)); + out.push_str("\n\n"); + + if item.has_options() { + let ordered = ordered_options(item, placement, form); + for (position, source) in ordered.iter().enumerate() { + let mark = if source.correct { " ✓" } else { "" }; + let note = source + .student_text() + .map(|t| format!(": {}", markup::to_markdown(t))) + .unwrap_or_default(); + out.push_str(&format!( + "{}. {}{mark}{note}\n", + letter(position), + markup::to_markdown(&source.text) + )); + } + out.push('\n'); + } + + solution_body(out, item); + objectives_line(out, item, course); + review_line(out, item, course); + out.push('\n'); +} + +/// The model answer, explanation, rubric, and accepted answers, when present. +fn solution_body(out: &mut String, item: &Item) { + let Some(solution) = item.solution.as_ref().filter(|s| !s.is_empty()) else { + if !item.has_options() { + // An open-response question with no written solution is unfinished, and + // saying so in the document is more useful than a silent blank. + out.push_str("_No solution written yet._\n\n"); + } + return; + }; + + if let Some(answer) = &solution.model_answer { + out.push_str(&format!( + "**Model answer.** {}\n\n", + markup::to_markdown(answer) + )); + } + if let Some(explanation) = &solution.explanation { + out.push_str(&markup::to_markdown(explanation)); + out.push_str("\n\n"); + } + if !solution.rubric.is_empty() { + out.push_str("**Rubric**\n\n"); + for criterion in &solution.rubric { + let points = criterion + .points + .map(|p| format!(" ({} pt)", trim_number(p))) + .unwrap_or_default(); + out.push_str(&format!( + "- {}{points}\n", + markup::to_markdown(&criterion.description) + )); + } + out.push('\n'); + } + if !solution.accepted.is_empty() { + let joined: Vec = solution + .accepted + .iter() + .map(|a| markup::to_markdown(a)) + .collect(); + out.push_str(&format!("**Accepted answers:** {}\n\n", joined.join("; "))); + } +} + +/// The `Tests:` line naming the objectives this item measures. +fn objectives_line(out: &mut String, item: &Item, course: &CourseFile) { + if item.learning_targets.is_empty() { + return; + } + let texts: Vec = item + .learning_targets + .iter() + .map(|id| course.text_for(id)) + .collect(); + out.push_str(&format!("**Tests:** {}\n\n", texts.join("; "))); +} + +/// The `Review:` line, resolving each citation to a short label, linked when a URL +/// resolves. +fn review_line(out: &mut String, item: &Item, course: &CourseFile) { + let Some(solution) = item.solution.as_ref() else { + return; + }; + if solution.review.is_empty() { + return; + } + let cites: Vec = solution.review.iter().map(|c| cite(course, c)).collect(); + out.push_str(&format!("**Review:** {}\n\n", cites.join("; "))); +} + +/// Resolves one citation to Markdown, mirroring the lecture reading style +/// `` `KKW` [§6.1](url) ``. +fn cite(course: &CourseFile, citation: &Citation) -> String { + if let Some(text) = &citation.text { + if citation.reference.is_none() { + return text.clone(); + } + } + let Some(key) = &citation.reference else { + return citation.display(); + }; + let Some(reference) = course.references.get(key) else { + return citation.display(); + }; + let label = reference.label_or(key); + let locator = citation.locator.as_deref().unwrap_or(""); + match citation.href(reference) { + Some(url) if !locator.is_empty() => format!("`{label}` [{locator}]({url})"), + Some(url) => format!("`{label}` [{}]({url})", reference.title), + None if !locator.is_empty() => format!("`{label}` {locator}"), + None => format!("`{label}`"), + } +} + +/// The `## Question N` heading, marking a bonus item. +fn heading(number: usize, placement: &Placement) -> String { + let bonus = if placement.bonus { " (bonus)" } else { "" }; + format!("## Question {number}{bonus}\n\n") +} + +/// The italic level-and-points line under a solutions heading. +fn meta_line(placement: &Placement, item: &Item) -> String { + let level = placement.level.unwrap_or(item.level); + let mut parts = vec![format!("Level {} ({})", level.code(), level.name())]; + if let Some(points) = placement.points { + parts.push(format!("{} point(s)", trim_number(points))); + } + format!("_{}_\n\n", parts.join(" · ")) +} + +/// The options in the order the form prints them. +/// +/// Salted with the item's global id, the same value the Typst and QTI exports use, +/// so a worksheet built for form B lists options in the order that form's paper and +/// its Canvas quiz do. +fn ordered_options<'a>(item: &'a Item, placement: &Placement, form: &Form) -> Vec<&'a Choice> { + let shown = item.administered(&placement.key, &placement.distractors); + select::option_order(form, &placement.item, shown.len()) + .into_iter() + .map(|i| shown[i]) + .collect() +} + +/// The printed letter for a zero-based position. +fn letter(position: usize) -> char { + (b'A' + (position as u8 % 26)) as char +} + +/// Formats a point value without a trailing `.0`. +fn trim_number(value: f64) -> String { + if value.fract() == 0.0 { + format!("{}", value as i64) + } else { + let s = format!("{value:.2}"); + s.trim_end_matches('0').trim_end_matches('.').to_string() + } +} + +/// Escapes a double quote for a YAML double-quoted scalar. +fn yaml_quote(s: &str) -> String { + s.replace('\\', "\\\\").replace('"', "\\\"") +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::assessment::{Assessment, Kind, Platform}; + + /// Writes a course and a bank to a temp directory and loads them, the same way + /// the catalog tests do, so this exercises only public API. The `tag` keeps each + /// test in its own directory, so tests running in parallel do not clobber a + /// shared `course.yaml`. + fn catalog(tag: &str) -> Catalog { + let dir = std::env::temp_dir().join(format!("cb-practice-{tag}-{}", std::process::id())); + let _ = std::fs::remove_dir_all(&dir); + std::fs::create_dir_all(dir.join("banks")).unwrap(); + std::fs::write( + dir.join("course.yaml"), + r#" +course: { code: BIOSC 1000, title: Biochemistry, term: 2026f } +references: + kkw: + label: KKW + title: The molecules of life + base_url: https://example.org/kkw/ +lectures: + L1.1: { title: Enthalpy } +learning_objectives: + lo-enthalpy: + text: Define enthalpy and explain the constant-pressure result. + lectures: [L1.1] + order: 1 +"#, + ) + .unwrap(); + std::fs::write( + dir.join("banks").join("l11.yaml"), + r#" +bank: { id: l11, title: L1.1 } +items: + - id: q-enthalpy-001 + status: draft + level: 1 + stem: At constant pressure, the heat exchanged equals which quantity? + learning_targets: [lo-enthalpy] + options: + - { id: A, text: "the enthalpy change", correct: true, feedback_student: "Right: P dV work is folded into H." } + - { id: B, text: "the internal energy change", misconception: "ignores expansion work" } + - { id: C, text: "zero" } + solution: + explanation: "Because H = U + PV, at constant P the P dV term is the expansion work, so q_p equals the change in H." + review: + - { ref: kkw, locator: "§6.4", path: "6/A/#4" } + - id: q-enthalpy-op-001 + status: draft + level: 2 + format: open_response + stem: Explain why, at constant pressure, the heat exchanged equals the enthalpy change. + learning_targets: [lo-enthalpy] + solution: + model_answer: "At constant pressure the P dV expansion work is folded into H = U + PV, so q_p is the change in H." + rubric: + - { description: "states H = U + PV", points: 1 } + - { description: "identifies q_p with the enthalpy change", points: 1 } + review: + - { ref: kkw, locator: "§6.4", path: "6/A/#4" } +"#, + ) + .unwrap(); + Catalog::load(&dir).expect("catalog loads") + } + + fn record() -> AssessmentFile { + AssessmentFile { + schema_version: "1.0".into(), + assessment: Assessment { + id: "hw-1".into(), + title: "Homework 1".into(), + term: None, + kind: Kind::Homework, + date: None, + platform: Platform::Canvas, + minutes_allowed: None, + attempts: None, + shuffle: None, + scoring_policy: None, + instructions: None, + notes: None, + }, + blueprint: None, + forms: Vec::new(), + items: vec![ + Placement { + number: 1, + item: "l11::q-enthalpy-001".into(), + version: None, + stem_digest: None, + distractors: Vec::new(), + variant: None, + fingerprint: None, + points: Some(1.0), + bonus: false, + key: vec!["A".into()], + level: None, + learning_targets: Vec::new(), + credit_overrides: Default::default(), + dropped: false, + dropped_as: None, + dropped_before_printing: false, + }, + Placement { + number: 2, + item: "l11::q-enthalpy-op-001".into(), + version: None, + stem_digest: None, + distractors: Vec::new(), + variant: None, + fingerprint: None, + points: Some(2.0), + bonus: false, + key: Vec::new(), + level: None, + learning_targets: Vec::new(), + credit_overrides: Default::default(), + dropped: false, + dropped_as: None, + dropped_before_printing: false, + }, + ], + } + } + + #[test] + fn worksheet_withholds_the_answer() { + let md = + worksheet(&catalog("worksheet"), &record(), &Options::default().form).expect("renders"); + assert!(md.contains("## Question 1")); + assert!(md.contains("A. the enthalpy change")); + // Nothing that reveals the key or the reasoning. + assert!(!md.contains('✓'), "no check marks on the worksheet:\n{md}"); + assert!(!md.contains("Model answer"), "no model answer:\n{md}"); + assert!(!md.contains("P dV"), "no explanation:\n{md}"); + assert!(!md.contains("Rubric")); + // The open-response question leaves room to write. + assert!(md.contains("answer-space")); + } + + #[test] + fn solutions_show_key_reasoning_rubric_and_review() { + let md = + solutions(&catalog("solutions"), &record(), &Options::default().form).expect("renders"); + assert!(md.contains("A. the enthalpy change ✓")); + assert!(md.contains("the internal energy change: ignores expansion work")); + assert!(md.contains("**Model answer.**")); + assert!(md.contains("H = U + PV")); + assert!(md.contains("**Rubric**")); + assert!(md.contains("states H = U + PV (1 pt)")); + assert!(md.contains("Tests:** Define enthalpy")); + // The review citation resolves to the label and a link. + assert!( + md.contains("`KKW` [§6.4](https://example.org/kkw/6/A/#4)"), + "{md}" + ); + } + + #[test] + fn front_matter_titles_each_document() { + let ws = worksheet( + &catalog("front-matter-ws"), + &record(), + &Options::default().form, + ) + .expect("renders"); + assert!(ws.contains("title: \"Homework 1: Questions\""), "{ws}"); + let sol = solutions( + &catalog("front-matter-sol"), + &record(), + &Options::default().form, + ) + .expect("renders"); + assert!(sol.contains("title: \"Homework 1: Solutions\""), "{sol}"); + // Neither document may title itself with the bare assessment name: two + // PDFs called "Homework 1" are indistinguishable once downloaded, and + // the pair that gets confused is the one with the answers in it. + assert!(!ws.contains("title: \"Homework 1\""), "{ws}"); + assert!(!sol.contains("title: \"Homework 1\""), "{sol}"); + } +} diff --git a/src/export/qti.rs b/src/export/qti.rs index 99fa46c..11030cc 100644 --- a/src/export/qti.rs +++ b/src/export/qti.rs @@ -46,10 +46,16 @@ const IMSMD_NS: &str = "http://www.imsglobal.org/xsd/imsmd_v1p2"; /// The content packaging schema location. const IMSCP_SCHEMA: &str = "http://www.imsglobal.org/xsd/imscp_v1p1 imscp_v1p1.xsd \ http://www.imsglobal.org/xsd/imsmd_v1p2 imsmd_v1p2p2.xsd"; +/// The Canvas metadata namespace, used by `assessment_meta.xml`. Canvas stores +/// nearly every quiz setting here (scoring policy, shuffle, results visibility, +/// attempts, publish state); the QTI file itself carries only the questions and +/// `cc_maxattempts`. +const CANVAS_NS: &str = "http://canvas.instructure.com/xsd/cccv1p0"; +/// The schema location Canvas emits alongside `CANVAS_NS`. +const CANVAS_SCHEMA: &str = "http://canvas.instructure.com/xsd/cccv1p0 \ + https://canvas.instructure.com/xsd/cccv1p0.xsd"; -// --------------------------------------------------------------------------- // A very small XML tree -// --------------------------------------------------------------------------- /// One XML element. #[derive(Debug, Clone)] @@ -155,11 +161,157 @@ fn escape_attr(s: &str) -> String { escape_text(s).replace('"', """) } -// --------------------------------------------------------------------------- +/// Renders authoring markup to HTML for Canvas, remapping math delimiters to the +/// forms Canvas's MathJax recognizes. +/// +/// Canvas loads MathJax and typesets a text field only when it finds one of its +/// own delimiters: `\( ... \)` inline, or `$$ ... $$` as a display block. Bare `$ ... $` +/// is not a delimiter Canvas reads, so the tool's authoring convention (`$ ... $` +/// inline) would otherwise reach the quiz as literal dollar-sign text. Here inline +/// `$ ... $` becomes `\( ... \)` and display `$$ ... $$` is left as it is. +/// +/// The split runs before inline markup, the same way the site export handles it, +/// so a subscript like `$q_p$` is not read as an emphasis span. Apart from the +/// math, the paragraph and line handling matches [`markup::to_html`]. +fn to_html_canvas(src: &str) -> String { + let paragraphs: Vec = src + .split("\n\n") + .map(|p| p.trim()) + .filter(|p| !p.is_empty()) + .map(|p| { + let joined = p + .lines() + .map(|l| l.trim()) + .filter(|l| !l.is_empty()) + .collect::>() + .join(" "); + format!("

{}

", canvas_segments(&joined)) + }) + .collect(); + if paragraphs.is_empty() { + String::new() + } else { + paragraphs.join("\n") + } +} + +/// Splits one line on math delimiters and renders each run: prose through the +/// shared escape, symbol, and inline-markup pipeline; math wrapped in the Canvas +/// delimiter for its kind, its interior escaped for HTML transport so a `<` inside +/// math survives as `<` and is decoded back before MathJax reads it. +fn canvas_segments(line: &str) -> String { + let mut out = String::new(); + let mut rest = line; + while let Some(at) = rest.find('$') { + if at > 0 { + out.push_str(&canvas_prose(&rest[..at])); + } + let after = &rest[at..]; + if let Some(display) = after.strip_prefix("$$") { + if let Some(end) = display.find("$$") { + out.push_str(&format!("$${}$$", markup::escape_html(&display[..end]))); + rest = &display[end + 2..]; + continue; + } + } + let inline = &after[1..]; + match inline.find('$') { + Some(end) => { + out.push_str(&format!("\\({}\\)", markup::escape_html(&inline[..end]))); + rest = &inline[end + 1..]; + } + None => { + // An unterminated `$` is ordinary text. + out.push_str(&canvas_prose(after)); + rest = ""; + } + } + } + if !rest.is_empty() { + out.push_str(&canvas_prose(rest)); + } + out +} + +/// A prose run: escape, the symbol table, then inline markup, the pipeline +/// [`markup::to_html`] uses. +fn canvas_prose(text: &str) -> String { + markup::apply_inline(&markup::apply_symbols(&markup::escape_html(text), true)) +} + +/// The per-option feedback to show, given the attempt policy. +/// +/// A single-attempt quiz is summative, so it shows the full post-submission +/// feedback ([`crate::item::Choice::student_text`]: the misconception and +/// explanation). A +/// multi-attempt quiz is formative, so it shows only the hint, withholding the +/// misconception a student would otherwise read before their next try. When a +/// distractor has no hint, a multi-attempt quiz shows nothing for it rather than +/// falling back to the misconception. +fn option_feedback<'a>(choice: &'a crate::item::Choice, opts: &QtiOptions) -> Option<&'a str> { + if opts.attempts == 1 { + choice.student_text() + } else { + choice.hint.as_deref() + } +} + +/// Renders `cc_maxattempts` the way Canvas's QTI dialect actually reads it: a +/// plain decimal count for a fixed number of attempts, or the literal string +/// `unlimited` for unbounded attempts. +/// +/// Canvas does not treat `-1` (this module's in-memory sentinel for "unlimited", +/// see [`QtiOptions::attempts`]) as a number meaning unlimited. Writing it out as +/// `-1` produces XML that still imports without error, so this is exactly the +/// kind of silent-misbehavior trap the module-level docs warn about: the quiz +/// comes in looking fine and only turns out wrong once a student hits the +/// attempt limit. Every other non-negative value round-trips as a plain number. +fn max_attempts_field(attempts: i64) -> String { + if attempts < 0 { + "unlimited".to_string() + } else { + attempts.to_string() + } +} + // Package construction -// --------------------------------------------------------------------------- + +/// The Canvas quiz type, written to `assessment_meta.xml` as ``. +/// +/// [`QuizType::Assignment`] is an ordinary graded quiz. A graded survey scores +/// on completion rather than correctness, so it cannot show right/wrong marks; +/// don't pair it with per-option feedback that presumes a correct answer. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum QuizType { + /// A graded quiz (the default). + Assignment, + /// An ungraded practice quiz. + PracticeQuiz, + /// A graded survey: points for completing, not for correctness. + GradedSurvey, + /// An ungraded survey. + Survey, +} + +impl QuizType { + /// The `` string Canvas expects. + fn as_str(self) -> &'static str { + match self { + QuizType::Assignment => "assignment", + QuizType::PracticeQuiz => "practice_quiz", + QuizType::GradedSurvey => "graded_survey", + QuizType::Survey => "survey", + } + } +} /// Options for a QTI export. +/// +/// The fields fall into two groups. A few (`form`, `include_feedback`) shape the +/// QTI questions themselves. The rest map onto `assessment_meta.xml`, the Canvas +/// sidecar file that actually carries the quiz's Details-tab settings: putting a +/// setting here is what makes it survive import, since the QTI file cannot hold +/// it. #[derive(Debug, Clone)] pub struct QtiOptions { /// Which form's option order to use. @@ -169,11 +321,37 @@ pub struct QtiOptions { /// immediately. pub include_feedback: bool, /// Whether to let Canvas shuffle answers on top of the form's own order. + /// Emitted as `` in the meta file. pub shuffle_in_canvas: bool, - /// Maximum attempts; `-1` for unlimited. + /// Maximum attempts; `-1` for unlimited. More than one attempt also switches + /// per-option feedback from the misconception to the hint, so a formative + /// quiz nudges rather than reveals. Emitted as `cc_maxattempts` in the QTI + /// file and `` in the meta file. pub attempts: i64, - /// How repeated attempts are scored. + /// How repeated attempts are scored. Emitted as ``. pub scoring_policy: ScoringPolicy, + /// The Canvas quiz type. Emitted as ``. + pub quiz_type: QuizType, + /// Whether students may review their own submissions and see which answers + /// were marked wrong — the "Let Students See Their Quiz Responses" checkbox. + /// This is the switch that makes the per-answer hints visible at all, so it + /// defaults on. Emitted as ``: empty when true, `always` when + /// false. + pub let_students_see_responses: bool, + /// Whether students see the keyed-correct answer highlighted on review — the + /// "Let Students See The Correct Answers" checkbox. Only has an effect when + /// `let_students_see_responses` is true. Emitted as ``. + pub show_correct_answers: bool, + /// Whether Canvas shows one question per page. Emitted as + /// ``. + pub one_question_at_a_time: bool, + /// A time limit in minutes, or `None` for untimed. Emitted as ``. + pub time_limit_minutes: Option, + /// Whether the imported quiz is published (available to students) on arrival. + /// Defaults off so an import never surprises a class with a live quiz before + /// the instructor has looked at it. Emitted as `` and the + /// assignment's ``. + pub published: bool, } impl Default for QtiOptions { @@ -186,20 +364,44 @@ impl Default for QtiOptions { shuffle_options: false, }, include_feedback: true, - shuffle_in_canvas: false, - attempts: 1, + shuffle_in_canvas: true, + // Unlimited attempts, which also selects the hint-only feedback path + // in `option_feedback` rather than the full reveal, so a quiz is + // formative by default. Pass `attempts: 1` explicitly for a + // single-shot summative export. + attempts: -1, scoring_policy: ScoringPolicy::KeepHighest, + // The remaining defaults describe the formative graded quiz worked + // out with the course staff: a normal graded quiz, students may + // review their responses and see the correct answer, all questions + // on one page, untimed, and left unpublished for the instructor to + // publish after review. + quiz_type: QuizType::Assignment, + let_students_see_responses: true, + show_correct_answers: true, + one_question_at_a_time: false, + time_limit_minutes: None, + published: false, } } } /// A rendered QTI package, ready to write. +/// +/// The archive mirrors a Canvas Common Cartridge quiz export: the manifest sits +/// at the root, and the quiz QTI file and its `assessment_meta.xml` sidecar sit +/// together in a folder named for the assessment identifier. `quiz_filename` and +/// `meta_filename` are the in-archive paths, including that folder. #[derive(Debug, Clone)] pub struct Package { - /// The name of the quiz XML file inside the archive. + /// The in-archive path of the quiz XML file (e.g. `g.../a1_1.xml`). pub quiz_filename: String, - /// The quiz XML. + /// The quiz QTI XML. pub quiz_xml: String, + /// The in-archive path of the settings sidecar (e.g. `g.../assessment_meta.xml`). + pub meta_filename: String, + /// The `assessment_meta.xml` XML, carrying the Canvas quiz settings. + pub meta_xml: String, /// The manifest XML. pub manifest_xml: String, } @@ -216,10 +418,12 @@ impl Package { /// Returns [`Error::Io`] on a write failure. pub fn write_zip(&self, path: &std::path::Path) -> Result<()> { let mut z = ZipBuilder::new(); - // Canvas only recognizes the archive as QTI when the manifest sits at the - // root rather than inside a directory. + // Canvas only recognizes the archive as a course/quiz package when the + // manifest sits at the root. The quiz and its settings sidecar live in a + // per-assessment folder, the layout Canvas's own export uses. z.add_text("imsmanifest.xml", &self.manifest_xml); z.add_text(&self.quiz_filename, &self.quiz_xml); + z.add_text(&self.meta_filename, &self.meta_xml); z.write_to(path) } } @@ -247,11 +451,19 @@ pub fn build(catalog: &Catalog, record: &AssessmentFile, opts: &QtiOptions) -> R let mut problems = Vec::new(); let mut items = Vec::new(); + // Accumulated to fill `` in the meta file. Canvas recomputes + // this from the question points on import, so it is advisory, but it should + // still agree with the questions. Bonus placements are excluded, matching how + // they sit outside the graded total. + let mut total_points = 0.0_f64; for placement in select::layout(record, &opts.form) { let entry = catalog.require(&placement.item)?; let item = &entry.item; - if item.key_indices().is_empty() { + // A choice question needs a keyed option or Canvas cannot score it. An + // open-response question has no options and is graded by hand, so the + // absence of a key is expected; it exports as a Canvas essay. + if item.format.has_options() && item.key_indices().is_empty() { problems.push(format!( "question {} ({}) has no keyed option, so Canvas cannot score it", placement.number, placement.item @@ -261,10 +473,15 @@ pub fn build(catalog: &Catalog, record: &AssessmentFile, opts: &QtiOptions) -> R let points = placement .points .unwrap_or_else(|| item.points(default_points)); + if !placement.bonus { + total_points += points; + } + let shown = item.administered(&placement.key, &placement.distractors); items.push(build_item( &record.assessment.id, &placement.item, item, + &shown, points, opts, )); @@ -274,18 +491,14 @@ pub fn build(catalog: &Catalog, record: &AssessmentFile, opts: &QtiOptions) -> R return Err(Error::Invalid(problems)); } - let metadata = vec![ - metadata_field("cc_maxattempts", &opts.attempts.to_string()), - metadata_field("cc_quiz_scoring_policy", opts.scoring_policy.as_str()), - metadata_field( - "cc_shuffle_answers", - if opts.shuffle_in_canvas { - "true" - } else { - "false" - }, - ), - ]; + // The QTI file carries only `cc_maxattempts`, matching what Canvas's own + // export writes. Scoring policy, shuffle, and the results-visibility settings + // live in `assessment_meta.xml` instead; writing them here too would create a + // second, potentially disagreeing source of truth. + let metadata = vec![metadata_field( + "cc_maxattempts", + &max_attempts_field(opts.attempts), + )]; let title = if record.forms.len() > 1 { format!("{} (form {})", record.assessment.title, opts.form.id) @@ -309,17 +522,33 @@ pub fn build(catalog: &Catalog, record: &AssessmentFile, opts: &QtiOptions) -> R .attr("xsi:schemaLocation", QTI_SCHEMA) .child(assessment); - let quiz_filename = format!("{}.xml", slug_filename(&record.assessment.id)); + // Canvas's layout: a folder named for the assessment identifier holds both + // the quiz QTI file and its settings sidecar. + let quiz_dir = assessment_id.clone(); + let quiz_href = format!("{quiz_dir}/{}.xml", slug_filename(&record.assessment.id)); + let meta_href = format!("{quiz_dir}/assessment_meta.xml"); + let meta_resource_id = format!("{assessment_id}_meta"); + + // The assignment wrapper Canvas creates for a graded quiz needs its own + // stable identifier, derived like every other id here. + let assignment_id = qti_id(&format!("{}/assignment", record.assessment.id)); + + let meta = build_assessment_meta(&assessment_id, &assignment_id, &title, total_points, opts); + let manifest = build_manifest( - &quiz_filename, + &quiz_href, + &meta_href, &assessment_id, + &meta_resource_id, &title, &record.assessment.id, ); Ok(Package { - quiz_filename, + quiz_filename: quiz_href, quiz_xml: root.document(), + meta_filename: meta_href, + meta_xml: meta.document(), manifest_xml: manifest.document(), }) } @@ -337,9 +566,21 @@ pub fn build(catalog: &Catalog, record: &AssessmentFile, opts: &QtiOptions) -> R /// # Returns /// /// The element. -fn build_item(assessment_id: &str, uid: &str, item: &Item, points: f64, opts: &QtiOptions) -> Node { - let order = select::option_order(&opts.form, uid, item.options.len()); - let ordered: Vec<&crate::item::Choice> = order.iter().map(|i| &item.options[*i]).collect(); +fn build_item( + assessment_id: &str, + uid: &str, + item: &Item, + shown: &[&crate::item::Choice], + points: f64, + opts: &QtiOptions, +) -> Node { + // An open-response item is an essay in Canvas: no choices, graded by hand. + if !item.format.has_options() { + return build_essay_item(assessment_id, uid, item, points, opts); + } + + let order = select::option_order(&opts.form, uid, shown.len()); + let ordered: Vec<&crate::item::Choice> = order.iter().map(|i| shown[*i]).collect(); // Option identifiers are numeric, mirroring Canvas's own exports, and are // derived from the item id so they survive regeneration. @@ -367,14 +608,14 @@ fn build_item(assessment_id: &str, uid: &str, item: &Item, points: f64, opts: &Q .map(|(o, id)| { Node::new("response_label") .attr("ident", id.clone()) - .child(mattext(&markup::to_html(&o.text))) + .child(mattext(&to_html_canvas(&o.text))) }) .collect(); let presentation = Node::new("presentation") .child(mattext(&format!( "
{}
", - markup::to_html(&item.stem) + to_html_canvas(&item.stem) ))) .child( Node::new("response_lid") @@ -398,7 +639,7 @@ fn build_item(assessment_id: &str, uid: &str, item: &Item, points: f64, opts: &Q // feedback. `continue="Yes"` is what allows scoring to be evaluated after. if opts.include_feedback { for (o, id) in ordered.iter().zip(opt_ids.iter()) { - if o.student_text().is_none() { + if option_feedback(o, opts).is_none() { continue; } resprocessing = resprocessing.child( @@ -483,13 +724,13 @@ fn build_item(assessment_id: &str, uid: &str, item: &Item, points: f64, opts: &Q if opts.include_feedback { for (o, id) in ordered.iter().zip(opt_ids.iter()) { - if let Some(text) = o.student_text() { + if let Some(text) = option_feedback(o, opts) { node = node.child( Node::new("itemfeedback") .attr("ident", format!("{id}_fb")) .child( Node::new("flow_mat") - .child(mattext(&format!("
{}
", markup::to_html(text)))), + .child(mattext(&format!("
{}
", to_html_canvas(text)))), ), ); } @@ -499,6 +740,104 @@ fn build_item(assessment_id: &str, uid: &str, item: &Item, points: f64, opts: &Q node } +/// Builds a Canvas essay item for an open-response question. +/// +/// An essay has no choices and no automatic score: the `` condition leaves +/// grading to the instructor. When feedback is on and a model answer exists, it +/// rides along as general feedback so a student sees it after submitting. +/// +/// This mapping has not been round-tripped through a live Canvas import in this +/// build, so verify it against your instance before relying on it for a graded +/// quiz. +/// +/// # Arguments +/// +/// * `assessment_id` - salts the generated ids. +/// * `uid` - the item's global id. +/// * `item` - the item. +/// * `points` - points as administered. +/// * `opts` - export options. +/// +/// # Returns +/// +/// The element. +fn build_essay_item( + assessment_id: &str, + uid: &str, + item: &Item, + points: f64, + opts: &QtiOptions, +) -> Node { + let item_meta = Node::new("itemmetadata").child(Node::new("qtimetadata").children(vec![ + metadata_field("question_type", item.format.qti_type()), + metadata_field("points_possible", &format!("{points:.2}")), + metadata_field("assessment_question_identifierref", &qti_id(uid)), + ])); + + let presentation = Node::new("presentation") + .child(mattext(&format!( + "
{}
", + to_html_canvas(&item.stem) + ))) + .child( + Node::new("response_str") + .attr("ident", "response1") + .attr("rcardinality", "Single") + .child( + Node::new("render_fib").child( + Node::new("response_label") + .attr("ident", "answer1") + .attr("rshuffle", "No"), + ), + ), + ); + + // The model answer, shown as general feedback after submission. + let model = item + .solution + .as_ref() + .and_then(|s| s.model_answer.as_deref()) + .filter(|_| opts.include_feedback); + + let mut condition = Node::new("respcondition") + .attr("continue", "No") + .child(Node::new("conditionvar").child(Node::new("other"))); + if model.is_some() { + condition = condition.child( + Node::new("displayfeedback") + .attr("feedbacktype", "Response") + .attr("linkrefid", "general_fb"), + ); + } + + let resprocessing = Node::new("resprocessing") + .child( + Node::new("outcomes").child( + Node::new("decvar") + .attr("maxvalue", "100") + .attr("minvalue", "0") + .attr("varname", "SCORE") + .attr("vartype", "Decimal"), + ), + ) + .child(condition); + + let mut node = Node::new("item") + .attr("ident", qti_id(&format!("{assessment_id}/{uid}"))) + .attr("title", item.display_title()) + .child(item_meta) + .child(presentation) + .child(resprocessing); + + if let Some(text) = model { + node = node.child(Node::new("itemfeedback").attr("ident", "general_fb").child( + Node::new("flow_mat").child(mattext(&format!("
{}
", to_html_canvas(text)))), + )); + } + + node +} + /// A `` pair. /// /// # Arguments @@ -527,15 +866,29 @@ fn metadata_field(label: &str, entry: &str) -> Node { /// /// # Arguments /// -/// * `quiz_filename` - the quiz XML file name. -/// * `assessment_id` - the assessment identifier, reused as the resource id. +/// * `quiz_href` - the in-archive path of the quiz XML file. +/// * `meta_href` - the in-archive path of the settings sidecar. +/// * `assessment_id` - the assessment identifier, reused as the quiz resource id. +/// * `meta_resource_id` - the resource id for the sidecar. /// * `title` - the human title. /// * `salt` - salts the manifest identifier. /// /// # Returns /// /// The manifest element. -fn build_manifest(quiz_filename: &str, assessment_id: &str, title: &str, salt: &str) -> Node { +/// +/// The quiz resource declares the sidecar as a ``, and the sidecar is +/// its own `learning-application-resource`, the wiring Canvas uses to find +/// `assessment_meta.xml` for a quiz. Without the dependency Canvas imports the +/// questions but ignores the settings file. +fn build_manifest( + quiz_href: &str, + meta_href: &str, + assessment_id: &str, + meta_resource_id: &str, + title: &str, + salt: &str, +) -> Node { let lom = Node::new("imsmd:lom").child( Node::new("imsmd:general").child( Node::new("imsmd:title").child( @@ -546,6 +899,22 @@ fn build_manifest(quiz_filename: &str, assessment_id: &str, title: &str, salt: & ), ); + let quiz_resource = Node::new("resource") + .attr("identifier", assessment_id) + .attr("type", "imsqti_xmlv1p2/imscc_xmlv1p1/assessment") + .attr("href", quiz_href) + .child(Node::new("file").attr("href", quiz_href)) + .child(Node::new("dependency").attr("identifierref", meta_resource_id)); + + let meta_resource = Node::new("resource") + .attr("identifier", meta_resource_id) + .attr( + "type", + "associatedcontent/imscc_xmlv1p1/learning-application-resource", + ) + .attr("href", meta_href) + .child(Node::new("file").attr("href", meta_href)); + Node::new("manifest") .attr("identifier", format!("man{}", qti_id(salt))) .attr("xmlns", IMSCP_NS) @@ -566,16 +935,112 @@ fn build_manifest(quiz_filename: &str, assessment_id: &str, title: &str, salt: & ), ) .child( - Node::new("resources").child( - Node::new("resource") - .attr("identifier", assessment_id) - .attr("type", "imsqti_xmlv1p2") - .attr("href", quiz_filename) - .child(Node::new("file").attr("href", quiz_filename)), - ), + Node::new("resources") + .child(quiz_resource) + .child(meta_resource), ) } +/// Builds the `assessment_meta.xml` element: the Canvas sidecar that carries the +/// quiz's Details-tab settings. +/// +/// The QTI file holds the questions and `cc_maxattempts` and nothing else, so +/// this is where scoring policy, answer shuffling, results visibility, the quiz +/// type, timing, and publish state actually live. The element and child order +/// follow a Canvas-emitted file; Canvas is order-tolerant on import, but matching +/// its shape keeps the output diffable against a real export. +/// +/// Two mappings are worth stating. `` is the inverse of the "Let +/// Students See Their Quiz Responses" checkbox: empty means they may review, +/// `always` means they never can — so students only ever see the per-answer hints +/// when it is empty. And `` takes the raw `-1` sentinel for +/// unlimited (unlike the QTI file's `cc_maxattempts`, which spells it `unlimited`). +/// +/// # Arguments +/// +/// * `quiz_identifier` - the quiz identifier, reused as `quiz_identifierref`. +/// * `assignment_identifier` - the wrapping assignment's identifier. +/// * `title` - the human title. +/// * `points_possible` - the summed non-bonus points. +/// * `opts` - export options. +/// +/// # Returns +/// +/// The `` element. +fn build_assessment_meta( + quiz_identifier: &str, + assignment_identifier: &str, + title: &str, + points_possible: f64, + opts: &QtiOptions, +) -> Node { + let bool_str = |b: bool| if b { "true" } else { "false" }; + // Canvas prints whole point totals without a decimal (e.g. `8`), matching its + // own export. + let points = format!("{points_possible}"); + let workflow_state = if opts.published { + "published" + } else { + "unpublished" + }; + + // "Let Students See Their Quiz Responses" on -> hide_results empty; off -> + // `always` (never show results, which also hides the hints). + let hide_results = Node::new("hide_results"); + let hide_results = if opts.let_students_see_responses { + hide_results + } else { + hide_results.text("always") + }; + + let time_limit = match opts.time_limit_minutes { + Some(m) => Node::new("time_limit").text(m.to_string()), + None => Node::new("time_limit"), + }; + + let assignment = Node::new("assignment") + .attr("identifier", assignment_identifier) + .child(Node::new("title").text(title)) + .child(Node::new("due_at")) + .child(Node::new("lock_at")) + .child(Node::new("unlock_at")) + .child(Node::new("workflow_state").text(workflow_state)) + .child(Node::new("quiz_identifierref").text(quiz_identifier)) + .child(Node::new("points_possible").text(points.clone())) + .child(Node::new("grading_type").text("points")) + .child(Node::new("submission_types").text("online_quiz")) + .child(Node::new("omit_from_final_grade").text("false")); + + Node::new("quiz") + .attr("identifier", quiz_identifier) + .attr("xmlns", CANVAS_NS) + .attr("xmlns:xsi", XSI_NS) + .attr("xsi:schemaLocation", CANVAS_SCHEMA) + .child(Node::new("title").text(title)) + .child(Node::new("shuffle_answers").text(bool_str(opts.shuffle_in_canvas))) + .child(Node::new("scoring_policy").text(opts.scoring_policy.as_str())) + .child(hide_results) + .child(Node::new("quiz_type").text(opts.quiz_type.as_str())) + .child(Node::new("points_possible").text(points)) + .child(Node::new("require_lockdown_browser").text("false")) + .child(Node::new("require_lockdown_browser_for_results").text("false")) + .child(Node::new("require_lockdown_browser_monitor").text("false")) + .child(Node::new("lockdown_browser_monitor_data")) + .child(Node::new("show_correct_answers").text(bool_str(opts.show_correct_answers))) + .child(Node::new("anonymous_submissions").text("false")) + .child(Node::new("could_be_locked").text("false")) + .child(time_limit) + .child(Node::new("allowed_attempts").text(opts.attempts.to_string())) + .child(Node::new("one_question_at_a_time").text(bool_str(opts.one_question_at_a_time))) + .child(Node::new("cant_go_back").text("false")) + .child(Node::new("available").text(bool_str(opts.published))) + .child(Node::new("one_time_results").text("false")) + .child(Node::new("show_correct_answers_last_attempt").text("false")) + .child(Node::new("only_visible_to_overrides").text("false")) + .child(Node::new("module_locked").text("false")) + .child(assignment) +} + /// A deterministic QTI identifier: `g` followed by 32 hex characters. /// /// # Arguments @@ -675,11 +1140,91 @@ mod tests { } #[test] - fn manifest_points_at_the_quiz_file() { - let m = build_manifest("exam-4.xml", "gabc", "Exam 4", "exam-4").render(0); - assert!(m.contains("imsqti_xmlv1p2")); - assert!(m.contains("href=\"exam-4.xml\"")); + fn max_attempts_field_is_unlimited_not_negative_one() { + // The bug this guards against: writing the in-memory `-1` sentinel + // straight into the XML instead of translating it to the literal + // Canvas expects. `-1` imports without error and silently fails to + // behave as unlimited. + assert_eq!(max_attempts_field(-1), "unlimited"); + assert_eq!(max_attempts_field(1), "1"); + assert_eq!(max_attempts_field(3), "3"); + assert_eq!(max_attempts_field(0), "0"); + } + + #[test] + fn manifest_points_at_the_quiz_and_its_meta() { + let m = build_manifest( + "gabc/exam-4.xml", + "gabc/assessment_meta.xml", + "gabc", + "gabc_meta", + "Exam 4", + "exam-4", + ) + .render(0); + assert!(m.contains("imsqti_xmlv1p2/imscc_xmlv1p1/assessment")); + assert!(m.contains("href=\"gabc/exam-4.xml\"")); assert!(m.contains("Exam 4")); + // The settings sidecar must be declared and depended on, or Canvas + // imports the questions and silently drops every quiz setting. + assert!(m.contains("href=\"gabc/assessment_meta.xml\"")); + assert!(m.contains("learning-application-resource")); + assert!(m.contains("")); + } + + #[test] + fn meta_carries_the_discussed_defaults() { + let opts = QtiOptions::default(); + let m = build_assessment_meta("gquiz", "gassign", "A1.1: Enthalpy", 8.0, &opts).render(0); + // The settings that make this a formative, unlimited-attempts quiz. + assert!(m.contains("keep_highest")); + assert!(m.contains("-1")); + assert!(m.contains("assignment")); + assert!(m.contains("true")); + // Canvas does the shuffling. The form fixes an option order for the + // printed paper, but an online quiz has no printed order to preserve and + // `data::canvas` matches responses back by answer text rather than by + // letter, so a per-student shuffle costs the analysis nothing. + assert!(m.contains("true")); + // Default is unpublished, so an import never goes live unreviewed. + assert!(m.contains("false")); + assert!(m.contains("unpublished")); + // Canvas namespace, so the importer recognizes it as quiz settings. + assert!(m.contains(CANVAS_NS)); + } + + #[test] + fn shuffling_can_be_turned_off_per_assessment() { + // A record with `shuffle: false` is the case where the printed order has + // to be preserved online too, such as an item whose options are a + // sequence ("first ... then ...") that reads wrong in any other order. + let opts = QtiOptions { + shuffle_in_canvas: false, + ..QtiOptions::default() + }; + let m = build_assessment_meta("gq", "ga", "T", 1.0, &opts).render(0); + assert!( + m.contains("false"), + "{m}" + ); + } + + #[test] + fn hide_results_tracks_letting_students_see_responses() { + // Default (on): hide_results is empty, so students may review and the + // per-answer hints are visible. + let on = QtiOptions::default(); + let m = build_assessment_meta("gq", "ga", "T", 1.0, &on).render(0); + assert!(m.contains(""), "{m}"); + assert!(!m.contains("always")); + + // Off: hide_results is `always`, which also suppresses the hints. + let off = QtiOptions { + let_students_see_responses: false, + ..QtiOptions::default() + }; + let m = build_assessment_meta("gq", "ga", "T", 1.0, &off).render(0); + assert!(m.contains("always"), "{m}"); } #[test] @@ -687,4 +1232,252 @@ mod tests { assert_eq!(slug_filename("exam-4 2026s"), "exam-4_2026s"); assert_eq!(slug_filename(""), "quiz"); } + + #[test] + fn canvas_math_uses_mathjax_delimiters() { + // Inline $...$ becomes \( ... \), which Canvas typesets in the text flow. + assert_eq!( + to_html_canvas("Enthalpy, $\\Delta H$"), + "

Enthalpy, \\(\\Delta H\\)

" + ); + // Display $$...$$ is left as a Canvas block. + assert_eq!(to_html_canvas("$$E = mc^2$$"), "

$$E = mc^2$$

"); + // A subscript inside math is not read as emphasis. + assert_eq!( + to_html_canvas("$q_p = \\Delta H$"), + "

\\(q_p = \\Delta H\\)

" + ); + // A `<` inside math is escaped for transport; Canvas decodes it before + // MathJax reads it. + assert_eq!(to_html_canvas("$a < b$"), "

\\(a < b\\)

"); + // Prose with no math is escaped and wrapped, same as `to_html`. + assert_eq!(to_html_canvas("a & b"), "

a & b

"); + } + + fn distractor() -> crate::item::Choice { + crate::item::Choice { + id: "B".into(), + text: "Internal energy".into(), + correct: false, + credit: None, + explanation: Some("Only at constant volume.".into()), + hint: Some("Reconsider what stays constant in an open flask.".into()), + misconception: Some("Uses the constant-volume result.".into()), + error_type: None, + defensible: false, + defense: None, + feedback_student: None, + selection_rate_expected: None, + retired: None, + } + } + + #[test] + fn one_attempt_shows_the_misconception_and_more_show_the_hint() { + let choice = distractor(); + let mut opts = QtiOptions { + attempts: 1, + ..QtiOptions::default() + }; + // student_text() prefers feedback_student, then the misconception. + assert_eq!( + option_feedback(&choice, &opts), + Some("Uses the constant-volume result.") + ); + + opts.attempts = 3; + assert_eq!( + option_feedback(&choice, &opts), + Some("Reconsider what stays constant in an open flask.") + ); + opts.attempts = -1; // unlimited is also multi-attempt + assert!( + option_feedback(&choice, &opts) + .unwrap() + .starts_with("Reconsider") + ); + + // Multiple attempts with no hint show nothing, so the misconception is + // not revealed before the next try. + let no_hint = crate::item::Choice { + hint: None, + ..distractor() + }; + assert_eq!(option_feedback(&no_hint, &opts), None); + } + + #[test] + fn default_options_export_as_unlimited_with_hints() { + // Locks in the two changes together: the default is unlimited + // attempts, and that default routes wrong-answer feedback through the + // hint rather than the full reveal. A quiz built with `..Default::default()` + // and no explicit `attempts` is formative unless the caller opts out. + let opts = QtiOptions::default(); + assert_eq!(opts.attempts, -1); + assert_eq!( + max_attempts_field(opts.attempts), + "unlimited", + "the exported cc_maxattempts field must read \"unlimited\", not \"-1\"" + ); + + let choice = distractor(); + assert_eq!( + option_feedback(&choice, &opts), + Some("Reconsider what stays constant in an open flask."), + "default options should show the hint, not the misconception reveal" + ); + } + + fn catalog(tag: &str) -> Catalog { + let dir = std::env::temp_dir().join(format!("cb-qti-{tag}-{}", std::process::id())); + let _ = std::fs::remove_dir_all(&dir); + std::fs::create_dir_all(dir.join("banks")).unwrap(); + std::fs::write( + dir.join("course.yaml"), + r#" +course: { code: BIOSC 1000, title: Biochemistry, term: 2026f } +lectures: + L1.1: { title: Enthalpy } +learning_objectives: + lo-enthalpy: + text: Define enthalpy. + lectures: [L1.1] + order: 1 +"#, + ) + .unwrap(); + std::fs::write( + dir.join("banks").join("b.yaml"), + r#" +bank: { id: b, title: Bank } +items: + - id: q-mcq + status: draft + level: 2 + format: single_best_answer + stem: "The heat at constant pressure equals a change in what?" + learning_targets: [lo-enthalpy] + options: + - { id: A, text: "Enthalpy, $\\Delta H$", correct: true } + - { id: B, text: "Internal energy, $\\Delta U$", misconception: "Constant-volume result." } + - id: q-open + status: draft + level: 3 + format: open_response + stem: "Show why $q_p = \\Delta H$." + learning_targets: [lo-enthalpy] + solution: + model_answer: "From $H = U + PV$ at constant pressure, $q_p = \\Delta H$." +"#, + ) + .unwrap(); + Catalog::load(&dir).expect("catalog loads") + } + + fn record() -> AssessmentFile { + use crate::assessment::{Assessment, Kind, Placement, Platform}; + AssessmentFile { + schema_version: "1.0".into(), + assessment: Assessment { + id: "a1.1".into(), + title: "Homework 1".into(), + term: None, + kind: Kind::Homework, + date: None, + platform: Platform::Canvas, + minutes_allowed: None, + attempts: None, + shuffle: None, + scoring_policy: None, + instructions: None, + notes: None, + }, + blueprint: None, + forms: Vec::new(), + items: vec![ + Placement { + number: 1, + item: "b::q-mcq".into(), + version: None, + stem_digest: None, + distractors: Vec::new(), + variant: None, + fingerprint: None, + points: Some(1.0), + bonus: false, + key: vec!["A".into()], + level: None, + learning_targets: Vec::new(), + credit_overrides: Default::default(), + dropped: false, + dropped_as: None, + dropped_before_printing: false, + }, + Placement { + number: 2, + item: "b::q-open".into(), + version: None, + stem_digest: None, + distractors: Vec::new(), + variant: None, + fingerprint: None, + points: Some(2.0), + bonus: false, + key: Vec::new(), + level: None, + learning_targets: Vec::new(), + credit_overrides: Default::default(), + dropped: false, + dropped_as: None, + dropped_before_printing: false, + }, + ], + } + } + + #[test] + fn an_open_response_item_exports_as_a_canvas_essay() { + let cat = catalog("essay"); + let rec = record(); + let pkg = build(&cat, &rec, &QtiOptions::default()) + .expect("an essay item does not block the export"); + // Both questions survive: the choice question and the open-response essay. + assert!( + pkg.quiz_xml.contains("multiple_choice_question"), + "the choice question should be present" + ); + assert!( + pkg.quiz_xml.contains("essay_question"), + "the open-response item should export as a Canvas essay" + ); + // The essay stem's math is rendered with Canvas delimiters, and the model + // answer rides along as feedback. + assert!(pkg.quiz_xml.contains("\\(q_p = \\Delta H\\)")); + assert!(pkg.quiz_xml.contains("H = U + PV")); + + // The QTI file carries only cc_maxattempts now; scoring and shuffle moved + // to the sidecar. + assert!(pkg.quiz_xml.contains("cc_maxattempts")); + assert!( + !pkg.quiz_xml.contains("cc_quiz_scoring_policy"), + "scoring policy belongs in assessment_meta.xml, not the QTI file" + ); + assert!(!pkg.quiz_xml.contains("cc_shuffle_answers")); + + // The package includes a settings sidecar, and the manifest points the + // quiz resource at it. The essay's 2 points are excluded from the graded + // total only if bonus; here both placements count, so points_possible is + // the sum the questions carry. + assert!(pkg.meta_filename.ends_with("assessment_meta.xml")); + assert!( + pkg.meta_xml + .contains("keep_highest") + ); + assert!( + pkg.meta_xml + .contains("-1") + ); + assert!(pkg.manifest_xml.contains(&pkg.meta_filename)); + assert!(pkg.manifest_xml.contains(" "json", + Format::Bibtex => "bib", + } + } + + /// The name used on the command line. + pub fn label(self) -> &'static str { + match self { + Format::Hayagriva => "hayagriva", + Format::CslJson => "csl-json", + Format::Bibtex => "bibtex", + } + } +} + +/// Renders a course's bibliography. +/// +/// # Arguments +/// +/// * `course` - the loaded course. +/// * `format` - which format to write. +/// +/// # Returns +/// +/// The document, keyed or ordered by citation key so the output is stable +/// between runs. +/// +/// # Errors +/// +/// Returns [`crate::error::Error::Other`] if the intermediate structure cannot +/// be serialized. +pub fn render(course: &CourseFile, format: Format) -> Result { + match format { + Format::Hayagriva => hayagriva(course), + Format::CslJson => csl_json(course), + Format::Bibtex => Ok(bibtex(course)), + } +} + +// --- Hayagriva --- + +/// One Hayagriva entry. +#[derive(Debug, Serialize)] +struct Entry { + #[serde(rename = "type")] + kind: &'static str, + title: String, + #[serde(skip_serializing_if = "Vec::is_empty")] + author: Vec, + #[serde(skip_serializing_if = "Option::is_none")] + date: Option, + #[serde(skip_serializing_if = "Option::is_none")] + edition: Option, + #[serde(skip_serializing_if = "Option::is_none")] + publisher: Option, + #[serde(rename = "page-range", skip_serializing_if = "Option::is_none")] + page_range: Option, + #[serde(rename = "serial-number", skip_serializing_if = "BTreeMap::is_empty")] + serial_number: BTreeMap<&'static str, String>, + #[serde(skip_serializing_if = "Option::is_none")] + url: Option, + #[serde(skip_serializing_if = "Option::is_none")] + parent: Option, +} + +/// The container a Hayagriva entry sits inside. +#[derive(Debug, Serialize)] +struct Parent { + #[serde(rename = "type")] + kind: &'static str, + title: String, + #[serde(skip_serializing_if = "Option::is_none")] + volume: Option, + #[serde(skip_serializing_if = "Option::is_none")] + issue: Option, + #[serde(skip_serializing_if = "Option::is_none")] + publisher: Option, +} + +/// Renders the bibliography as Hayagriva YAML. +fn hayagriva(course: &CourseFile) -> Result { + let entries: BTreeMap<&String, Entry> = course + .references + .iter() + .map(|(key, reference)| (key, entry(reference))) + .collect(); + yaml::to_string(&entries) +} + +/// Maps one reference onto a Hayagriva entry. +fn entry(reference: &Reference) -> Entry { + let mut serial: BTreeMap<&'static str, String> = BTreeMap::new(); + for (field, value) in [ + ("doi", &reference.doi), + ("isbn", &reference.isbn), + ("arxiv", &reference.arxiv), + ("pmid", &reference.pmid), + ("pmcid", &reference.pmcid), + ] { + if let Some(value) = value { + serial.insert(field, value.clone()); + } + } + + // A chapter sits in a book and an article sits in a periodical, and + // Hayagriva wants that said with a parent rather than a flat field. + let parent = reference + .container + .as_ref() + .map(|title| match reference.kind { + ReferenceKind::Chapter => Parent { + kind: "book", + title: title.clone(), + volume: reference.volume.clone(), + issue: None, + publisher: reference.publisher.clone(), + }, + _ => Parent { + kind: "periodical", + title: title.clone(), + volume: reference.volume.clone(), + issue: reference.issue.clone(), + publisher: None, + }, + }); + + Entry { + kind: hayagriva_kind(reference.kind), + title: reference.title.clone(), + author: reference.authors.clone(), + date: reference.year, + edition: reference.edition.clone(), + // The publisher belongs to the container when there is one. + publisher: if parent.is_some() { + None + } else { + reference.publisher.clone() + }, + page_range: reference.pages.clone(), + serial_number: serial, + url: reference.url.clone(), + parent, + } +} + +/// The Hayagriva entry type for a course reference kind. +fn hayagriva_kind(kind: ReferenceKind) -> &'static str { + match kind { + ReferenceKind::Book => "book", + ReferenceKind::Chapter => "chapter", + // Hayagriva has no preprint type. An article with no periodical parent + // is what a preprint is anyway. + ReferenceKind::Article | ReferenceKind::Preprint => "article", + ReferenceKind::Thesis => "thesis", + ReferenceKind::Website => "web", + ReferenceKind::Software | ReferenceKind::Dataset => "repository", + ReferenceKind::Video => "video", + ReferenceKind::Other => "misc", + } +} + +// --- CSL-JSON --- + +/// Renders the bibliography as CSL-JSON. +fn csl_json(course: &CourseFile) -> Result { + let items: Vec = course + .references + .iter() + .map(|(key, reference)| csl_item(key, reference)) + .collect(); + serde_json::to_string_pretty(&items) + .map(|text| format!("{text}\n")) + .map_err(crate::error::Error::other) +} + +/// One CSL-JSON item. +fn csl_item(key: &str, reference: &Reference) -> Value { + let mut item = json!({ + "id": key, + "type": csl_kind(reference.kind), + "title": reference.title, + }); + let map = item.as_object_mut().expect("built from a JSON object"); + + if !reference.authors.is_empty() { + map.insert( + "author".to_string(), + Value::Array(reference.authors.iter().map(|a| csl_name(a)).collect()), + ); + } + if let Some(year) = reference.year { + map.insert("issued".to_string(), json!({ "date-parts": [[year]] })); + } + for (field, value) in [ + ("container-title", &reference.container), + ("publisher", &reference.publisher), + ("volume", &reference.volume), + ("issue", &reference.issue), + ("page", &reference.pages), + ("edition", &reference.edition), + ("DOI", &reference.doi), + ("ISBN", &reference.isbn), + ("PMID", &reference.pmid), + ("PMCID", &reference.pmcid), + ("URL", &reference.url), + ] { + if let Some(value) = value { + map.insert(field.to_string(), Value::String(value.clone())); + } + } + item +} + +/// Splits `Family, Given` into a CSL name, or keeps it whole. +/// +/// A name with no comma is not a name this code can take apart — an +/// organization, or a single mononym — so it goes in `literal`, which is what +/// CSL has the field for. +fn csl_name(author: &str) -> Value { + match author.split_once(',') { + Some((family, given)) => json!({ + "family": family.trim(), + "given": given.trim(), + }), + None => json!({ "literal": author.trim() }), + } +} + +/// The CSL type for a course reference kind. +fn csl_kind(kind: ReferenceKind) -> &'static str { + match kind { + ReferenceKind::Book => "book", + ReferenceKind::Chapter => "chapter", + ReferenceKind::Article => "article-journal", + ReferenceKind::Preprint => "article", + ReferenceKind::Thesis => "thesis", + ReferenceKind::Website => "webpage", + ReferenceKind::Software => "software", + ReferenceKind::Dataset => "dataset", + ReferenceKind::Video => "motion_picture", + ReferenceKind::Other => "document", + } +} + +// --- BibTeX --- + +/// Renders the bibliography as BibTeX. +fn bibtex(course: &CourseFile) -> String { + let mut out = String::new(); + for (key, reference) in &course.references { + out.push_str(&bibtex_entry(key, reference)); + out.push('\n'); + } + out +} + +/// One BibTeX entry. +fn bibtex_entry(key: &str, reference: &Reference) -> String { + let mut fields: Vec<(&str, String)> = vec![("title", reference.title.clone())]; + if !reference.authors.is_empty() { + fields.push(("author", reference.authors.join(" and "))); + } + if let Some(year) = reference.year { + fields.push(("year", year.to_string())); + } + if let Some(container) = &reference.container { + let field = match reference.kind { + ReferenceKind::Chapter => "booktitle", + _ => "journal", + }; + fields.push((field, container.clone())); + } + for (field, value) in [ + ("volume", &reference.volume), + ("number", &reference.issue), + ("edition", &reference.edition), + ("publisher", &reference.publisher), + ("doi", &reference.doi), + ("isbn", &reference.isbn), + ("url", &reference.url), + ("note", &reference.note), + ] { + if let Some(value) = value { + fields.push((field, value.clone())); + } + } + if let Some(pages) = &reference.pages { + fields.push(("pages", en_dash(pages))); + } + + let body: String = fields + .iter() + .map(|(field, value)| format!(" {field} = {{{}}},\n", escape_tex(value))) + .collect(); + format!("@{}{{{key},\n{body}}}\n", bibtex_kind(reference.kind)) +} + +/// The BibTeX entry type for a course reference kind. +fn bibtex_kind(kind: ReferenceKind) -> &'static str { + match kind { + ReferenceKind::Book => "book", + ReferenceKind::Chapter => "incollection", + ReferenceKind::Article => "article", + ReferenceKind::Thesis => "phdthesis", + ReferenceKind::Website => "online", + // BibTeX proper has nothing for these. `misc` with a `note` is what + // every style guide says to do, and biblatex users can convert. + ReferenceKind::Preprint + | ReferenceKind::Software + | ReferenceKind::Dataset + | ReferenceKind::Video + | ReferenceKind::Other => "misc", + } +} + +/// Escapes the characters BibTeX treats as syntax. +/// +/// Deliberately short: a publisher called `John Wiley & Sons` is the case that +/// actually occurs, and escaping more than this risks mangling the `$...$` in a +/// title that carries real mathematics. +fn escape_tex(value: &str) -> String { + value.replace('&', "\\&").replace('%', "\\%") +} + +/// Turns a hyphenated page range into the en dash BibTeX expects. +fn en_dash(pages: &str) -> String { + let parts: Vec<&str> = pages.split('-').collect(); + if parts.len() == 2 && parts.iter().all(|p| !p.is_empty()) { + return format!("{}--{}", parts[0].trim(), parts[1].trim()); + } + pages.to_string() +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::course::ReferenceRole; + + fn course() -> CourseFile { + let mut c = CourseFile::skeleton("BIOSC 1540", "Computational Biology", "2026f"); + c.references.insert( + "ismail2023bioinformatics".into(), + Reference { + label: Some("IBB".into()), + kind: ReferenceKind::Book, + role: ReferenceRole::Supplemental, + title: "Bioinformatics: A practical guide".into(), + authors: vec!["Ismail, H. D.".into()], + year: Some(2023), + publisher: Some("CRC Press".into()), + base_url: Some("https://library.example.org/ismail2023/".into()), + isbn: Some("9781032366423".into()), + ..Reference::default() + }, + ); + c.references.insert( + "altschul1990basic".into(), + Reference { + label: Some("BLAST".into()), + kind: ReferenceKind::Article, + role: ReferenceRole::Required, + title: "Basic local alignment search tool".into(), + authors: vec![ + "Altschul, S. F.".into(), + "Gish, W.".into(), + "Wiley & Sons".into(), + ], + year: Some(1990), + container: Some("Journal of Molecular Biology".into()), + volume: Some("215".into()), + issue: Some("3".into()), + pages: Some("403-410".into()), + doi: Some("10.1016/S0022-2836(05)80360-2".into()), + pmid: Some("2231712".into()), + ..Reference::default() + }, + ); + c + } + + #[test] + fn hayagriva_nests_an_article_under_its_periodical() { + let out = render(&course(), Format::Hayagriva).unwrap(); + assert!(out.contains("altschul1990basic:"), "{out}"); + assert!(out.contains("type: article"), "{out}"); + assert!(out.contains("type: periodical"), "{out}"); + assert!(out.contains("Journal of Molecular Biology"), "{out}"); + assert!(out.contains("page-range: 403-410"), "{out}"); + assert!(out.contains("doi: 10.1016/S0022-2836(05)80360-2"), "{out}"); + // Parses back as YAML, which is what Typst will do to it. + let back: serde_yaml_ng::Value = serde_yaml_ng::from_str(&out).unwrap(); + assert!(back.get("ismail2023bioinformatics").is_some()); + } + + #[test] + fn hayagriva_keeps_a_books_publisher_on_the_book() { + let out = render(&course(), Format::Hayagriva).unwrap(); + let parsed: serde_yaml_ng::Value = serde_yaml_ng::from_str(&out).unwrap(); + let book = parsed.get("ismail2023bioinformatics").unwrap(); + assert_eq!(book.get("publisher").unwrap().as_str(), Some("CRC Press")); + assert!(book.get("parent").is_none()); + // `label`, `role`, and `base_url` have nowhere to go and stay behind. + assert!(book.get("label").is_none()); + assert!(book.get("base_url").is_none()); + } + + #[test] + fn csl_json_splits_names_and_keeps_organizations_whole() { + let out = render(&course(), Format::CslJson).unwrap(); + let items: Vec = serde_json::from_str(&out).unwrap(); + let article = items + .iter() + .find(|i| i["id"] == "altschul1990basic") + .unwrap(); + assert_eq!(article["type"], "article-journal"); + assert_eq!(article["author"][0]["family"], "Altschul"); + assert_eq!(article["author"][0]["given"], "S. F."); + assert_eq!(article["author"][2]["literal"], "Wiley & Sons"); + assert_eq!(article["issued"]["date-parts"][0][0], 1990); + assert_eq!(article["DOI"], "10.1016/S0022-2836(05)80360-2"); + } + + #[test] + fn bibtex_escapes_ampersands_and_dashes_a_page_range() { + let out = render(&course(), Format::Bibtex).unwrap(); + assert!(out.contains("@article{altschul1990basic,"), "{out}"); + assert!( + out.contains("journal = {Journal of Molecular Biology},"), + "{out}" + ); + assert!(out.contains("pages = {403--410},"), "{out}"); + assert!(out.contains("Wiley \\& Sons"), "{out}"); + assert!(out.contains("@book{ismail2023bioinformatics,"), "{out}"); + } + + #[test] + fn a_course_with_no_bibliography_renders_empty_rather_than_failing() { + let mut c = course(); + c.references.clear(); + assert_eq!(render(&c, Format::Bibtex).unwrap(), ""); + assert_eq!(render(&c, Format::CslJson).unwrap().trim(), "[]"); + } +} diff --git a/src/export/report.rs b/src/export/report.rs index f084047..c049f81 100644 --- a/src/export/report.rs +++ b/src/export/report.rs @@ -36,7 +36,7 @@ use crate::course::CourseFile; use crate::date::Date; use crate::irt::Fit; use crate::students::{Cohort, Mastery, StudentSummary}; -use crate::taxonomy::Level; +use crate::taxonomy::{Level, Tier}; /// What to include in a student report. #[derive(Debug, Clone)] @@ -98,7 +98,7 @@ pub fn student( )); out.push_str(&format!("**{}**\n\n", summary.display_name())); - // ---------------------------------------------------------------- score + // --- score out.push_str(&format!( "You scored **{:.1} of {:.1} points ({:.0}%)**", summary.points, summary.points_possible, summary.percent @@ -144,17 +144,38 @@ pub fn student( } } - // ----------------------------------------------------------- objectives - if opts.objectives && !summary.objectives.is_empty() { + // --- objectives + // + // Objectives only. A target row would rest on one question and could only + // ever read "not enough questions to say", so a mixed table buried the few + // real classifications among dozens of non-statements. The targets appear + // where they can be acted on instead: under each lecture in "what to + // revise", and beside each missed question. + if opts.objectives && summary.objectives.iter().any(|o| o.tier == Tier::Objective) { out.push_str("## What this exam says about each learning objective\n\n"); out.push_str("| | Objective | You | Class | Items |\n|:--|:--|--:|--:|--:|\n"); - for o in &summary.objectives { - let you = format!("{:.0}%", o.rate * 100.0); + for o in summary + .objectives + .iter() + .filter(|o| o.tier == Tier::Objective) + { + // How much of the objective this exam reached, which is the scope of + // the claim the row makes. + let scope = if o.targets_total > 0 { + format!( + "{} ({} of {} targets tested)", + escape_pipes(&o.text), + o.targets_seen, + o.targets_total + ) + } else { + escape_pipes(&o.text) + }; out.push_str(&format!( - "| {} | {} | {} | {:.0}% | {} |\n", + "| {} | {} | {:.0}% | {:.0}% | {} |\n", o.status.symbol(), - escape_pipes(&o.text), - you, + scope, + o.rate * 100.0, o.cohort_rate * 100.0, o.n_items )); @@ -167,7 +188,7 @@ pub fn student( let thin: Vec<&str> = summary .objectives .iter() - .filter(|o| o.status == Mastery::NotEnoughEvidence) + .filter(|o| o.tier == Tier::Objective && o.status == Mastery::NotEnoughEvidence) .map(|o| o.text.as_str()) .collect(); if !thin.is_empty() { @@ -180,7 +201,7 @@ pub fn student( } } - // --------------------------------------------------------------- levels + // --- levels if opts.levels && summary.levels.len() > 1 { out.push_str("## Kinds of thinking\n\n"); out.push_str( @@ -231,12 +252,12 @@ pub fn student( } } - // ---------------------------------------------------------- what to do + // --- what to do if !summary.focus.is_empty() { out.push_str("## Where to put your time\n\n"); out.push_str("In this order:\n\n"); for (i, id) in summary.focus.iter().take(4).enumerate() { - let text = course.objective_text(id); + let text = course.text_for(id); out.push_str(&format!("{}. {}\n", i + 1, text)); } out.push('\n'); @@ -250,10 +271,7 @@ pub fn student( .map(|(id, _)| id.as_str()) .collect(); if !class_gaps.is_empty() { - let texts: Vec = class_gaps - .iter() - .map(|id| course.objective_text(id)) - .collect(); + let texts: Vec = class_gaps.iter().map(|id| course.text_for(id)).collect(); let refs: Vec<&str> = texts.iter().map(|s| s.as_str()).collect(); out.push_str(&format!( "Most of the class also struggled with {}, so expect it to come back in class. \ @@ -268,13 +286,13 @@ pub fn student( .strengths .iter() .take(4) - .map(|id| course.objective_text(id)) + .map(|id| course.text_for(id)) .collect(); let refs: Vec<&str> = texts.iter().map(|s| s.as_str()).collect(); out.push_str(&format!("You have clearly got {}.\n\n", list(&refs))); } - // --------------------------------------------------------- missed items + // --- missed items if opts.missed && !summary.missed.is_empty() { out.push_str("## Question by question\n\n"); out.push_str( @@ -288,6 +306,20 @@ pub fn student( String::new() }; out.push_str(&format!("**Question {}**{partial}\n\n", m.number)); + // Both tiers: the target says what this question asked, the + // objective says which row of the table above it counted toward. + for target in &m.learning_targets { + let objective = course.objective_for(target); + if objective == target.as_str() { + out.push_str(&format!("Asked you to: {}\n\n", course.text_for(target))); + } else { + out.push_str(&format!( + "Asked you to: {} \nCounts toward: {}\n\n", + course.text_for(target), + course.text_for(objective) + )); + } + } if let Some(text) = &m.feedback { out.push_str(&format!("{text}\n\n")); } else if let Some(misconception) = &m.misconception { @@ -356,7 +388,7 @@ pub fn cohort( .unwrap_or_else(|| "date not recorded".into()) )); - // ------------------------------------------------------------- summary + // --- summary let r = &analysis.reliability; out.push_str("## Summary\n\n"); out.push_str(&format!( @@ -408,7 +440,7 @@ pub fn cohort( out.push_str(&format!("> {w}\n\n")); } - // ------------------------------------------------------- revise queue + // --- revise queue let queue = analysis.revise_queue(); out.push_str("## What to revise\n\n"); if queue.is_empty() { @@ -477,7 +509,7 @@ pub fn cohort( } } - // ---------------------------------------------------------- item table + // --- item table out.push_str("## Every item\n\n"); out.push_str( "| Q | Item | Lv | p | r | D | Blank | Flags |\n|--:|:--|--:|--:|--:|--:|--:|:--|\n", @@ -535,7 +567,7 @@ pub fn cohort( } } - // -------------------------------------------------------- class gaps + // --- class gaps out.push_str("## Objectives the class did not meet\n\n"); if cohort.class_gaps.is_empty() { out.push_str("Every assessed objective cleared the mastery threshold.\n\n"); @@ -549,14 +581,14 @@ pub fn cohort( for (id, rate) in &cohort.class_gaps { out.push_str(&format!( "| {} | {:.0}% |\n", - escape_pipes(&course.objective_text(id)), + escape_pipes(&course.text_for(id)), rate * 100.0 )); } out.push('\n'); } - // ------------------------------------------------------ level coverage + // --- level coverage out.push_str("## Coverage and class performance by level\n\n"); let counts = record.level_counts(); out.push_str("| Level | Items | Class rate |\n|:--|--:|--:|\n"); @@ -576,7 +608,7 @@ pub fn cohort( out.push('\n'); if let Some(bp) = &record.blueprint { - let drift = crate::select::check_blueprint(record); + let drift = crate::select::check_blueprint(record, course); if !drift.is_empty() { out.push_str("Blueprint drift:\n\n"); for d in &drift { @@ -587,7 +619,7 @@ pub fn cohort( let _ = bp; } - // -------------------------------------------------------- archetypes + // --- archetypes if !cohort.archetypes.is_empty() { out.push_str("## Patterns across students\n\n"); out.push_str( @@ -956,6 +988,10 @@ pub fn write_all_students( /// Per-objective class rates as a compact table, for pasting into a syllabus /// review or a curriculum committee document. /// +/// Objectives only. A committee document listing three hundred learning targets +/// is not read, and the target rates are the wrong number to put in front of one +/// anyway: each rests on one or two questions. +/// /// # Arguments /// /// * `cohort` - the class. @@ -965,7 +1001,7 @@ pub fn write_all_students( /// /// A Markdown table. pub fn objective_summary(cohort: &Cohort, course: &CourseFile) -> String { - let mut out = String::from("| Objective | Class rate |\n|:--|--:|\n"); + let mut out = String::from("| Objective | Class rate | Items |\n|:--|--:|--:|\n"); let mut rows: Vec<(&String, &f64)> = cohort.objective_rates.iter().collect(); rows.sort_by(|a, b| { a.1.partial_cmp(b.1) @@ -973,10 +1009,20 @@ pub fn objective_summary(cohort: &Cohort, course: &CourseFile) -> String { .then_with(|| a.0.cmp(b.0)) }); for (id, rate) in rows { + // Items behind the rate, because a rate without a denominator is what + // makes a committee table misleading. + let n = cohort + .students + .iter() + .flat_map(|s| s.objectives.iter()) + .find(|o| &o.id == id) + .map(|o| o.n_items) + .unwrap_or(0); out.push_str(&format!( - "| {} | {:.0}% |\n", - escape_pipes(&course.objective_text(id)), - rate * 100.0 + "| {} | {:.0}% | {} |\n", + escape_pipes(&course.text_for(id)), + rate * 100.0, + n )); } out diff --git a/src/export/site.rs b/src/export/site.rs new file mode 100644 index 0000000..d7be2e1 --- /dev/null +++ b/src/export/site.rs @@ -0,0 +1,1025 @@ +// SPDX-License-Identifier: Prosperity-3.0.0 +// Copyright Scientific Computing Studio +// Source: https://git.scient.ing/education/coursebank + +//! Rendering an assessment for a Quarto course website, with solutions gated +//! behind a per-page password. +//! +//! This is the path that publishes to the web rather than to Canvas or a printed +//! exam. It produces three things for one assessment: +//! +//! 1. A `_questions.qmd` partial: the questions as Quarto fenced divs, with an +//! empty, hidden `.qsol` slot per item. A page includes it with +//! `{{< include _questions.qmd >}}`. It carries no answers. +//! 2. A `-solutions.json` bundle: every solution rendered to an HTML fragment, +//! then encrypted. The ciphertext ships in the page, but the plaintext never +//! does, so a student cannot read answers from the source or the network tab. +//! 3. A password: a fresh, random, per-bundle password that decrypts the bundle. +//! It is printed for the instructor and stored nowhere, so it cannot be +//! recovered from the files. Hand it out, and rotate it after a due date. +//! +//! The browser side is [`assets`]: `questions.css` styles the questions and the +//! unlocked solutions, and `solutions.js` derives the key from the typed password, +//! decrypts, and injects each fragment. The crypto here matches that script byte +//! for byte: PBKDF2-HMAC-SHA256 at 250,000 iterations derives a 256-bit key, and +//! AES-256-GCM encrypts each fragment under a fresh 96-bit IV with the 128-bit tag +//! appended to the ciphertext. Get any parameter wrong and a correct password +//! would fail to authenticate. +//! +//! Two rules carry over from the other exporters. A question paper never contains +//! the answer: the `.qsol` slot is empty in the qmd and the answer lives only in +//! the encrypted bundle. And option order comes from the form's seed, so the +//! printed letters in the questions and the letters the solution refers to are the +//! same order. + +use crate::assessment::{AssessmentFile, Form, Placement}; +use crate::catalog::Catalog; +use crate::course::CourseFile; +use crate::error::{Error, Result}; +use crate::item::{Choice, Citation, Item, Solution}; +use crate::markup; +use crate::select; +use crate::taxonomy::Format; + +use aes_gcm::Aes256Gcm; +use aes_gcm::aead::generic_array::GenericArray; +use aes_gcm::aead::{Aead, KeyInit}; +use base64::Engine as _; +use base64::engine::general_purpose::STANDARD as B64; +use serde::ser::SerializeMap; +use serde::{Serialize, Serializer}; + +/// PBKDF2 iteration count. Must match `solutions.js` and `make_solutions.py`. +const ITERATIONS: u32 = 250_000; + +/// Crockford base32 without the ambiguous `i`, `l`, `o`, `u`. Thirty-two symbols +/// means five bits each, so sixteen of them carry eighty bits of entropy. +const PW_ALPHABET: &[u8; 32] = b"0123456789abcdefghjkmnpqrstvwxyz"; + +/// The bundled browser assets, installed once per site (not per page). +/// +/// # Returns +/// +/// Pairs of file name and verbatim contents: the stylesheet and the unlock +/// script. Write them into the site's static directory and wire them in +/// `_quarto.yml`. +pub fn assets() -> [(&'static str, &'static str); 2] { + [ + ( + "questions.css", + include_str!("../assets/site/questions.css"), + ), + ("solutions.js", include_str!("../assets/site/solutions.js")), + ] +} + +/// How to render one assessment for the site. +#[derive(Debug, Clone)] +pub struct Options { + /// The form whose option order to print. + pub form: Form, + /// A password to encrypt with. `None` generates a fresh random one, which is + /// the intended path; pass `Some` only to re-encrypt a bundle with a known + /// password. + pub password: Option, +} + +impl Options { + /// Builds options for a form, generating the password. + /// + /// # Arguments + /// + /// * `form` - the form whose option order to print. + /// + /// # Returns + /// + /// Options that will generate a fresh password at render time. + pub fn new(form: Form) -> Options { + Options { + form, + password: None, + } + } +} + +/// Everything one render produces, ready to write. +#[derive(Debug, Clone)] +pub struct Rendered { + /// The assessment id, used for the bundle file name and the gate's + /// `data-bundle` attribute. + pub page: String, + /// The `_questions.qmd` partial. + pub questions_qmd: String, + /// The `-solutions.json` bundle, serialized and newline-terminated. + pub solutions_json: String, + /// The password that decrypts the bundle. Print it; it is stored nowhere. + pub password: String, +} + +/// Renders an assessment into the questions partial and the encrypted bundle. +/// +/// # Arguments +/// +/// * `catalog` - the loaded course, for items, objectives, and references. +/// * `record` - the assembled assessment. +/// * `opts` - the form and an optional password. +/// +/// # Returns +/// +/// The partial, the serialized bundle, and the password. +/// +/// # Errors +/// +/// Returns [`Error::Unresolved`] when a placement names an item the catalog does +/// not hold, and [`Error::Other`] on a CSPRNG, encryption, or serialization +/// failure. +pub fn render(catalog: &Catalog, record: &AssessmentFile, opts: Options) -> Result { + let page = record.assessment.id.clone(); + let questions_qmd = questions_qmd(catalog, record, &opts.form, &page)?; + let fragments = solution_fragments(catalog, record, &opts.form)?; + let password = match opts.password { + Some(p) => p, + None => gen_password()?, + }; + let bundle = build_bundle(&page, &password, &fragments)?; + let json = serde_json::to_string_pretty(&bundle) + .map_err(|e| Error::other(format!("could not serialize the solutions bundle: {e}")))?; + Ok(Rendered { + page, + questions_qmd, + solutions_json: format!("{json}\n"), + password, + }) +} + +/// Builds the `_questions.qmd` partial. +fn questions_qmd( + catalog: &Catalog, + record: &AssessmentFile, + form: &Form, + page: &str, +) -> Result { + let mut out = String::new(); + out.push_str( + "\n\n", + ); + out.push_str(&format!( + "::: {{.solutions-gate data-bundle=\"{page}-solutions.json\"}}\n:::\n\n" + )); + + let mut number = 0usize; + let default_points = catalog.course.policy.points_per_item; + let layout = select::layout(record, form); + for placement in layout.iter().filter(|p| !p.dropped) { + let item = &catalog.require(&placement.item)?.item; + number += 1; + out.push_str(&question_block( + number, + placement, + item, + form, + default_points, + )); + out.push('\n'); + } + + // Leave exactly one trailing newline. + while out.ends_with("\n\n") { + out.pop(); + } + Ok(out) +} + +/// One `.q` block: head, stem, choices or a writing box, then the empty slot. +fn question_block( + number: usize, + placement: &Placement, + item: &Item, + form: &Form, + default_points: f64, +) -> String { + let id = &item.id; + let points = placement + .points + .unwrap_or_else(|| item.points(default_points)); + let unit = if (points - 1.0).abs() < f64::EPSILON { + "point" + } else { + "points" + }; + + let mut b = String::new(); + b.push_str(&format!("::: {{.q #{id}}}\n")); + b.push_str(":::: {.q-head}\n"); + b.push_str(&format!( + "[Question {number}]{{.q-num}} [{}]{{.q-kind}} [{} {unit}]{{.q-points}}\n", + kind_label(item.format), + trim_number(points), + )); + b.push_str("::::\n\n"); + + b.push_str(":::: {.q-stem}\n"); + b.push_str(&markup::to_markdown(&item.stem)); + b.push_str("\n::::\n\n"); + + if item.has_options() { + b.push_str(":::: {.q-choices}\n"); + let shown = item.administered(&placement.key, &placement.distractors); + let order = select::option_order(form, &placement.item, shown.len()); + for (position, &source) in order.iter().enumerate() { + b.push_str(&format!( + "{}. {}\n", + position + 1, + markup::to_markdown(&shown[source].text) + )); + } + b.push_str("::::\n\n"); + } else { + b.push_str(":::: {.q-response aria-hidden=\"true\"}\n::::\n\n"); + } + + b.push_str(&format!( + ":::: {{.qsol data-solution-for=\"{id}\" hidden=\"true\"}}\n::::\n" + )); + b.push_str(":::\n"); + b +} + +/// The human label for a format, shown in the question head. +fn kind_label(format: Format) -> &'static str { + match format { + Format::SingleBestAnswer => "Single best answer", + Format::MultipleResponse => "Multiple response", + Format::TrueFalse => "True or false", + Format::OpenResponse => "Open response", + } +} + +// --- the solution fragments --- + +/// Renders each item's solution to an HTML fragment, in printed order. +/// +/// An item with nothing to show (an open-response question whose solution is +/// empty) is left out, so its slot simply never unlocks. +fn solution_fragments( + catalog: &Catalog, + record: &AssessmentFile, + form: &Form, +) -> Result> { + let mut out = Vec::new(); + let layout = select::layout(record, form); + for placement in layout.iter().filter(|p| !p.dropped) { + let item = &catalog.require(&placement.item)?.item; + if let Some(html) = fragment(&catalog.course, placement, item, form) { + out.push((item.id.clone(), html)); + } + } + Ok(out) +} + +/// The fragment for one item, or `None` when there is nothing to show. +fn fragment( + course: &CourseFile, + placement: &Placement, + item: &Item, + form: &Form, +) -> Option { + if item.has_options() { + Some(choice_fragment(course, placement, item, form)) + } else { + open_fragment(course, item) + } +} + +/// A single-best-answer or multiple-response fragment: the key, the model answer, +/// the explanation, then per-distractor feedback. +fn choice_fragment(course: &CourseFile, placement: &Placement, item: &Item, form: &Form) -> String { + let shown = item.administered(&placement.key, &placement.distractors); + let order = select::option_order(form, &placement.item, shown.len()); + let printed: Vec<(usize, &Choice)> = order + .iter() + .enumerate() + .map(|(position, &source)| (position, shown[source])) + .collect(); + + let mut out = String::new(); + + let keyed: Vec = printed + .iter() + .filter(|(_, c)| c.correct) + .map(|(position, c)| { + format!( + "{} — {}", + letter(*position), + inline_html(&c.text) + ) + }) + .collect(); + out.push_str( + "

Correct answer ", + ); + out.push_str(&keyed.join("; ")); + out.push_str("

\n"); + + if let Some(solution) = item.solution.as_ref() { + if let Some(model) = &solution.model_answer { + out.push_str(&format!( + "
{}
\n", + block_html(model) + )); + } + if let Some(explanation) = &solution.explanation { + out.push_str(&explain_html(explanation)); + } + } + + let distractors: Vec<(usize, &Choice)> = printed + .iter() + .copied() + .filter(|(_, c)| !c.correct) + .collect(); + if distractors + .iter() + .any(|(_, c)| c.misconception.is_some() || why_wrong(c).is_some()) + { + out.push_str("\n"); + out.push_str("
    \n"); + for (position, c) in &distractors { + if c.misconception.is_none() && why_wrong(c).is_none() { + continue; + } + out.push_str(&format!( + "
  • {}\n", + letter(*position) + )); + out.push_str("
    \n"); + if let Some(mis) = &c.misconception { + out.push_str(&format!( + " {}\n", + inline_html(mis) + )); + } + if let Some(why) = why_wrong(c) { + out.push_str(&format!( + " {}\n", + inline_html(why) + )); + } + out.push_str("
  • \n"); + } + out.push_str("
\n"); + } + + push_reference(&mut out, course, item.solution.as_ref()); + out +} + +/// An open-response fragment: the model answer, the explanation, the rubric, and +/// the accepted variants. +fn open_fragment(course: &CourseFile, item: &Item) -> Option { + let solution = item.solution.as_ref().filter(|s| !s.is_empty())?; + let mut out = String::new(); + + if let Some(model) = &solution.model_answer { + out.push_str(&format!( + "
{}
\n", + block_html(model) + )); + } + + if let Some(explanation) = &solution.explanation { + out.push_str(&explain_html(explanation)); + } + + if !solution.rubric.is_empty() { + let caption = match solution.rubric_points() { + Some(total) => { + let unit = if (total - 1.0).abs() < f64::EPSILON { + "point" + } else { + "points" + }; + format!("Rubric — {} {unit}", trim_number(total)) + } + None => "Rubric".to_string(), + }; + out.push_str("\n"); + out.push_str(&format!(" \n")); + out.push_str( + " \n", + ); + out.push_str(" \n"); + for criterion in &solution.rubric { + let pts = criterion.points.map(trim_number).unwrap_or_default(); + out.push_str(&format!( + " \n", + inline_html(&criterion.description) + )); + } + out.push_str(" \n
{caption}
PtsCriterion
{pts}{}
\n"); + } + + if !solution.accepted.is_empty() { + let joined = solution + .accepted + .iter() + .map(|a| inline_html(a)) + .collect::>() + .join("; "); + out.push_str(&format!( + "

Also accepted {joined}

\n" + )); + } + + push_reference(&mut out, course, Some(solution)); + Some(out) +} + +/// The student-facing reason a distractor is wrong: the instructor explanation +/// first, then any student feedback. +fn why_wrong(choice: &Choice) -> Option<&str> { + choice + .explanation + .as_deref() + .or(choice.feedback_student.as_deref()) +} + +/// Appends the `Source:` line when the solution carries review citations. +fn push_reference(out: &mut String, course: &CourseFile, solution: Option<&Solution>) { + let Some(solution) = solution else { return }; + if solution.review.is_empty() { + return; + } + let sources = solution + .review + .iter() + .map(|c| cite_html(course, c)) + .collect::>() + .join("; "); + out.push_str(&format!("

Source: {sources}

\n")); +} + +/// Renders one citation to HTML, linking it when a URL resolves. +fn cite_html(course: &CourseFile, citation: &Citation) -> String { + if let Some(text) = &citation.text { + if citation.reference.is_none() { + return markup::escape_html(text); + } + } + let Some(key) = &citation.reference else { + return markup::escape_html(&citation.display()); + }; + let Some(reference) = course.references.get(key) else { + return markup::escape_html(&citation.display()); + }; + let label = reference.label_or(key); + let locator = citation.locator.as_deref().unwrap_or(""); + let body = if locator.is_empty() { + markup::escape_html(label) + } else { + format!( + "{} {}", + markup::escape_html(label), + markup::escape_html(locator) + ) + }; + match citation.href(reference) { + Some(url) => format!("{body}", markup::escape_html(&url)), + None => body, + } +} + +// --- math-aware markup --- + +/// One run of source text, split on math delimiters. +enum Segment<'a> { + /// Prose, formatted with the shared escape, symbol, and inline rules. + Text(&'a str), + /// A math span, passed through with its interior escaped for the browser. + Math { inner: &'a str, display: bool }, +} + +/// Splits source into prose and `$…$` or `$$…$$` math runs. +/// +/// The split runs on raw source so a formatting rule can never reach inside math: +/// an underscore in `$q_p$` is a subscript, not the start of an emphasis span. +fn split_math(src: &str) -> Vec> { + let mut segments = Vec::new(); + let mut rest = src; + while let Some(at) = rest.find('$') { + if at > 0 { + segments.push(Segment::Text(&rest[..at])); + } + let after = &rest[at..]; + if let Some(display) = after.strip_prefix("$$") { + if let Some(end) = display.find("$$") { + segments.push(Segment::Math { + inner: &display[..end], + display: true, + }); + rest = &display[end + 2..]; + continue; + } + } + let inline = &after[1..]; + match inline.find('$') { + Some(end) => { + segments.push(Segment::Math { + inner: &inline[..end], + display: false, + }); + rest = &inline[end + 1..]; + } + None => { + // An unterminated `$` is treated as ordinary text. + segments.push(Segment::Text(after)); + rest = ""; + } + } + } + if !rest.is_empty() { + segments.push(Segment::Text(rest)); + } + segments +} + +/// Formats a prose run: HTML-escape, then the symbol table and inline markup. +fn format_prose(text: &str) -> String { + markup::apply_inline(&markup::apply_symbols(&markup::escape_html(text), true)) +} + +/// Emits a math run with its delimiters, escaping the interior so the browser +/// hands MathJax clean text (a `<` inside math becomes `<`, which the DOM +/// decodes back before MathJax reads it). +fn render_math(inner: &str, display: bool) -> String { + let delim = if display { "$$" } else { "$" }; + format!("{delim}{}{delim}", markup::escape_html(inner)) +} + +/// Converts authoring markup to an inline HTML string, preserving math. +/// +/// No paragraph wrapping: the caller supplies the surrounding element. +fn inline_html(src: &str) -> String { + let mut out = String::new(); + for segment in split_math(src.trim()) { + match segment { + Segment::Text(text) => out.push_str(&format_prose(text)), + Segment::Math { inner, display } => out.push_str(&render_math(inner, display)), + } + } + out +} + +/// Like [`inline_html`], but wraps each blank-line-separated paragraph in `

`. +fn block_html(src: &str) -> String { + src.split("\n\n") + .map(str::trim) + .filter(|p| !p.is_empty()) + .map(|p| format!("

{}

", inline_html(p))) + .collect::>() + .join("") +} + +/// Renders a solution explanation as one `

` per blank-line- +/// separated paragraph. A single-paragraph explanation emits exactly one such +/// paragraph, unchanged from before; a multi-paragraph one keeps its breaks, and +/// every paragraph carries the class `questions.css` already styles. +fn explain_html(src: &str) -> String { + src.split("\n\n") + .map(str::trim) + .filter(|p| !p.is_empty()) + .map(|p| format!("

{}

\n", inline_html(p))) + .collect() +} + +// --- the encrypted bundle ----- + +/// The encrypted solutions bundle, matching the `solutions.js` v1 format. +#[derive(Debug, Clone, Serialize)] +pub struct Bundle { + v: u8, + page: String, + kdf: Kdf, + cipher: &'static str, + items: OrderedItems, +} + +#[derive(Debug, Clone, Serialize)] +struct Kdf { + name: &'static str, + hash: &'static str, + iterations: u32, + salt: String, +} + +#[derive(Debug, Clone, Serialize)] +struct Enc { + iv: String, + ct: String, +} + +/// Item entries serialized as a JSON object in insertion order, so the bundle +/// lists solutions in the order the questions appear rather than sorted by id. +#[derive(Debug, Clone)] +struct OrderedItems(Vec<(String, Enc)>); + +impl Serialize for OrderedItems { + fn serialize(&self, serializer: S) -> std::result::Result { + let mut map = serializer.serialize_map(Some(self.0.len()))?; + for (id, enc) in &self.0 { + map.serialize_entry(id, enc)?; + } + map.end() + } +} + +/// Generates a random per-bundle password. +/// +/// Sixteen Crockford base32 symbols, grouped in fours, for eighty bits of entropy +/// from the system CSPRNG, for example `k7m4-9p2q-r8tx-3wn6`. +/// +/// # Returns +/// +/// The password, to print for the instructor. +/// +/// # Errors +/// +/// Returns [`Error::Other`] if the system CSPRNG is unavailable. +pub fn gen_password() -> Result { + let mut raw = [0u8; 16]; + csprng(&mut raw)?; + // Thirty-two divides 256, so the modulo is unbiased. + let symbols: Vec = raw + .iter() + .map(|b| PW_ALPHABET[(*b % 32) as usize]) + .collect(); + let groups: Vec = symbols + .chunks(4) + .map(|c| String::from_utf8_lossy(c).into_owned()) + .collect(); + Ok(groups.join("-")) +} + +/// Encrypts rendered fragments into a bundle. +/// +/// # Arguments +/// +/// * `page` - the assessment id, stored as `page` and echoed by the script. +/// * `password` - the password to derive the key from. +/// * `fragments` - item id and solution HTML, in the order to list them. +/// +/// # Returns +/// +/// The bundle, ready to serialize as `-solutions.json`. +/// +/// # Errors +/// +/// Returns [`Error::Other`] on a CSPRNG or encryption failure. +pub fn build_bundle(page: &str, password: &str, fragments: &[(String, String)]) -> Result { + let mut salt = [0u8; 16]; + csprng(&mut salt)?; + let key = derive_key(password, &salt); + let cipher = Aes256Gcm::new_from_slice(&key) + .map_err(|_| Error::other("the derived AES key was the wrong length"))?; + + let mut items = Vec::with_capacity(fragments.len()); + for (id, html) in fragments { + let mut iv = [0u8; 12]; + csprng(&mut iv)?; + let nonce = GenericArray::from_slice(&iv); + let ct = cipher + .encrypt(nonce, html.as_bytes()) + .map_err(|_| Error::other("AES-GCM encryption failed"))?; + items.push(( + id.clone(), + Enc { + iv: B64.encode(iv), + ct: B64.encode(ct), + }, + )); + } + + Ok(Bundle { + v: 1, + page: page.to_string(), + kdf: Kdf { + name: "PBKDF2", + hash: "SHA-256", + iterations: ITERATIONS, + salt: B64.encode(salt), + }, + cipher: "AES-GCM", + items: OrderedItems(items), + }) +} + +/// Derives the AES-256 key with PBKDF2-HMAC-SHA256. +fn derive_key(password: &str, salt: &[u8]) -> [u8; 32] { + let mut key = [0u8; 32]; + pbkdf2::pbkdf2_hmac::(password.as_bytes(), salt, ITERATIONS, &mut key); + key +} + +/// Fills a buffer with cryptographically secure random bytes. +fn csprng(buf: &mut [u8]) -> Result<()> { + getrandom::getrandom(buf).map_err(|e| Error::other(format!("system CSPRNG unavailable: {e}"))) +} + +// --- small helpers ------- + +/// Formats a point value: no decimal when whole, at most two places otherwise. +fn trim_number(value: f64) -> String { + if value.fract() == 0.0 { + format!("{}", value as i64) + } else { + let s = format!("{value:.2}"); + s.trim_end_matches('0').trim_end_matches('.').to_string() + } +} + +/// The printed letter for a position: 0 is A, 1 is B, and so on. +fn letter(position: usize) -> char { + char::from(b'A' + (position % 26) as u8) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn form() -> Form { + Form { + id: "A".into(), + seed: 0, + shuffle_items: false, + shuffle_options: false, + } + } + + fn catalog(tag: &str) -> Catalog { + let dir = std::env::temp_dir().join(format!("cb-site-{tag}-{}", std::process::id())); + let _ = std::fs::remove_dir_all(&dir); + std::fs::create_dir_all(dir.join("banks")).unwrap(); + std::fs::write( + dir.join("course.yaml"), + r#" +course: { code: BIOSC 1000, title: Biochemistry, term: 2026f } +references: + kkw: + label: KKW + title: The molecules of life + base_url: https://example.org/kkw/ +lectures: + L1.1: { title: Enthalpy } +learning_objectives: + lo-enthalpy: + text: Define enthalpy and explain the constant-pressure result. + lectures: [L1.1] + order: 1 +"#, + ) + .unwrap(); + std::fs::write( + dir.join("banks").join("b.yaml"), + r#" +bank: { id: b, title: Bank } +items: + - id: q-mcq + status: draft + level: 2 + format: single_best_answer + stem: "The heat at constant pressure equals a change in what?" + learning_targets: [lo-enthalpy] + options: + - { id: A, text: "Enthalpy, $\\Delta H$", correct: true, feedback_student: "Right, $q_p = \\Delta H$." } + - { id: B, text: "Internal energy, $\\Delta U$", misconception: "Uses the constant-volume result.", explanation: "That holds only at constant volume." } + solution: + model_answer: "The change in enthalpy, $\\Delta H$." + review: + - { ref: kkw, locator: "§6.3" } + - id: q-open + status: draft + level: 3 + format: open_response + stem: "Show why $q_p = \\Delta H$." + learning_targets: [lo-enthalpy] + solution: + model_answer: "From $H = U + PV$ at constant pressure, $q_p = \\Delta H$." + explanation: | + At constant pressure the pressure-volume work is folded into H, so the heat equals the change in H. + + That is why a calorimeter run at constant pressure reads the enthalpy change directly. + rubric: + - { description: "States $H = U + PV$.", points: 1 } + - { description: "Reaches $q_p = \\Delta H$.", points: 1 } + accepted: ["$q_p = \\Delta H$ via $H = U + PV$"] +"#, + ) + .unwrap(); + Catalog::load(&dir).expect("catalog loads") + } + + fn record() -> AssessmentFile { + use crate::assessment::{Assessment, Kind, Platform}; + AssessmentFile { + schema_version: "1.0".into(), + assessment: Assessment { + id: "a1.1".into(), + title: "Homework 1".into(), + term: None, + kind: Kind::Homework, + date: None, + platform: Platform::Other, + minutes_allowed: None, + attempts: None, + shuffle: None, + scoring_policy: None, + instructions: None, + notes: None, + }, + blueprint: None, + forms: Vec::new(), + items: vec![ + Placement { + number: 1, + item: "b::q-mcq".into(), + version: None, + stem_digest: None, + distractors: Vec::new(), + variant: None, + fingerprint: None, + points: Some(1.0), + bonus: false, + key: vec!["A".into()], + level: None, + learning_targets: Vec::new(), + credit_overrides: Default::default(), + dropped: false, + dropped_as: None, + dropped_before_printing: false, + }, + Placement { + number: 2, + item: "b::q-open".into(), + version: None, + stem_digest: None, + distractors: Vec::new(), + variant: None, + fingerprint: None, + points: Some(2.0), + bonus: false, + key: Vec::new(), + level: None, + learning_targets: Vec::new(), + credit_overrides: Default::default(), + dropped: false, + dropped_as: None, + dropped_before_printing: false, + }, + ], + } + } + + #[test] + fn questions_partial_has_no_front_matter_and_no_answers() { + let out = questions_qmd(&catalog("qmd"), &record(), &form(), "a1.1").unwrap(); + assert!(!out.starts_with("---"), "a partial carries no front matter"); + assert!(out.contains("::: {.solutions-gate data-bundle=\"a1.1-solutions.json\"}")); + assert!(out.contains("::: {.q #q-mcq}")); + assert!( + out.contains("[Question 1]{.q-num} [Single best answer]{.q-kind} [1 point]{.q-points}") + ); + // Choices are printed without letters; the correct flag never appears. + assert!(out.contains("1. Enthalpy, $\\Delta H$")); + assert!(!out.contains("correct")); + // The open-response question gets a writing box, both get an empty slot. + assert!(out.contains(":::: {.q-response aria-hidden=\"true\"}")); + assert!(out.contains(":::: {.qsol data-solution-for=\"q-mcq\" hidden=\"true\"}")); + assert!(out.contains("[2 points]{.q-points}")); + // No model answer leaks into the questions. + assert!(!out.contains("sol-model")); + } + + #[test] + fn a_choice_fragment_marks_the_key_and_explains_the_distractor() { + let cat = catalog("choice"); + let rec = record(); + let frags = solution_fragments(&cat, &rec, &form()).unwrap(); + let mcq = &frags.iter().find(|(id, _)| id == "q-mcq").unwrap().1; + assert!(mcq.contains("Correct answer")); + assert!(mcq.contains("A — Enthalpy, $\\Delta H$")); + assert!(mcq.contains( + "

The change in enthalpy, $\\Delta H$.

" + )); + assert!(mcq.contains("B")); + assert!(mcq.contains("Uses the constant-volume result.")); + assert!(mcq.contains("That holds only at constant volume.")); + assert!(mcq.contains("

Source: KKW §6.3

")); + } + + #[test] + fn an_open_fragment_has_a_rubric_table_with_a_total() { + let cat = catalog("open"); + let rec = record(); + let frags = solution_fragments(&cat, &rec, &form()).unwrap(); + let open = &frags.iter().find(|(id, _)| id == "q-open").unwrap().1; + assert!(open.contains("Rubric — 2 points")); + assert!(open.contains("1States $H = U + PV$.")); + assert!(open.contains("Also accepted")); + } + + #[test] + fn an_open_fragment_shows_the_model_answer_and_the_explanation() { + // Both fields render, in that order, so an open-response solution reads as + // the answer followed by the reasoning, the same as a choice fragment. A + // multi-paragraph explanation keeps its breaks as separate paragraphs. + let cat = catalog("open-explain"); + let rec = record(); + let frags = solution_fragments(&cat, &rec, &form()).unwrap(); + let open = &frags.iter().find(|(id, _)| id == "q-open").unwrap().1; + assert!( + open.contains("
"), + "model answer shown:\n{open}" + ); + assert_eq!( + open.matches("

").count(), + 2, + "each explanation paragraph is its own styled

:\n{open}" + ); + assert!( + open.contains("folded into H"), + "first paragraph present:\n{open}" + ); + assert!( + open.contains("reads the enthalpy change directly"), + "second paragraph present:\n{open}" + ); + let model_at = open.find("sol-model").unwrap(); + let explain_at = open.find("sol-explain").unwrap(); + assert!( + model_at < explain_at, + "model answer comes before the explanation" + ); + } + + #[test] + fn inline_html_keeps_math_verbatim_and_escapes_prose() { + // A subscript inside math survives; angle brackets outside math are escaped. + assert_eq!(inline_html("value $q_p$ < 5"), "value $q_p$ < 5"); + // Bold outside math becomes a tag; a dollar-math run is passed through. + assert_eq!( + inline_html("**H** is $H = U + PV$"), + "H is $H = U + PV$" + ); + } + + #[test] + fn the_bundle_round_trips_through_the_kdf_and_cipher() { + use aes_gcm::aead::generic_array::GenericArray; + let fragments = vec![ + ("q-mcq".to_string(), "

alpha

".to_string()), + ("q-open".to_string(), "

beta

".to_string()), + ]; + let bundle = build_bundle("a1.1", "enthalpy2026", &fragments).unwrap(); + assert_eq!(bundle.v, 1); + assert_eq!(bundle.page, "a1.1"); + assert_eq!(bundle.cipher, "AES-GCM"); + assert_eq!(bundle.kdf.iterations, ITERATIONS); + // Insertion order is preserved in the serialized object. + let json = serde_json::to_string(&bundle).unwrap(); + assert!(json.find("q-mcq").unwrap() < json.find("q-open").unwrap()); + + // Decrypt the first item the way the browser would and check the plaintext. + let salt = B64.decode(&bundle.kdf.salt).unwrap(); + let key = derive_key("enthalpy2026", &salt); + let cipher = Aes256Gcm::new_from_slice(&key).unwrap(); + let (_, enc) = &bundle.items.0[0]; + let iv = B64.decode(&enc.iv).unwrap(); + let ct = B64.decode(&enc.ct).unwrap(); + let pt = cipher + .decrypt(GenericArray::from_slice(&iv), ct.as_ref()) + .unwrap(); + assert_eq!(String::from_utf8(pt).unwrap(), "

alpha

"); + } + + #[test] + fn a_generated_password_has_the_expected_shape() { + let pw = gen_password().unwrap(); + assert_eq!(pw.len(), 19); // 16 symbols + 3 dashes + assert_eq!(pw.matches('-').count(), 3); + assert!( + pw.chars() + .all(|c| c == '-' || PW_ALPHABET.contains(&(c as u8))) + ); + } + + #[test] + fn the_two_browser_assets_are_bundled() { + let assets = assets(); + assert_eq!(assets[0].0, "questions.css"); + assert_eq!(assets[1].0, "solutions.js"); + assert!(assets[0].1.contains(".qsol")); + assert!(assets[1].1.contains("AES-GCM")); + } +} diff --git a/src/export/typst.rs b/src/export/typst.rs index bb7fd28..88e61c5 100644 --- a/src/export/typst.rs +++ b/src/export/typst.rs @@ -50,6 +50,7 @@ //! the template by someone who has not read this comment. pub mod config; +pub mod diagnostic; pub mod payload; pub mod template; pub mod value; @@ -68,6 +69,7 @@ use crate::Layout; use crate::assessment::{AssessmentFile, Form}; use crate::catalog::Catalog; use crate::error::{Error, Result}; +use crate::markup; /// What to render. #[derive(Debug, Clone)] @@ -194,6 +196,17 @@ pub fn render(catalog: &Catalog, record: &AssessmentFile, opts: &Options) -> Res } let mut warnings = payload::check(&payload, &opts.config); + for (slot, body) in &bodies { + if markup::needs_chem_import(&template.source, body) { + warnings.push(format!( + "the `{}` slot carries a chemical formula, but the template {} does not import \ + whalogen, so Typst will stop at `unknown variable: ce`; add `{}`", + slot.as_str(), + template.origin, + markup::CHEM_IMPORT + )); + } + } if template.is_inert() { warnings.push(format!( "the template {} declares no coursebank markers, so no questions were injected; add \ @@ -418,35 +431,45 @@ mod tests { number: 1, item: "b::q-1".into(), version: None, + stem_digest: None, + distractors: Vec::new(), + variant: None, fingerprint: None, points: None, bonus: false, key: vec!["A".into()], level: None, - learning_objectives: Vec::new(), + learning_targets: Vec::new(), credit_overrides: Default::default(), dropped: true, + dropped_as: None, + dropped_before_printing: false, }, Placement { number: 2, item: "b::q-2".into(), version: None, + stem_digest: None, + distractors: Vec::new(), + variant: None, fingerprint: None, points: None, bonus: false, key: vec!["B".into()], level: None, - learning_objectives: Vec::new(), + learning_targets: Vec::new(), credit_overrides: Default::default(), dropped: false, + dropped_as: None, + dropped_before_printing: false, }, ], }; let printable: Vec = select::layout(&record, &Options::default().form) .into_iter() - .filter(|p| !p.dropped) + .filter(|p| p.was_printed()) .map(|p| p.number) .collect(); - assert_eq!(printable, vec![2]); + assert_eq!(printable, vec![1, 2]); } } diff --git a/src/export/typst/config.rs b/src/export/typst/config.rs index 2c8675c..01406d8 100644 --- a/src/export/typst/config.rs +++ b/src/export/typst/config.rs @@ -43,11 +43,25 @@ pub enum Variant { Key, /// A bubble sheet matching the form. AnswerSheet, + /// One student's diagnostic, which carries no questions. + StudentReport, + /// The instructor's class diagnostic. + CohortReport, } impl Variant { /// Every variant, in the order `export` writes them. - pub const ALL: [Variant; 3] = [Variant::Exam, Variant::Key, Variant::AnswerSheet]; + pub const ALL: [Variant; 5] = [ + Variant::Exam, + Variant::Key, + Variant::AnswerSheet, + Variant::StudentReport, + Variant::CohortReport, + ]; + + /// The variants `export typst` produces. A report is not built from an + /// assessment record alone, so `export typst` must not default to it. + pub const EXAM: [Variant; 3] = [Variant::Exam, Variant::Key, Variant::AnswerSheet]; /// The token used on the command line, in config keys, and in file names. pub fn as_str(self) -> &'static str { @@ -55,6 +69,8 @@ impl Variant { Variant::Exam => "exam", Variant::Key => "key", Variant::AnswerSheet => "answer-sheet", + Variant::StudentReport => "student-report", + Variant::CohortReport => "cohort-report", } } @@ -103,6 +119,8 @@ impl Variant { Variant::Exam => "", Variant::Key => "-key", Variant::AnswerSheet => "-answer-sheet", + Variant::StudentReport => "-student", + Variant::CohortReport => "-cohort", } } } @@ -310,6 +328,58 @@ impl Default for Fields { } } +/// Which sections of a student diagnostic are emitted. +/// +/// Everything defaults on. Each flag corresponds to one block that +/// [`crate::typst::diagnostic::student_value`] would otherwise write +/// unconditionally: turning one off drops it from `cb-data` as an empty array +/// rather than omitting the key, so a template that checks `len() > 0` (the +/// pattern the bundled templates use) simply renders nothing for that section +/// without needing to guard against a missing key. +/// +/// This governs whole sections. Which fields survive *within* a question — +/// feedback, hints, misconceptions, worked solutions — is decided when the +/// diagnostic itself is built, not here. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct StudentSections { + /// The per-level ("kinds of thinking") comparison to the class. + #[serde(default = "yes")] + pub levels: bool, + /// The per-objective mastery table. + #[serde(default = "yes")] + pub objectives: bool, + /// Objectives called out as strengths. + #[serde(default = "yes")] + pub strengths: bool, + /// Objectives called out as focus areas. + #[serde(default = "yes")] + pub focus: bool, + /// Which questions were dropped from scoring. + #[serde(default = "yes")] + pub dropped_questions: bool, + /// Lectures to revisit for missed objectives. + #[serde(default = "yes")] + pub review_lectures: bool, + /// Suggested study groups and their readings. + #[serde(default = "yes")] + pub study: bool, +} + +impl Default for StudentSections { + fn default() -> StudentSections { + StudentSections { + levels: true, + objectives: true, + strengths: true, + focus: true, + dropped_questions: true, + review_lectures: true, + study: true, + } + } +} + /// The resolved configuration for rendering one variant. #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] #[serde(deny_unknown_fields)] @@ -360,6 +430,11 @@ pub struct RenderConfig { #[serde(default = "yes")] pub number_from_record: bool, + /// Which sections of a student diagnostic are emitted. Ignored by every + /// variant except [`Variant::StudentReport`]. + #[serde(default)] + pub student_sections: StudentSections, + /// Anything else you want the template to see, carried through untouched. /// /// This is the escape hatch that keeps the crate out of your layout @@ -392,6 +467,7 @@ impl RenderConfig { stimulus: StimulusMode::Inline, fields: Fields::default(), number_from_record: true, + student_sections: StudentSections::default(), extra: BTreeMap::new(), }; match variant { @@ -422,6 +498,30 @@ impl RenderConfig { calibration: false, }; } + Variant::StudentReport => { + // Belt and braces. The student payload is built by + // `typst::diagnostic`, which has no question text to reveal in the + // first place; this says so in the one place someone would look. + config.reveal = Reveal::Nothing; + config.stimulus = StimulusMode::Omit; + config.fields = Fields { + uid: false, + title: false, + points: true, + level: true, + objectives: true, + topics: false, + assets: false, + source_letters: false, + design: false, + calibration: false, + }; + } + Variant::CohortReport => { + config.reveal = Reveal::Everything; + config.stimulus = StimulusMode::Omit; + config.fields.calibration = true; + } } config } @@ -518,6 +618,9 @@ pub struct Overrides { /// See [`RenderConfig::fields`]. #[serde(default, skip_serializing_if = "Option::is_none")] pub fields: Option, + /// See [`RenderConfig::student_sections`]. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub student_sections: Option, /// See [`RenderConfig::number_from_record`]. #[serde(default, skip_serializing_if = "Option::is_none")] pub number_from_record: Option, @@ -561,6 +664,9 @@ impl Overrides { if let Some(v) = &self.fields { config.fields = v.clone(); } + if let Some(v) = &self.student_sections { + config.student_sections = v.clone(); + } if let Some(v) = self.number_from_record { config.number_from_record = v; } @@ -607,6 +713,8 @@ defaults: extra: accent: '#017ab9' font: 'Libertinus Serif' + show-bubbles: true + bubble-radius: '0.42em' variants: exam: @@ -624,6 +732,20 @@ variants: answer-sheet: reveal: nothing + + # Every student-report section defaults on. Uncomment what you don't want; + # the CLI's `--no-levels`, `--no-objectives`, `--no-strengths`, `--no-focus`, + # `--no-dropped-questions`, `--no-review-lectures`, and `--no-study` flags + # set these same fields for a single run without editing this file. + # student-report: + # student_sections: + # levels: false + # objectives: false + # strengths: false + # focus: false + # dropped_questions: false + # review_lectures: false + # study: false "#; /// Serde default: `true`. diff --git a/src/export/typst/diagnostic.rs b/src/export/typst/diagnostic.rs new file mode 100644 index 0000000..ea05931 --- /dev/null +++ b/src/export/typst/diagnostic.rs @@ -0,0 +1,1347 @@ +// SPDX-License-Identifier: Prosperity-3.0.0 +// Copyright Scientific Computing Studio +// Source: https://git.scient.ing/education/coursebank + +//! Handing a diagnostic to a Typst template. +//! +//! The same arrangement the exam export uses: this module decides what the +//! template is told, the template decides what it looks like, and the two meet at +//! a marker comment. Nothing here knows about page size or colour. +//! +//! Two slots are filled, both already defined for the exam path: +//! +//! | Slot | Injected | +//! |:--|:--| +//! | `meta` | `#let cb-meta = (...)` — course, assessment, generator, class size | +//! | `data` | `#let cb-data = (...)` — one student's diagnostic, or the class's | +//! +//! The `questions` slot is deliberately unused. It exists to emit question stems, +//! and a diagnostic has none to emit: see +//! [`crate::diagnostic::StudentDiagnostic`], which has no field for one. A student +//! report template that wanted to print a stem would have nothing to print it +//! from, which is the property worth preserving. +//! +//! Prose fields — objective text, the feedback written for a chosen option, a +//! reading's focus sentence — go through the same markup path as an exam stem, so +//! `$\Delta G$` in `course.yaml` renders the same way in a report as it does on +//! the paper. + +use crate::assessment::AssessmentFile; +use crate::catalog::Catalog; +use crate::diagnostic::{ + Bin, CohortDiagnostic, CohortObjectiveRow, CohortQuestionRow, StudentDiagnostic, StudyGroup, +}; +use crate::error::{Error, Result}; +use crate::layout::Layout; +use crate::typst::config::{RenderConfig, Variant}; +use crate::typst::payload::markup_value; +use crate::typst::template::{self, Origin, Slot}; +use crate::typst::value::Value; + +/// A rendered diagnostic document. +#[derive(Debug, Clone)] +pub struct Document { + /// Which document this is. + pub variant: Variant, + /// Where the template came from. + pub origin: Origin, + /// Which slots it declared. + pub slots: Vec, + /// The Typst source. + pub text: String, + /// Advisory problems. + pub warnings: Vec, +} + +/// The identity block every diagnostic carries. +#[derive(Debug, Clone)] +pub struct Meta { + /// Course code. + pub course_code: String, + /// Course title. + pub course_title: String, + /// The term. + pub term: String, + /// Institution, when the course names one. + pub institution: Option, + /// Assessment id. + pub assessment_id: String, + /// Assessment title. + pub assessment_title: String, + /// The administration date, as `YYYY-MM-DD`. + pub date: Option, + /// The day the report was generated. + pub generated_on: String, + /// The tool version, so a report found later can be traced. + pub version: String, + /// How many students sat the assessment. + pub n_students: usize, + /// The course's mastery threshold, so a template can draw the line in the + /// same place the classification used. + pub mastery_threshold: f64, + /// How many items an objective needs before it is classified at all. + pub min_items_for_mastery: usize, +} + +impl Meta { + /// Builds the identity block. + /// + /// # Arguments + /// + /// * `catalog` - the loaded course. + /// * `record` - the assessment record. + /// * `n_students` - the cohort size. + /// + /// # Returns + /// + /// The block. + pub fn new(catalog: &Catalog, record: &AssessmentFile, n_students: usize) -> Meta { + Meta { + course_code: catalog.course.course.code.clone(), + course_title: catalog.course.course.title.clone(), + term: record + .assessment + .term + .clone() + .unwrap_or_else(|| catalog.course.course.term.clone()), + institution: catalog.course.course.institution.clone(), + assessment_id: record.assessment.id.clone(), + assessment_title: record.assessment.title.clone(), + date: record.assessment.date.map(|d| d.to_string()), + generated_on: crate::date::Date::today().to_string(), + version: crate::VERSION.to_string(), + n_students, + mastery_threshold: catalog.course.policy.mastery_threshold, + min_items_for_mastery: catalog.course.policy.min_items_for_mastery, + } + } + + /// The metadata as a Typst value. + fn value(&self, config: &RenderConfig) -> Value { + let mut course = Value::dict(); + course.insert("code", Value::str(&self.course_code)); + course.insert("title", Value::str(&self.course_title)); + course.insert("term", Value::str(&self.term)); + if let Some(institution) = &self.institution { + course.insert("institution", Value::str(institution)); + } + + let mut assessment = Value::dict(); + assessment.insert("id", Value::str(&self.assessment_id)); + assessment.insert("title", Value::str(&self.assessment_title)); + assessment.insert_some("date", self.date.as_ref().map(Value::str)); + + let mut generator = Value::dict(); + generator.insert("tool", Value::str("coursebank")); + generator.insert("version", Value::str(&self.version)); + generator.insert("on", Value::str(&self.generated_on)); + + let mut policy = Value::dict(); + policy.insert("mastery-threshold", Value::Float(self.mastery_threshold)); + policy.insert( + "min-items-for-mastery", + Value::Int(self.min_items_for_mastery as i64), + ); + + let mut out = Value::dict(); + out.insert("course", course); + out.insert("assessment", assessment); + out.insert("generator", generator); + out.insert("policy", policy); + // `students-tested` is the honest name: it is how many people sat this + // assessment, which is not the enrolment. `class-size` stays as an alias + // so a template forked before this change keeps working. + out.insert("students-tested", Value::Int(self.n_students as i64)); + out.insert("class-size", Value::Int(self.n_students as i64)); + out.insert("extra", extra_value(config)); + out + } +} + +/// The render config's `extra` block, carried through untouched. +fn extra_value(config: &RenderConfig) -> Value { + let mut out = Value::dict(); + for (key, value) in &config.extra { + out.insert(key.clone(), crate::typst::value::from_yaml(value, false)); + } + out +} + +/// One student's diagnostic as a Typst value. +/// +/// `config.student_sections` decides which of `levels`, `objectives`, +/// `strengths`, `focus`, `dropped-questions`, `review-lectures`, and `study` +/// are populated; a section turned off is emitted as an empty array rather +/// than left out of the dictionary, so a template need not guard against a +/// missing key. +/// +/// # Arguments +/// +/// * `diagnostic` - the assembled diagnostic. +/// * `config` - the render config, for markup handling and section toggles. +/// +/// # Returns +/// +/// A dictionary the template binds as `cb-data`. +pub fn student_value(diagnostic: &StudentDiagnostic, config: &RenderConfig) -> Value { + let content = config.content.is_content(); + let mut out = Value::dict(); + + out.insert("student-key", Value::str(&diagnostic.student_key)); + out.insert_some("name", diagnostic.name.as_ref().map(Value::str)); + out.insert_some("sid", diagnostic.sid.as_ref().map(Value::str)); + out.insert_some("email", diagnostic.email.as_ref().map(Value::str)); + out.insert_some("form", diagnostic.form.as_ref().map(Value::str)); + + let mut score = Value::dict(); + score.insert("points", Value::Float(diagnostic.score.points)); + score.insert("possible", Value::Float(diagnostic.score.points_possible)); + score.insert("percent", Value::Float(diagnostic.score.percent)); + score.insert("bonus", Value::Float(diagnostic.score.bonus_points)); + score.insert("correct", Value::Int(diagnostic.score.correct as i64)); + score.insert("items", Value::Int(diagnostic.score.n_items as i64)); + out.insert("score", score); + + if let Some(standing) = &diagnostic.standing { + let mut value = Value::dict(); + value.insert("class-mean", Value::Float(standing.class_mean)); + value.insert("class-sd", Value::Float(standing.class_sd)); + value.insert("band", Value::str(&standing.band)); + value.insert_some("theta", standing.theta.map(Value::Float)); + value.insert_some("theta-se", standing.theta_se.map(Value::Float)); + out.insert("standing", value); + } + + let levels = if config.student_sections.levels { + diagnostic.levels.as_slice() + } else { + &[] + }; + out.insert( + "levels", + Value::Array( + levels + .iter() + .map(|level| { + let mut value = Value::dict(); + value.insert("level", Value::Int(level.level as i64)); + value.insert("name", Value::str(&level.name)); + value.insert("blurb", Value::str(&level.blurb)); + value.insert("items", Value::Int(level.n_items as i64)); + value.insert("rate", Value::Float(level.rate)); + value.insert_some("class-rate", level.class_rate.map(Value::Float)); + value.insert_some("comparison", level.comparison.as_ref().map(Value::str)); + value + }) + .collect(), + ), + ); + + let objectives = if config.student_sections.objectives { + diagnostic.objectives.as_slice() + } else { + &[] + }; + out.insert( + "objectives", + Value::Array( + objectives + .iter() + .map(|objective| { + let mut value = Value::dict(); + value.insert("id", Value::str(&objective.id)); + value.insert("text", markup_value(&objective.text, content)); + value.insert_some("unit", objective.unit.as_ref().map(Value::str)); + value.insert("items", Value::Int(objective.n_items as i64)); + value.insert("credit", Value::Float(objective.credit)); + value.insert("rate", Value::Float(objective.rate)); + value.insert("lower", Value::Float(objective.lower)); + value.insert("upper", Value::Float(objective.upper)); + value.insert_some("class-rate", objective.class_rate.map(Value::Float)); + value.insert("status", Value::str(&objective.status)); + value.insert("symbol", Value::str(&objective.symbol)); + value.insert("confident", Value::Bool(objective.confident)); + value.insert("thin-evidence", Value::Bool(objective.thin_evidence)); + value.insert( + "levels", + Value::Array( + objective + .levels + .iter() + .map(|l| Value::Int(*l as i64)) + .collect(), + ), + ); + value + }) + .collect(), + ), + ); + + let objective_refs = |rows: &[crate::diagnostic::ObjectiveRef]| -> Value { + Value::Array( + rows.iter() + .map(|row| { + let mut value = Value::dict(); + value.insert("id", Value::str(&row.id)); + value.insert("text", markup_value(&row.text, content)); + value.insert("rate", Value::Float(row.rate)); + value.insert("items", Value::Int(row.n_items as i64)); + value + }) + .collect(), + ) + }; + out.insert( + "strengths", + objective_refs(if config.student_sections.strengths { + &diagnostic.strengths + } else { + &[] + }), + ); + out.insert( + "focus", + objective_refs(if config.student_sections.focus { + &diagnostic.focus + } else { + &[] + }), + ); + + out.insert( + "questions", + Value::Array( + diagnostic + .questions + .iter() + .map(|question| { + let mut value = Value::dict(); + value.insert("number", Value::Int(question.number as i64)); + value.insert_some("position", question.position.map(|p| Value::Int(p as i64))); + value.insert_some("level", question.level.map(|l| Value::Int(l as i64))); + value.insert( + "targets", + Value::Array(question.targets.iter().map(Value::str).collect()), + ); + value.insert_some("correct", question.correct.map(Value::Bool)); + value.insert("credit", Value::Float(question.credit)); + value.insert("bonus", Value::Bool(question.bonus)); + value.insert("dropped", Value::Bool(question.dropped)); + value.insert("blank", Value::Bool(question.blank)); + value.insert_some("class-rate", question.class_rate.map(Value::Float)); + // Both tiers, as a list of pairs: the target says what this + // question asked, the objective says which row of the table + // above it counted toward. `objective` is absent when the + // tagged id is an objective with no targets, so the template + // does not print one sentence twice. + value.insert( + "measured", + Value::Array( + question + .measured + .iter() + .map(|m| { + let mut pair = Value::dict(); + pair.insert_some( + "objective", + m.objective + .as_ref() + .map(|text| markup_value(text, content)), + ); + pair.insert("target", markup_value(&m.target, content)); + pair + }) + .collect(), + ), + ); + for (key, text) in [ + ("feedback", question.feedback.as_ref()), + ("hint", question.hint.as_ref()), + ("misconception", question.misconception.as_ref()), + ("worked", question.worked.as_ref()), + ] { + value.insert_some(key, text.map(|t| markup_value(t, content))); + } + value.insert( + "taught-in", + Value::Array(question.taught_in.iter().map(Value::str).collect()), + ); + value.insert( + "review", + Value::Array( + question + .review + .iter() + .map(|reading| { + let mut entry = Value::dict(); + entry.insert("citation", Value::str(&reading.citation)); + entry.insert_some( + "title", + reading.title.as_ref().map(Value::str), + ); + entry.insert_some("url", reading.url.as_ref().map(Value::str)); + entry + }) + .collect(), + ), + ); + value + }) + .collect(), + ), + ); + + let dropped_questions = if config.student_sections.dropped_questions { + diagnostic.dropped_questions.as_slice() + } else { + &[] + }; + out.insert( + "dropped-questions", + Value::Array( + dropped_questions + .iter() + .map(|dropped| { + let mut value = Value::dict(); + value.insert("number", Value::Int(dropped.number as i64)); + value.insert("full-credit", Value::Bool(dropped.full_credit)); + value + }) + .collect(), + ), + ); + + let review_lectures = if config.student_sections.review_lectures { + diagnostic.review_lectures.as_slice() + } else { + &[] + }; + out.insert( + "review-lectures", + Value::Array( + review_lectures + .iter() + .map(|lecture| { + let mut value = Value::dict(); + value.insert("lecture", Value::str(&lecture.lecture)); + value.insert("title", Value::str(&lecture.title)); + value.insert_some("url", lecture.url.as_ref().map(Value::str)); + value.insert("targets-missed", Value::Int(lecture.n_targets as i64)); + value.insert("questions-missed", Value::Int(lecture.n_questions as i64)); + value.insert( + "questions", + Value::Array( + lecture + .questions + .iter() + .map(|n| Value::Int(*n as i64)) + .collect(), + ), + ); + value.insert( + "slides", + Value::Array( + lecture + .slides + .iter() + .map(|n| Value::Int(*n as i64)) + .collect(), + ), + ); + value.insert( + "targets", + Value::Array( + lecture + .targets + .iter() + .map(|text| markup_value(text, content)) + .collect(), + ), + ); + value + }) + .collect(), + ), + ); + + let study = if config.student_sections.study { + diagnostic.study.as_slice() + } else { + &[] + }; + out.insert( + "study", + Value::Array( + study + .iter() + .map(|group| study_value(group, content)) + .collect(), + ), + ); + + out +} + +/// One study group as a Typst value. +fn study_value(group: &StudyGroup, content: bool) -> Value { + let mut out = Value::dict(); + out.insert("objective", Value::str(&group.objective)); + out.insert("text", markup_value(&group.text, content)); + out.insert("rate", Value::Float(group.rate)); + out.insert( + "readings", + Value::Array( + group + .readings + .iter() + .map(|reading| { + let mut value = Value::dict(); + value.insert("citation", Value::str(&reading.citation)); + value.insert("lecture", Value::str(&reading.lecture)); + value.insert("lecture-title", Value::str(&reading.lecture_title)); + value.insert_some("url", reading.url.as_ref().map(Value::str)); + value.insert_some( + "focus", + reading.focus.as_ref().map(|t| markup_value(t, content)), + ); + value.insert_some( + "summary", + reading.summary.as_ref().map(|t| markup_value(t, content)), + ); + value.insert("supplemental", Value::Bool(reading.supplemental)); + value + }) + .collect(), + ), + ); + out +} + +/// The class diagnostic as a Typst value. +/// +/// # Arguments +/// +/// * `diagnostic` - the assembled diagnostic. +/// * `config` - the render config, for markup handling. +/// +/// # Returns +/// +/// A dictionary the template binds as `cb-data`. +pub fn cohort_value(diagnostic: &CohortDiagnostic, config: &RenderConfig) -> Value { + let content = config.content.is_content(); + let mut out = Value::dict(); + + out.insert("students", Value::Int(diagnostic.n_students as i64)); + out.insert("items", Value::Int(diagnostic.n_items as i64)); + + let mut distribution = Value::dict(); + distribution.insert("mean", Value::Float(diagnostic.distribution.mean)); + distribution.insert("median", Value::Float(diagnostic.distribution.median)); + distribution.insert("sd", Value::Float(diagnostic.distribution.sd)); + distribution.insert("min", Value::Float(diagnostic.distribution.min)); + distribution.insert("max", Value::Float(diagnostic.distribution.max)); + distribution.insert( + "bins", + Value::Array(diagnostic.distribution.bins.iter().map(bin_value).collect()), + ); + out.insert("distribution", distribution); + + let mut reliability = Value::dict(); + reliability.insert_some("alpha", diagnostic.reliability.alpha.map(Value::Float)); + reliability.insert_some("sem", diagnostic.reliability.sem.map(Value::Float)); + reliability.insert("mean-p", Value::Float(diagnostic.reliability.mean_p)); + reliability.insert_some( + "mean-point-biserial", + diagnostic.reliability.mean_point_biserial.map(Value::Float), + ); + reliability.insert( + "interpretation", + Value::str(&diagnostic.reliability.interpretation), + ); + out.insert("reliability", reliability); + + out.insert( + "levels", + Value::Array( + diagnostic + .levels + .iter() + .map(|level| { + let mut value = Value::dict(); + value.insert("level", Value::Int(level.level as i64)); + value.insert("name", Value::str(&level.name)); + value.insert("items", Value::Int(level.n_items as i64)); + value.insert("rate", Value::Float(level.rate)); + value + }) + .collect(), + ), + ); + + out.insert( + "objectives", + Value::Array( + diagnostic + .objectives + .iter() + .map(|o| cohort_objective_value(o, content)) + .collect(), + ), + ); + out.insert( + "gaps", + Value::Array( + diagnostic + .gaps + .iter() + .map(|o| cohort_objective_value(o, content)) + .collect(), + ), + ); + + out.insert( + "questions", + Value::Array( + diagnostic + .questions + .iter() + .map(|q| cohort_question_value(q, content)) + .collect(), + ), + ); + out.insert( + "revise", + Value::Array( + diagnostic + .revise + .iter() + .map(|q| cohort_question_value(q, content)) + .collect(), + ), + ); + // Separate from `questions` so no statistic can pick them up, and merged + // back in by the evidence section, which describes rather than measures. + out.insert( + "dropped-detail", + Value::Array( + diagnostic + .dropped_detail + .iter() + .map(|q| cohort_question_value(q, content)) + .collect(), + ), + ); + + out.insert( + "grades", + Value::Array( + diagnostic + .grades + .iter() + .map(|grade| { + let mut value = Value::dict(); + value.insert("letter", Value::str(&grade.letter)); + value.insert("low", Value::Float(grade.low)); + value.insert("high", Value::Float(grade.high)); + value.insert_some("gpa", grade.gpa.map(Value::Float)); + value.insert_some("attainment", grade.attainment.as_ref().map(Value::str)); + value.insert("group", Value::str(&grade.group)); + value.insert("count", Value::Int(grade.count as i64)); + value.insert("share", Value::Float(grade.share)); + value.insert("at-or-above", Value::Int(grade.at_or_above as i64)); + value + }) + .collect(), + ), + ); + + out.insert( + "lectures", + Value::Array( + diagnostic + .lectures + .iter() + .map(|lecture| { + let mut value = Value::dict(); + value.insert("lecture", Value::str(&lecture.lecture)); + value.insert("title", Value::str(&lecture.title)); + value.insert("items", Value::Int(lecture.n_items as i64)); + value.insert("objectives", Value::Int(lecture.n_objectives as i64)); + value.insert( + "objectives-below", + Value::Int(lecture.n_objectives_below as i64), + ); + value.insert("rate", Value::Float(lecture.rate)); + value.insert( + "questions", + Value::Array( + lecture + .questions + .iter() + .map(|n| Value::Int(*n as i64)) + .collect(), + ), + ); + value.insert_some( + "worst-objective", + lecture + .worst_objective + .as_ref() + .map(|text| markup_value(text, content)), + ); + value + }) + .collect(), + ), + ); + + out.insert( + "dropped-questions", + Value::Array( + diagnostic + .dropped_questions + .iter() + .map(|dropped| { + let mut value = Value::dict(); + value.insert("number", Value::Int(dropped.number as i64)); + value.insert("full-credit", Value::Bool(dropped.full_credit)); + value + }) + .collect(), + ), + ); + + let triage_rows = |rows: &[crate::diagnostic::TriageRow]| -> Value { + Value::Array(rows.iter().map(|row| triage_value(row, content)).collect()) + }; + let mut triage = Value::dict(); + triage.insert("discard", triage_rows(&diagnostic.triage.discard)); + triage.insert("rekey", triage_rows(&diagnostic.triage.rekey)); + triage.insert("revise", triage_rows(&diagnostic.triage.revise)); + triage.insert("reteach", triage_rows(&diagnostic.triage.reteach)); + triage.insert("bounded", triage_rows(&diagnostic.triage.bounded)); + triage.insert("clean", Value::Int(diagnostic.triage.clean as i64)); + out.insert("triage", triage); + + let predictions = &diagnostic.predictions; + let mut prediction = Value::dict(); + prediction.insert("predicted", Value::Int(predictions.n_predicted as i64)); + prediction.insert("calibrated", Value::Int(predictions.n_calibrated as i64)); + prediction.insert_some( + "mean-signed-error", + predictions.mean_signed_error.map(Value::Float), + ); + prediction.insert_some( + "mean-abs-error", + predictions.mean_abs_error.map(Value::Float), + ); + prediction.insert("within", Value::Int(predictions.n_within as i64)); + prediction.insert("band", Value::Int(predictions.n_band as i64)); + prediction.insert("band-hit", Value::Int(predictions.n_band_hit as i64)); + if let Some((number, expected, observed)) = predictions.biggest_surprise { + let mut surprise = Value::dict(); + surprise.insert("number", Value::Int(number as i64)); + surprise.insert("expected", Value::Float(expected)); + surprise.insert("observed", Value::Float(observed)); + prediction.insert("biggest-surprise", surprise); + } + out.insert("predictions", prediction); + + out.insert( + "forms", + Value::Array( + diagnostic + .forms + .iter() + .map(|form| { + let mut value = Value::dict(); + value.insert("id", Value::str(&form.id)); + value.insert("students", Value::Int(form.n_students as i64)); + value.insert("mean", Value::Float(form.mean)); + value.insert("sd", Value::Float(form.sd)); + value + }) + .collect(), + ), + ); + + out.insert( + "blueprint", + Value::Array(diagnostic.blueprint.iter().map(Value::str).collect()), + ); + + out.insert( + "patterns", + Value::Array( + diagnostic + .patterns + .iter() + .map(|pattern| { + let mut value = Value::dict(); + value.insert("label", Value::str(&pattern.label)); + value.insert("students", Value::Int(pattern.n_students as i64)); + let mut means = Value::dict(); + for (level, mean) in &pattern.level_means { + means.insert(format!("l{level}"), Value::Float(*mean)); + } + value.insert("level-means", means); + value + }) + .collect(), + ), + ); + + out.insert( + "warnings", + Value::Array(diagnostic.warnings.iter().map(Value::str).collect()), + ); + + out +} + +/// One triage row as a Typst value. +fn triage_value(row: &crate::diagnostic::TriageRow, content: bool) -> Value { + let mut value = Value::dict(); + value.insert("number", Value::Int(row.number as i64)); + value.insert_some("item", row.item.as_ref().map(Value::str)); + value.insert_some("level", row.level.map(|l| Value::Int(l as i64))); + value.insert("p", Value::Float(row.p_value)); + value.insert_some("point-biserial", row.point_biserial.map(Value::Float)); + value.insert_some("discrimination", row.discrimination.map(Value::Float)); + value.insert( + "targets", + Value::Array( + row.targets + .iter() + .map(|text| markup_value(text, content)) + .collect(), + ), + ); + value.insert( + "taught-in", + Value::Array(row.taught_in.iter().map(Value::str).collect()), + ); + value.insert_some("option", row.option.as_ref().map(Value::str)); + value.insert_some("option-share", row.option_share.map(Value::Float)); + value.insert_some( + "option-point-biserial", + row.option_point_biserial.map(Value::Float), + ); + value.insert( + "reasons", + Value::Array( + row.reasons + .iter() + .map(|reason| markup_value(reason, content)) + .collect(), + ), + ); + value +} + +/// One histogram bin as a Typst value. +fn bin_value(bin: &Bin) -> Value { + let mut value = Value::dict(); + value.insert("low", Value::Int(bin.low as i64)); + value.insert("high", Value::Int(bin.high as i64)); + value.insert("count", Value::Int(bin.count as i64)); + value +} + +/// One class objective row as a Typst value. +fn cohort_objective_value(objective: &CohortObjectiveRow, content: bool) -> Value { + let mut value = Value::dict(); + value.insert("id", Value::str(&objective.id)); + value.insert("text", markup_value(&objective.text, content)); + value.insert("items", Value::Int(objective.n_items as i64)); + value.insert("rate", Value::Float(objective.rate)); + value.insert("meeting", Value::Int(objective.meeting as i64)); + value.insert("developing", Value::Int(objective.developing as i64)); + value.insert("not-yet", Value::Int(objective.not_yet as i64)); + value.insert("thin", Value::Int(objective.thin as i64)); + value.insert("below-threshold", Value::Bool(objective.below_threshold)); + value +} + +/// One class question row as a Typst value. +fn cohort_question_value(question: &CohortQuestionRow, content: bool) -> Value { + let mut value = Value::dict(); + value.insert("number", Value::Int(question.number as i64)); + value.insert_some("item", question.item.as_ref().map(Value::str)); + value.insert_some("level", question.level.map(|l| Value::Int(l as i64))); + value.insert( + "targets", + Value::Array(question.targets.iter().map(Value::str).collect()), + ); + value.insert( + "target-texts", + Value::Array( + question + .target_texts + .iter() + .map(|text| markup_value(text, content)) + .collect(), + ), + ); + value.insert_some( + "stem", + question + .stem + .as_ref() + .map(|text| markup_value(text, content)), + ); + value.insert("dropped", Value::Bool(question.dropped)); + value.insert( + "dropped-full-credit", + Value::Bool(question.dropped_full_credit), + ); + value.insert( + "taught-in", + Value::Array(question.taught_in.iter().map(Value::str).collect()), + ); + value.insert( + "lectures", + Value::Array(question.lectures.iter().map(Value::str).collect()), + ); + value.insert("difficulty-band", Value::str(&question.difficulty_band)); + value.insert( + "discrimination-band", + Value::str(&question.discrimination_band), + ); + value.insert("p", Value::Float(question.p_value)); + value.insert_some("point-biserial", question.point_biserial.map(Value::Float)); + value.insert_some("discrimination", question.discrimination.map(Value::Float)); + value.insert("blank-rate", Value::Float(question.blank_rate)); + value.insert( + "key", + Value::Array(question.key.iter().map(Value::str).collect()), + ); + value.insert( + "options", + Value::Array( + question + .options + .iter() + .map(|option| { + let mut value = Value::dict(); + value.insert("letter", Value::str(&option.letter)); + value.insert_some( + "text", + option.text.as_ref().map(|text| markup_value(text, content)), + ); + // The letter on each paper, so a statistic reported against + // the bank letter can be checked against a student's copy. + value.insert( + "printed", + Value::Array( + option + .printed + .iter() + .map(|printed| { + let mut pair = Value::dict(); + pair.insert("form", Value::str(&printed.form)); + pair.insert("letter", Value::str(&printed.letter)); + pair + }) + .collect(), + ), + ); + value.insert("count", Value::Int(option.count as i64)); + value.insert("rate", Value::Float(option.rate)); + value.insert("is-key", Value::Bool(option.is_key)); + value.insert_some("point-biserial", option.point_biserial.map(Value::Float)); + value.insert("nonfunctioning", Value::Bool(option.nonfunctioning)); + value + }) + .collect(), + ), + ); + value.insert( + "flags", + Value::Array(question.flags.iter().map(Value::str).collect()), + ); + value.insert( + "notes", + Value::Array( + question + .notes + .iter() + .map(|n| markup_value(n, content)) + .collect(), + ), + ); + value.insert( + "prediction-notes", + Value::Array( + question + .prediction_notes + .iter() + .map(|n| markup_value(n, content)) + .collect(), + ), + ); + value.insert("calibrated", Value::Bool(question.calibrated)); + let mut by_form = Value::dict(); + for (form, p) in &question.by_form { + by_form.insert(form.clone(), Value::Float(*p)); + } + value.insert("by-form", by_form); + value +} + +/// Emits a `#let` binding for a slot. +fn binding(name: &str, value: &Value) -> String { + format!("#let {name} = {}\n", value.to_typst(0)) +} + +/// Renders one student's report. +/// +/// # Arguments +/// +/// * `layout` - the course layout, for the template lookup. +/// * `meta` - the identity block. +/// * `diagnostic` - the student's diagnostic. +/// * `config` - the render config for this variant. +/// * `explicit` - a template path overriding the lookup. +/// +/// # Returns +/// +/// The rendered document. +/// +/// # Errors +/// +/// Returns [`Error::Io`] when an explicit template cannot be read and +/// [`Error::Invalid`] when a template's markers are malformed. +pub fn render_student( + layout: &Layout, + meta: &Meta, + diagnostic: &StudentDiagnostic, + config: &RenderConfig, + explicit: Option<&std::path::Path>, +) -> Result { + render( + layout, + Variant::StudentReport, + &meta.assessment_id, + meta, + student_value(diagnostic, config), + config, + explicit, + ) +} + +/// Renders the class report. +/// +/// # Arguments +/// +/// * `layout` - the course layout, for the template lookup. +/// * `meta` - the identity block. +/// * `diagnostic` - the class diagnostic. +/// * `config` - the render config for this variant. +/// * `explicit` - a template path overriding the lookup. +/// +/// # Returns +/// +/// The rendered document. +/// +/// # Errors +/// +/// As [`render_student`]. +pub fn render_cohort( + layout: &Layout, + meta: &Meta, + diagnostic: &CohortDiagnostic, + config: &RenderConfig, + explicit: Option<&std::path::Path>, +) -> Result { + render( + layout, + Variant::CohortReport, + &meta.assessment_id, + meta, + cohort_value(diagnostic, config), + config, + explicit, + ) +} + +/// The shared rendering path. +#[allow(clippy::too_many_arguments)] +fn render( + layout: &Layout, + variant: Variant, + assessment_id: &str, + meta: &Meta, + data: Value, + config: &RenderConfig, + explicit: Option<&std::path::Path>, +) -> Result { + let template = template::load(layout, variant, Some(assessment_id), explicit)?; + + let mut bodies = Vec::new(); + if template.wants(Slot::Meta) { + bodies.push(( + Slot::Meta, + binding(&config.meta_binding, &meta.value(config)), + )); + } + if template.wants(Slot::Data) { + bodies.push((Slot::Data, binding(&config.data_binding, &data))); + } + + let mut warnings = Vec::new(); + for (slot, body) in &bodies { + if crate::markup::needs_chem_import(&template.source, body) { + warnings.push(format!( + "the `{}` slot carries a chemical formula, but the template {} does not import \ + whalogen, so Typst will stop at `unknown variable: ce`; add `{}`", + slot.as_str(), + template.origin, + crate::markup::CHEM_IMPORT + )); + } + } + if template.is_inert() { + warnings.push(format!( + "the template {} declares no coursebank markers, so the report is empty; add `// \ + coursebank:data` where the body belongs", + template.origin + )); + } else if !template.wants(Slot::Data) { + warnings.push(format!( + "the template {} declares no `data` slot, so it received the metadata but not the \ + report itself", + template.origin + )); + } + if template.wants(Slot::Questions) { + return Err(Error::Invalid(vec![format!( + "the template {} declares a `questions` slot, but a diagnostic report carries no \ + questions to fill it with. Remove the marker: a report that reproduces the exam \ + cannot be returned before a makeup is given", + template.origin + )])); + } + + Ok(Document { + variant, + origin: template.origin.clone(), + slots: template.slots(), + text: template.render(&bodies), + warnings, + }) +} + +/// The file stem a student's report is written under. +/// +/// Uses the student key rather than the name: a key is unique, filesystem-safe, +/// and already a pseudonym when the store is pseudonymized. +/// +/// # Arguments +/// +/// * `assessment_id` - the assessment id. +/// * `student_key` - the student key. +/// +/// # Returns +/// +/// The stem, with no extension. +pub fn student_stem(assessment_id: &str, student_key: &str) -> String { + let safe: String = student_key + .chars() + .map(|c| { + if c.is_ascii_alphanumeric() || c == '-' || c == '_' { + c + } else { + '-' + } + }) + .collect(); + format!("{assessment_id}-{safe}") +} + +/// Summarizes what a set of rendered reports covered, for the command line. +/// +/// # Arguments +/// +/// * `written` - how many files were written. +/// * `students` - how many students they cover. +/// +/// # Returns +/// +/// A sentence. +pub fn summary(written: usize, students: usize) -> String { + format!( + "{written} file(s) for {students} student(s); compile them with `typst compile` or the \ + loop in the guide" + ) +} + +/// Tallies what a class diagnostic would tell you to do next. +/// +/// Kept here rather than in the template so that the command line and the PDF +/// agree about what counts as a finding. +/// +/// # Arguments +/// +/// * `diagnostic` - the class diagnostic. +/// +/// # Returns +/// +/// Short lines, most important first. +pub fn headline(diagnostic: &CohortDiagnostic) -> Vec { + let mut out = Vec::new(); + out.push(format!( + "{} student(s), mean {:.0}% (SD {:.1}), median {:.0}%", + diagnostic.n_students, + diagnostic.distribution.mean, + diagnostic.distribution.sd, + diagnostic.distribution.median + )); + if !diagnostic.gaps.is_empty() { + out.push(format!( + "{} objective(s) the class did not meet; worst is {} at {:.0}%", + diagnostic.gaps.len(), + diagnostic.gaps[0].id, + diagnostic.gaps[0].rate * 100.0 + )); + } + if !diagnostic.revise.is_empty() { + let numbers: Vec = diagnostic + .revise + .iter() + .take(6) + .map(|q| format!("q{}", q.number)) + .collect(); + out.push(format!( + "{} question(s) to look at before reuse: {}", + diagnostic.revise.len(), + numbers.join(", ") + )); + } + if diagnostic.forms.len() > 1 { + let spread = diagnostic + .forms + .iter() + .map(|f| f.mean) + .fold(f64::NEG_INFINITY, f64::max) + - diagnostic + .forms + .iter() + .map(|f| f.mean) + .fold(f64::INFINITY, f64::min); + out.push(format!( + "{} forms, {:.0} points apart at the mean", + diagnostic.forms.len(), + spread + )); + } + out +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::diagnostic::{Distribution, ReliabilityRow}; + + fn empty_cohort() -> CohortDiagnostic { + CohortDiagnostic { + n_students: 24, + n_items: 36, + distribution: Distribution { + mean: 72.5, + median: 74.0, + sd: 12.0, + min: 41.0, + max: 97.0, + bins: vec![Bin { + low: 70, + high: 80, + count: 9, + }], + }, + reliability: ReliabilityRow { + alpha: Some(0.71), + sem: Some(2.1), + mean_p: 0.72, + mean_point_biserial: Some(0.24), + interpretation: "acceptable for a classroom exam".into(), + }, + levels: Vec::new(), + objectives: Vec::new(), + gaps: Vec::new(), + grades: Vec::new(), + lectures: Vec::new(), + dropped_questions: Vec::new(), + questions: Vec::new(), + triage: crate::diagnostic::Triage::default(), + predictions: crate::diagnostic::PredictionSummary::default(), + revise: Vec::new(), + forms: Vec::new(), + blueprint: Vec::new(), + patterns: Vec::new(), + warnings: Vec::new(), + dropped_detail: Vec::new(), + } + } + + #[test] + fn report_markup_takes_the_same_path_as_a_paper() { + for source in [ + "$\\ce{H2O <=> H+ + OH-}$", + "the backbone $\\ce{-C=O}$ group", + "$K_w = [\\text{H}^+][\\text{OH}^-]$", + "see @fig:x where x < y", + "costs \\$5, and just $5", + ] { + let expected = crate::markup::to_typst(source); + assert_eq!( + markup_value(source, true).to_typst(0), + format!("[{expected}]"), + "content mode diverged on {source:?}" + ); + assert_eq!( + markup_value(source, false).to_typst(0), + Value::str(expected).to_typst(0), + "string mode diverged on {source:?}" + ); + } + } + + #[test] + fn a_report_carrying_chemistry_names_a_template_missing_the_import() { + let body = "#let cb-data = (stem: [#ce(\"H2O\")])"; + assert!(crate::markup::needs_chem_import( + "#import \"@preview/mitex:0.2.7\": mi\n// coursebank:data\n", + body + )); + assert!(!crate::markup::needs_chem_import( + crate::markup::CHEM_IMPORT, + body + )); + } + + #[test] + fn the_headline_leads_with_the_distribution() { + let lines = headline(&empty_cohort()); + assert!(lines[0].contains("24 student(s)"), "{lines:?}"); + assert!( + lines[0].contains("73%") || lines[0].contains("72%"), + "{lines:?}" + ); + } + + #[test] + fn a_student_stem_is_filesystem_safe() { + assert_eq!(student_stem("e1", "s-9f8e7d"), "e1-s-9f8e7d"); + assert_eq!(student_stem("e1", "ada@x.edu"), "e1-ada-x-edu"); + } + + #[test] + fn the_cohort_value_carries_no_question_text() { + let config = RenderConfig::for_variant(Variant::CohortReport); + let text = cohort_value(&empty_cohort(), &config).to_typst(0); + assert!(text.contains("reliability")); + assert!(!text.contains("stem")); + } +} diff --git a/src/export/typst/payload.rs b/src/export/typst/payload.rs index dc966ab..12d572f 100644 --- a/src/export/typst/payload.rs +++ b/src/export/typst/payload.rs @@ -50,7 +50,7 @@ pub struct Payload { pub form: FormInfo, /// Counts and sums, so a template does not have to derive them. pub totals: Totals, - /// Learning objectives referenced by the printed items, by id. + /// Learning objectives and targets referenced by the printed items, by id. #[serde(skip_serializing_if = "BTreeMap::is_empty")] pub objectives: BTreeMap, /// Shared stimuli, by id. Populated when the render config says stimuli are @@ -268,9 +268,9 @@ pub struct Question { /// letter. Omitted unless the config reveals the key. #[serde(skip_serializing_if = "Option::is_none")] pub credit_overrides: Option>, - /// Learning objective ids. + /// Learning target ids. #[serde(skip_serializing_if = "Vec::is_empty")] - pub learning_objectives: Vec, + pub learning_targets: Vec, /// Topic tags. #[serde(skip_serializing_if = "Vec::is_empty")] pub topics: Vec, @@ -386,10 +386,10 @@ pub fn build( .or_insert(0) += 1; } - let objectives = if placement.learning_objectives.is_empty() { - item.learning_objectives.clone() + let objectives = if placement.learning_targets.is_empty() { + item.learning_targets.clone() } else { - placement.learning_objectives.clone() + placement.learning_targets.clone() }; if config.fields.objectives { objective_ids.extend(objectives.iter().cloned()); @@ -411,11 +411,12 @@ pub fn build( _ => (None, None), }; - let order = select::option_order(form, &placement.item, item.options.len()); + let shown = item.administered(&placement.key, &placement.distractors); + let order = select::option_order(form, &placement.item, shown.len()); let options: Vec = order .iter() .enumerate() - .map(|(position, source_index)| option(&item.options[*source_index], position, config)) + .map(|(position, source_index)| option(shown[*source_index], position, config)) .collect(); let key = if config.reveal.shows_key() { @@ -467,7 +468,7 @@ pub fn build( options, key, credit_overrides, - learning_objectives: if config.fields.objectives { + learning_targets: if config.fields.objectives { objectives } else { Vec::new() @@ -501,14 +502,10 @@ pub fn build( let objectives = objective_ids .into_iter() .filter_map(|id| { - course.learning_objectives.get(&id).map(|o| { - ( - id, - Objective { - text: markup::to_typst(&o.text), - unit: o.unit.clone(), - }, - ) + (course.is_objective(&id) || course.is_target(&id)).then(|| { + let text = markup::to_typst(&course.text_for(&id)); + let unit = course.objective_unit(&id).map(str::to_string); + (id, Objective { text, unit }) }) }) .collect(); @@ -662,11 +659,7 @@ fn calibration(item: &Item) -> Option> { /// The token for a response format. fn format_token(format: Format) -> &'static str { - match format { - Format::SingleBestAnswer => "single_best_answer", - Format::MultipleResponse => "multiple_response", - Format::TrueFalse => "true_false", - } + format.as_str() } /// The token for an administration platform. @@ -964,12 +957,12 @@ fn question_value(question: &Question, config: &RenderConfig) -> Value { }), ); - if !question.learning_objectives.is_empty() { + if !question.learning_targets.is_empty() { root.insert( - "learning-objectives", + "learning-targets", Value::Array( question - .learning_objectives + .learning_targets .iter() .map(|s| Value::str(s.as_str())) .collect(), @@ -1074,8 +1067,27 @@ fn numeric_map(map: &BTreeMap) -> Value { ) } -/// Emits authored markup as either a content block or a quoted string. -fn markup_value(source: &str, content: bool) -> Value { +/// Emits one markup-bearing field for a diagnostic report. +/// +/// The conversion itself lives in [`markup::to_typst`], which is the single +/// entry point for turning authoring markup into Typst: it rewrites inline +/// LaTeX math, routes mhchem through whalogen, and escapes the characters Typst +/// treats specially in content mode. This path used to carry its own copy of +/// the math rewriting, which drifted — the reports escaped nothing outside math +/// and silently disagreed with the exam papers about an unmatched `$`. +/// +/// # Arguments +/// +/// * `source` - the authoring source. +/// * `content` - whether to emit a content block rather than a quoted string. +/// A string is evaluated by the template with `eval(.., mode: "markup")`, so +/// both modes need the same escaping. +/// +/// # Returns +/// +/// The value to place in the payload. +pub(crate) fn markup_value(source: &str, content: bool) -> Value { + let source = markup::to_typst(source); if content { Value::content(source) } else { diff --git a/src/export/typst/template.rs b/src/export/typst/template.rs index a73942c..ddc7e4a 100644 --- a/src/export/typst/template.rs +++ b/src/export/typst/template.rs @@ -76,6 +76,12 @@ const EMBEDDED_KEY: &str = include_str!("templates/key.typ"); /// The bundled answer sheet template. const EMBEDDED_ANSWER_SHEET: &str = include_str!("templates/answer-sheet.typ"); +/// The bundled student diagnostic template. +const EMBEDDED_STUDENT_REPORT: &str = include_str!("templates/student-report.typ"); + +/// The bundled class diagnostic template. +const EMBEDDED_COHORT_REPORT: &str = include_str!("templates/cohort-report.typ"); + /// An injection point a template can declare. #[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] pub enum Slot { @@ -174,6 +180,8 @@ pub fn embedded(variant: Variant) -> &'static str { Variant::Exam => EMBEDDED_EXAM, Variant::Key => EMBEDDED_KEY, Variant::AnswerSheet => EMBEDDED_ANSWER_SHEET, + Variant::StudentReport => EMBEDDED_STUDENT_REPORT, + Variant::CohortReport => EMBEDDED_COHORT_REPORT, } } diff --git a/src/export/typst/templates/answer-sheet.typ b/src/export/typst/templates/answer-sheet.typ index 7258c86..be776fd 100644 --- a/src/export/typst/templates/answer-sheet.typ +++ b/src/export/typst/templates/answer-sheet.typ @@ -12,6 +12,8 @@ // templates/typst.yaml under `extra` rather than editing the geometry here, and // check one printed page against your scanner before running a class through it. +#import "@preview/mitex:0.2.7": mi + // coursebank:begin data #let cb-data = ( course: (code: "COURSE 101", title: "Sample Course", term: "2026s"), diff --git a/src/export/typst/templates/cohort-report.typ b/src/export/typst/templates/cohort-report.typ new file mode 100644 index 0000000..43536c5 --- /dev/null +++ b/src/export/typst/templates/cohort-report.typ @@ -0,0 +1,1438 @@ +// coursebank — class diagnostic template +// +// This file is a template, not generated output. `coursebank report cohort` +// replaces only the marked regions below. +// +// coursebank template dump --variant cohort-report +// typst watch templates/cohort-report.typ +// +// Markers: +// +// // coursebank:begin meta course, assessment, how many sat it, policy +// // coursebank:end meta +// // coursebank:begin data the class diagnostic +// // coursebank:end data +// +// This document is for you, not for the class. It carries item statistics, the +// triage lists, and the per-option breakdown: everything the student report +// withholds except the question text itself, which is in the bank where it +// belongs. Do not hand it out, because an option table tells a reader which +// letter was keyed. +// +// The order is deliberate. Where the class landed comes first, because that is +// the fact that changes next week. What to do with each question comes before +// the full item table, because the table is a reference and the triage lists are +// a work queue. + +#import "@preview/mitex:0.2.7": mi + +// ───────────────────────────────────────────────────────────────────────────── +// Data +// ───────────────────────────────────────────────────────────────────────────── + +// coursebank:begin meta +#let cb-meta = ( + course: (code: "COURSE 101", title: "Sample Course", term: "2026f"), + assessment: (id: "sample", title: "Sample assessment", date: "2026-01-01"), + generator: (tool: "coursebank", version: "0.0.0", on: "2026-01-02"), + policy: (mastery-threshold: 0.75, min-items-for-mastery: 2), + students-tested: 24, + class-size: 24, + extra: (:), +) +// coursebank:end meta + +// coursebank:begin data +#let cb-data = ( + students: 24, + items: 36, + distribution: ( + mean: 71.2, + median: 73.0, + sd: 11.4, + min: 44.0, + max: 94.0, + bins: ( + (low: 40, high: 50, count: 1), + (low: 50, high: 60, count: 3), + (low: 60, high: 70, count: 6), + (low: 70, high: 80, count: 9), + (low: 80, high: 90, count: 4), + (low: 90, high: 100, count: 1), + ), + ), + grades: ( + (letter: "A", low: 93.0, high: 100.0, gpa: 4.0, group: "A", count: 1, share: 0.04, at-or-above: 1), + (letter: "B", low: 83.0, high: 92.9, gpa: 3.0, group: "B", count: 5, share: 0.21, at-or-above: 6), + (letter: "C", low: 73.0, high: 82.9, gpa: 2.0, group: "C", count: 10, share: 0.42, at-or-above: 16), + (letter: "D", low: 63.0, high: 72.9, gpa: 1.0, group: "D", count: 6, share: 0.25, at-or-above: 22), + (letter: "F", low: 0.0, high: 62.9, gpa: 0.0, group: "F", count: 2, share: 0.08, at-or-above: 24), + ), + reliability: ( + alpha: 0.71, + sem: 2.1, + mean-p: 0.72, + mean-point-biserial: 0.24, + interpretation: "Reliability is 0.71, acceptable for a classroom exam.", + ), + levels: ( + (level: 1, name: "Remember", items: 6, rate: 0.91), + (level: 3, name: "Apply", items: 9, rate: 0.64), + ), + lectures: ( + ( + lecture: "L1.2", + title: "Entropy", + items: 4, + objectives: 3, + objectives-below: 2, + rate: 0.48, + questions: (10, 11, 12, 26), + worst-objective: [A sample objective the class struggled with.], + ), + ), + objectives: ( + ( + id: "lo-sample-gap", + text: [A sample objective the class struggled with.], + items: 3, + rate: 0.41, + meeting: 4, + developing: 7, + not-yet: 13, + thin: 0, + below-threshold: true, + ), + ), + gaps: (), + questions: ( + ( + number: 1, + item: "bank::q-sample-001", + level: 1, + targets: ("t-sample-gap",), + target-texts: ([A sample learning target the class struggled with.],), + taught-in: ("Entropy (L1.2), slides 4, 5",), + lectures: ("L1.2",), + difficulty-band: "moderate", + discrimination-band: "poor", + p: 0.42, + point-biserial: 0.05, + discrimination: 0.10, + blank-rate: 0.0, + key: ("B",), + stem: [A sample question stem, shown so the option shares can be read against what was asked.], + options: ( + ( + letter: "A", + text: [A sample distractor built on a real misconception.], + printed: ((form: "A", letter: "C"), (form: "B", letter: "A")), + count: 9, rate: 0.375, is-key: false, point-biserial: 0.11, nonfunctioning: false, + ), + ( + letter: "B", + text: [A sample key.], + printed: ((form: "A", letter: "D"), (form: "B", letter: "C")), + count: 10, rate: 0.417, is-key: true, point-biserial: 0.05, nonfunctioning: false, + ), + ( + letter: "C", + text: [A second sample distractor.], + printed: ((form: "A", letter: "A"), (form: "B", letter: "D")), + count: 5, rate: 0.208, is-key: false, point-biserial: -0.2, nonfunctioning: false, + ), + ( + letter: "D", + text: [A distractor nobody chose.], + printed: ((form: "A", letter: "B"), (form: "B", letter: "B")), + count: 0, rate: 0.0, is-key: false, nonfunctioning: true, + ), + ), + flags: ("ambiguous",), + notes: ([Distractor A drew as many strong students as the key.],), + prediction-notes: ([You predicted about 60% correct and observed 42%.],), + calibrated: false, + by-form: (A: 0.55, B: 0.30), + ), + ), + triage: ( + discard: (), + rekey: ( + ( + number: 1, + item: "bank::q-sample-001", + level: 1, + p: 0.42, + point-biserial: 0.05, + discrimination: 0.1, + objectives: ([A sample objective the class struggled with.],), + taught-in: ("Entropy (L1.2), slides 4, 5",), + option: "A", + option-share: 0.375, + option-point-biserial: 0.11, + reasons: ([Option A drew 38% and tracks total score better than the key.],), + ), + ), + revise: (), + reteach: (), + bounded: (), + clean: 35, + ), + predictions: ( + predicted: 36, + calibrated: 0, + mean-signed-error: 0.14, + mean-abs-error: 0.19, + within: 21, + band: 36, + band-hit: 11, + biggest-surprise: (number: 14, expected: 0.45, observed: 0.86), + ), + dropped-questions: (), + dropped-detail: ( + ( + number: 9, + item: "bank::q-sample-009", + level: 2, + targets: ("t-sample-gap",), + target-texts: ([A sample learning target.],), + dropped: true, + dropped-full-credit: true, + stem: [A sample question that was thrown out after the exam.], + taught-in: ("Entropy (L1.2)",), + lectures: ("L1.2",), + p: 0.18, + blank-rate: 0.0, + key: ("A",), + options: ( + (letter: "A", text: [The keyed option, which few chose.], count: 4, rate: 0.18, is-key: true, nonfunctioning: false), + (letter: "B", text: [The option most students read as correct.], count: 15, rate: 0.68, is-key: false, nonfunctioning: false), + (letter: "C", text: [A third option.], count: 3, rate: 0.14, is-key: false, nonfunctioning: false), + ), + flags: (), + notes: (), + prediction-notes: (), + calibrated: false, + by-form: (:), + ), + ), + revise: (), + forms: ( + (id: "A", students: 12, mean: 73.5, sd: 10.2), + (id: "B", students: 12, mean: 68.9, sd: 12.1), + ), + blueprint: (), + patterns: (), + warnings: (), +) +// coursebank:end data + +// ───────────────────────────────────────────────────────────────────────────── +// Settings +// ───────────────────────────────────────────────────────────────────────────── + +#let extra = cb-meta.at("extra", default: (:)) +#let accent = rgb(extra.at("accent", default: "#1f4e79")) +#let body-font = extra.at("font", default: "Roboto") +#let body-size = eval(extra.at("font-size", default: "9.5pt")) +#let paper = extra.at("paper", default: "us-letter") + +#let show-options = extra.at("option-tables", default: true) +#let show-triage = extra.at("triage", default: true) +#let show-guide = extra.at("stat-guide", default: true) +#let show-map = extra.at("item-map", default: true) +#let show-lectures = extra.at("lecture-table", default: true) +#let show-predictions = extra.at("prediction-check", default: true) + +#let threshold = cb-meta.policy.at("mastery-threshold", default: 0.75) +#let n-tested = cb-meta.at("students-tested", default: cb-meta.at("class-size", default: 0)) + +// The same scale the student report uses, so the two documents look related. +#let size-tag = 0.62em +#let size-micro = 0.72em // column heads, badges, legends +#let size-meta = 0.9em // item ids, provenance +#let size-small = 0.88em // table cells +#let size-lead = 0.94em // section explanations +#let size-h2 = 1.02em +#let size-h1 = 1.15em +#let size-sub = 0.95em +#let size-display = 1.35em +#let size-title = 1.5em + +#let step = 1.0em +#let entry-gap = 0.9em +#let prose-pad = 3.2cm +#let prose-pad-inset = prose-pad - 1cm + +#let ok-color = rgb("#2a9d8f") +#let mid-color = rgb("#D19F1F") +#let bad-color = rgb("#E24E29") +#let thin-color = luma(150) + +#set page( + paper: paper, + margin: (x: 1.7cm, y: 2.0cm), + header: text(size: size-meta, fill: luma(110))[ + #cb-meta.course.code · #cb-meta.assessment.title · class diagnostic + #h(1fr) + instructor copy + ], + footer: context text(size: size-meta, fill: luma(110))[ + #h(1fr) + Page #counter(page).display("1 of 1", both: true) + ], +) +#set text(font: body-font, size: body-size, lang: "en") +#set par(justify: false, leading: 0.6em) +#show table: set text(number-width: "tabular") +#show heading.where(level: 1): it => block(above: 1.5em, below: entry-gap)[ + #text(size: size-h1, weight: "bold", fill: accent)[#it.body] + #v(-0.45em) + #line(length: 100%, stroke: 0.6pt + accent.lighten(55%)) +] +#show heading.where(level: 2): it => block(above: 1em, below: 0.4em)[ + #text(size: size-h2, weight: "bold")[#it.body] +] + +// ───────────────────────────────────────────────────────────────────────────── +// Helpers +// ───────────────────────────────────────────────────────────────────────────── + +#let markup(v) = if type(v) == str { eval(v, mode: "markup") } else { v } +#let pct(rate) = str(calc.round(rate * 100)) + "%" +#let num(value, digits: 2) = str(calc.round(value, digits: digits)) +#let signed(value) = (if value >= 0 { "+" } else { "" }) + num(value) +#let plural(n, one, many) = if n == 1 { one } else { many } + +#let th(body) = text(size: size-micro, fill: luma(95), tracking: 0.04em)[#body] + +#let explain(body) = block(below: entry-gap)[ + #pad(right: prose-pad)[#text(size: size-lead, fill: luma(95))[#body]] +] + +#let rate-color(rate) = { + if rate >= threshold { ok-color } else if rate >= threshold * 0.6 { mid-color } else { bad-color } +} + +// The conventional reading of a corrected item-total correlation. Below about +// 0.20 an item is doing little sorting; negative means it sorts backwards. +#let r-color(r) = { + if r == none { thin-color } else if r < 0.0 { bad-color } else if r < 0.20 { mid-color } else { ok-color } +} + +#let group-color(group) = { + if group == "A" { ok-color } else if group == "B" { ok-color.lighten(25%) } else if group == "C" { + mid-color + } else if group == "D" { bad-color.lighten(25%) } else { bad-color } +} + +#let badge(label, color) = box( + fill: color.lighten(82%), + radius: 3pt, + inset: (x: 4pt, y: 2pt), +)[#text(size: size-micro, weight: "bold", fill: color.darken(12%))[#label]] + +// Flags travel as machine names. Printed in full they wrap a table column into +// three lines; these are the same findings in the space available. +#let flag-labels = ( + negative_discrimination: "negative r", + low_discrimination: "low r", + distractor_outperforms_key: "distractor > key", + key_underperforms: "key split", + too_easy: "easy", + too_hard: "hard", + high_rapid_guess: "rapid", + nonfunctioning_distractor: "dead option", + dif_flagged: "group gap", + ambiguous: "ambiguous", + design_mismatch: "off prediction", +) +#let flag-badge(flag) = badge(flag-labels.at(flag, default: flag), bad-color) + +#let bar(rate, color: accent, width: 2.4cm) = { + let r = calc.max(0.0, calc.min(1.0, rate)) + box(baseline: 0.15em)[ + #stack( + dir: ltr, + box(width: width * r, height: 0.58em, fill: color, radius: (left: 2pt)), + box(width: width * (1.0 - r), height: 0.58em, fill: luma(232), radius: (right: 2pt)), + ) + ] +} + +#let stat-card(label, value, note: none) = { + let parts = ( + text(size: size-micro, fill: luma(95), tracking: 0.06em)[#upper(label)], + text(size: size-display, weight: "bold", fill: accent)[#value], + ) + if note != none { parts.push(text(size: size-meta, fill: luma(95))[#note]) } + block(width: 100%, fill: luma(247), radius: 4pt, inset: (x: 9pt, y: 8pt))[ + #stack(dir: ttb, spacing: step * 0.4, ..parts) + ] +} + +// One question's identity line, used wherever a question is discussed rather +// than tabulated. +#let question-head(row, color) = { + let parts = ( + { + let bits = (text(weight: "bold", fill: color)[Question #row.number],) + let item = row.at("item", default: none) + if item != none { bits.push(text(size: size-meta, fill: luma(120))[#item]) } + let level = row.at("level", default: none) + if level != none { bits.push(text(size: size-meta, fill: luma(120))[L#level]) } + bits.join(h(0.5em)) + }, + ) + let stats = () + stats.push("p = " + num(row.p)) + let r = row.at("point-biserial", default: none) + if r != none { stats.push("r = " + signed(r)) } + let d = row.at("discrimination", default: none) + if d != none { stats.push("D = " + signed(d)) } + parts.push(text(size: size-meta, fill: luma(110))[#stats.join(" · ")]) + stack(dir: ttb, spacing: step * 0.4, ..parts) +} + +// The objective and lecture context for one question. +#let question-context(row) = { + let parts = () + // A question row carries ids in `targets` and prose in `target-texts`; a + // triage row carries the prose under `targets`. One lookup covers both. This + // is the target tier on purpose: a row about one item should name the + // performance that item measured, not the broader claim it rolls up to. + let targets = row.at("target-texts", default: row.at("targets", default: ())) + if targets.len() > 0 { + parts.push(text(size: size-meta, fill: luma(105))[ + #text(weight: "bold")[Measured:] #targets.map(t => markup(t)).join([; ]) + ]) + } + let taught = row.at("taught-in", default: ()) + if taught.len() > 0 { + parts.push(text(size: size-meta, fill: luma(105))[ + #text(weight: "bold")[Taught in:] #taught.join("; ") + ]) + } + if parts.len() == 0 { return none } + stack(dir: ttb, spacing: step * 0.65, ..parts) +} + +// ───────────────────────────────────────────────────────────────────────────── +// Heading and headline numbers +// ───────────────────────────────────────────────────────────────────────────── + +#block(below: entry-gap)[ + #stack( + dir: ttb, + spacing: step * 0.5, + text(size: size-title, weight: "bold")[#cb-meta.assessment.title · class diagnostic], + text(size: size-sub, fill: luma(110))[ + #cb-meta.course.code · #cb-meta.course.term + #{ + let date = cb-meta.assessment.at("date", default: none) + if date != none [ · administered #date ] + } + ], + ) +] + +#let dist = cb-data.distribution +#let rel = cb-data.reliability +#let grades = cb-data.at("grades", default: ()) + +#grid( + columns: (1fr, 1fr, 1fr, 1fr), + gutter: 9pt, + stat-card("students", str(cb-data.students), note: str(cb-data.items) + " scored items"), + stat-card("mean", str(calc.round(dist.mean)) + "%", note: "median " + str(calc.round(dist.median)) + "%"), + stat-card( + "spread", + "SD " + num(dist.sd, digits: 1), + note: str(calc.round(dist.min)) + "–" + str(calc.round(dist.max)) + "%", + ), + stat-card( + "reliability", + { + let alpha = rel.at("alpha", default: none) + if alpha != none { num(alpha) } else { "n/a" } + }, + note: { + let sem = rel.at("sem", default: none) + if sem != none { "SEM " + num(sem, digits: 1) + " items" } else { "KR-20" } + }, + ), +) + +#block(above: entry-gap)[ + #pad(right: prose-pad)[#text(size: size-lead, fill: luma(90))[#rel.interpretation]] +] + +// ───────────────────────────────────────────────────────────────────────────── +// Where the class landed +// ───────────────────────────────────────────────────────────────────────────── + +#let bins = dist.at("bins", default: ()) + +#if grades.len() > 0 [ + = Where the class landed + + #explain[ + Scores binned by the course's own letter scale rather than by ten-point + interval, because the bands are what the class will see. The running column + is cumulative: how many students are at that letter or above. + ] + + #let peak = calc.max(..grades.map(g => g.count), 1) + #let height = 2.8cm + + #align(center)[ + #grid( + columns: (1fr,) * grades.len(), + column-gutter: 3pt, + ..grades + .rev() + .map(g => { + let share = g.count / peak + let color = group-color(g.at("group", default: g.letter)) + stack( + dir: ttb, + spacing: 3pt, + align(center + bottom)[ + #box(height: height)[ + #align(bottom + center)[ + #stack( + dir: ttb, + spacing: 2pt, + align(center)[ + #text(size: size-micro, weight: "bold", fill: if g.count > 0 { color.darken(20%) } else { + thin-color + })[#g.count] + ], + box( + width: 100%, + height: height * share, + fill: color.lighten(30%), + stroke: 0.5pt + color, + radius: (top: 2pt), + ), + ) + ] + ] + ], + align(center)[#text(size: size-small, weight: "bold")[#g.letter]], + align(center)[#text(size: size-tag, fill: luma(115))[#num(g.low, digits: 0)+]], + ) + }), + ) + ] + + #v(0.5em) + + // The sentence the histogram is for. Whichever band the median sits in is + // where the class is, and the share at or below the failing band is the number + // that decides whether this was an exam problem or a teaching one. + #let failing = grades.filter(g => g.at("group", default: "") == "F") + #let n-failing = if failing.len() > 0 { failing.at(0).count } else { 0 } + #let top = grades.filter(g => g.count > 0) + + #pad(right: prose-pad)[ + #text(size: size-lead)[ + The median score of #num(dist.median, digits: 0)% falls in the + #{ + let median-band = grades.filter(g => dist.median >= g.low) + if median-band.len() > 0 { [*#median-band.at(0).letter*] } else { [lowest] } + } + band. + #if n-failing > 0 [ + #n-failing of #cb-data.students students, #pct(n-failing / calc.max(cb-data.students, 1)), + are below the lowest passing band. + ] + #if top.len() > 0 [ + The highest band anyone reached is *#top.at(0).letter*. + ] + ] + ] +] else if bins.len() > 0 [ + = Score distribution + + #explain[ + Ten-point bins, because `course.yaml` sets no `policy.grade_scale`. Add one + and this becomes a letter-grade histogram, which is the version worth + reading. + ] + + #let peak = calc.max(..bins.map(b => b.count), 1) + #let height = 3.0cm + #align(center)[ + #grid( + columns: (1fr,) * bins.len(), + column-gutter: 5pt, + ..bins.map(b => { + let share = b.count / peak + stack( + dir: ttb, + spacing: 3pt, + align(center + bottom)[ + #box(height: height)[ + #align(bottom)[ + #box( + width: 100%, + height: height * share, + fill: accent.lighten(if b.low >= 70 { 55% } else { 25% }), + radius: (top: 2pt), + ) + ] + ] + ], + align(center)[#text(size: size-micro, weight: "bold")[#b.count]], + align(center)[#text(size: size-tag, fill: luma(110))[#b.low]], + ) + }), + ) + ] + #align(center)[#text(size: size-micro, fill: luma(120))[percent scored, in ten-point bins]] +] + +// ───────────────────────────────────────────────────────────────────────────── +// Forms +// ───────────────────────────────────────────────────────────────────────────── + +#let forms = cb-data.at("forms", default: ()) + +#if forms.len() > 1 [ + = Forms + + #explain[ + Forms differ only in order, so their means should differ only by who sat + them. A persistent gap points at an item whose permutation made it easier or + harder, and the per-question columns further down are where to look. + ] + + #table( + columns: (auto, auto, auto, auto), + stroke: none, + align: (left, right, right, right), + inset: (x: 7pt, y: 5pt), + fill: (_, row) => if calc.odd(row) { luma(250) } else { white }, + table.header(th[FORM], th[STUDENTS], th[MEAN], th[SD]), + ..forms + .map(form => ( + [*#form.id*], + [#form.students], + [#num(form.mean, digits: 1)%], + [#num(form.sd, digits: 1)], + )) + .flatten(), + ) +] + +// ───────────────────────────────────────────────────────────────────────────── +// How to read the item statistics +// ───────────────────────────────────────────────────────────────────────────── + +#if show-guide [ + = How to read the item statistics + + #explain[ + Three numbers describe each question, and they only mean anything together. + ] + + #table( + columns: (auto, 1fr), + stroke: none, + align: (left + top, left + top), + inset: (x: 7pt, y: 5pt), + fill: (_, row) => if calc.odd(row) { luma(250) } else { white }, + table.header(th[], th[WHAT IT TELLS YOU]), + [*p* #text(size: size-meta, fill: luma(110))[difficulty]], + [ + The share who answered correctly, counting blanks as wrong. On its own it + says almost nothing: an item everyone passes may be a deliberate anchor, + and an item everyone misses is only a problem if it also failed to sort + students. + ], + + [*r* #text(size: size-meta, fill: luma(110))[discrimination]], + [ + The corrected item-total correlation: how well this question agrees with + the rest of the exam. This is the column to read first. Conventional + guidance puts 0.40 and above at excellent, 0.30 to 0.39 good, 0.20 to 0.29 + marginal, and below 0.20 poor. Negative means the students who did well + overall did worse here, which is almost always a keying error or a stem + with a second reading. + ], + + [*D* #text(size: size-meta, fill: luma(110))[upper minus lower]], + [ + The same idea computed crudely: the pass rate in the top quarter of the + class minus the pass rate in the bottom quarter. It uses only the extremes + and throws the middle away, so where D and r disagree, trust r. + ], + ) + + #block(above: entry-gap, breakable: false)[ + #pad(right: prose-pad)[ + #text(size: size-lead, fill: luma(95))[ + One constraint governs all of this: *discrimination is bounded by + difficulty*. An item that nearly everyone passes has almost no variance + left to correlate with anything, so a low r on a question at 90% correct + is arithmetic rather than a fault. That is why the triage below separates + questions whose weak r is explained by their difficulty from questions + that had room to sort students and did not. + ] + ] + ] +] + +// ───────────────────────────────────────────────────────────────────────────── +// The item map +// ───────────────────────────────────────────────────────────────────────────── + +#let questions = cb-data.at("questions", default: ()) + +#if show-map and questions.len() > 0 [ + = The exam at a glance + + #explain[ + Every question placed by difficulty and discrimination. The top-left cell is + where a well-built exam puts most of its items. The bottom row is where the + work is, and the right-hand column is where low discrimination is at least + partly explained by the ceiling. + ] + + #let d-bands = (("hard", "hard"), ("moderate", "moderate"), ("too easy", "at the ceiling")) + #let r-bands = ( + ("excellent", "excellent", ok-color), + ("good", "good", ok-color), + ("marginal", "marginal", mid-color), + ("poor", "poor", bad-color), + ("negative", "negative", bad-color), + ) + + #let in-cell(r-key, d-key) = questions.filter(q => ( + q.at("discrimination-band", default: "") == r-key and q.at("difficulty-band", default: "") == d-key + )) + + #table( + columns: (3.4cm, 1fr, 1fr, 1fr), + stroke: 0.5pt + luma(225), + align: left + top, + inset: 5pt, + fill: (x, y) => { + if y == 0 or x == 0 { white } else if in-cell(r-bands.at(y - 1).at(0), d-bands.at(x - 1).at(0)).len() > 0 { + r-bands.at(y - 1).at(2).lighten(93%) + } else { white } + }, + table.header(th[], ..d-bands.map(pair => align(center)[#th[#upper(pair.at(1))]])), + ..r-bands + .map(band => ( + { + let range = if band.at(0) == "excellent" { + "r ≥ 0.40" + } else if band.at(0) == "good" { + "0.30–0.39" + } else if band.at(0) == "marginal" { + "0.20–0.29" + } else if band.at(0) == "poor" { "0.00–0.19" } else { "r < 0" } + stack( + dir: ttb, + spacing: step * 0.3, + text(size: size-small, weight: "bold", fill: band.at(2))[#band.at(1)], + text(size: size-tag, fill: luma(120))[#range], + ) + }, + ..d-bands.map(pair => { + let inside = in-cell(band.at(0), pair.at(0)) + if inside.len() == 0 { + text(size: size-micro, fill: luma(200))[—] + } else { + text(size: size-small)[#inside.map(q => str(q.number)).join(" ")] + } + }), + )) + .flatten(), + ) + + #v(0.4em) + #pad(right: prose-pad)[ + #text(size: size-micro, fill: luma(120))[ + Difficulty: hard is 35% correct or below, at the ceiling is 85% or above. + A question with no variance at all appears in no cell. + ] + ] +] + +// ───────────────────────────────────────────────────────────────────────────── +// Triage +// ───────────────────────────────────────────────────────────────────────────── + +#let triage = cb-data.at("triage", default: (:)) + +#let triage-block(rows, title, color, lead) = { + if rows.len() == 0 { return none } + block(breakable: true, above: entry-gap, width: 100%)[ + == #title #text(size: size-meta, fill: luma(110), weight: "regular")[ + #rows.len() #plural(rows.len(), "question", "questions") + ] + + #pad(right: prose-pad)[#text(size: size-lead, fill: luma(95))[#lead]] + + #for row in rows [ + #block( + breakable: false, + above: step * 1.6, + width: 100%, + inset: (left: 0.6em), + stroke: (left: 2pt + color.lighten(45%)), + )[ + #{ + let parts = (question-head(row, color),) + let context_ = question-context(row) + if context_ != none { parts.push(context_) } + for reason in row.at("reasons", default: ()) { + parts.push(text(size: size-small)[#markup(reason)]) + } + let option = row.at("option", default: none) + if option != none { + let share = row.at("option-share", default: none) + let r = row.at("option-point-biserial", default: none) + let bits = ("option " + option,) + if share != none { bits.push("chosen by " + pct(share)) } + if r != none { bits.push("r = " + signed(r)) } + parts.push(text(size: size-meta, fill: luma(110))[ + #text(weight: "bold")[The option in question:] #bits.join(", ") + ]) + } + pad(right: prose-pad-inset)[#stack(dir: ttb, spacing: step * 0.6, ..parts)] + } + ] + ] + ] +} + +#if show-triage [ + #pagebreak(weak: true) + = What to do with each question + + #explain[ + Every question that raised something, sorted by what it asks of you. The + first three lists are decisions about items and are exclusive: there is no + point rewriting a distractor on a question you are about to discard. The + fourth is not about the items at all. + #{ + let clean = triage.at("clean", default: 0) + if clean > 0 [ + #clean #plural(clean, "question", "questions") raised nothing and are not + listed. + ] + } + ] + + #triage-block( + triage.at("discard", default: ()), + [Consider dropping from the score], + bad-color, + [ + The evidence says these did not measure what they were scored on. Dropping + a question raises every student's percentage, so it is a decision about + fairness rather than about the item: a question that sorted students + backwards contributed noise to every score it touched. + ], + ) + + #triage-block( + triage.at("rekey", default: ()), + [Consider credit for a second answer], + mid-color, + [ + A distractor here drew a substantial share of the class *and* tracked + overall performance at least as well as the key. That is the pattern of a + second defensible reading rather than a popular mistake. Either credit it + for this administration or rewrite the stem to exclude it. + ], + ) + + #triage-block( + triage.at("revise", default: ()), + [Rewrite before reusing], + mid-color, + [ + Weak but salvageable. These had room to separate students and did not, or + carry an option nobody chose, which makes a four-option question really a + three-option one. + ], + ) + + #triage-block( + triage.at("reteach", default: ()), + [Teach again], + accent, + [ + These questions worked. They separated the class cleanly, and the class + still missed them, which makes this a list about your teaching plan rather + than about your item bank. This is the list to bring to the next class + meeting. + ], + ) + + #triage-block( + triage.at("bounded", default: ()), + [No action needed], + ok-color, + [ + These carry a low-discrimination flag that their difficulty explains. They + are listed so the flag does not send you rewriting a question that is doing + exactly what an anchor item should do. + ], + ) +] + +// ───────────────────────────────────────────────────────────────────────────── +// By level +// ───────────────────────────────────────────────────────────────────────────── + +#if cb-data.levels.len() > 0 [ + + #block(breakable: false)[ + = By level + + #explain[ + Where the class sits on each kind of thinking. A drop from one level to the + next is expected; a cliff is worth reading as a gap in what was practised + rather than in what was taught. + ] + + #table( + columns: (auto, auto, auto, 1fr), + stroke: none, + align: (left + horizon, right + horizon, right + horizon, left + horizon), + inset: (x: 7pt, y: 5pt), + fill: (_, row) => if calc.odd(row) { luma(250) } else { white }, + table.header(th[LEVEL], th[ITEMS], th[CLASS], th[]), + ..cb-data + .levels + .map(level => ( + [#level.level · *#level.name*], + [#level.items], + [#pct(level.rate)], + bar(level.rate, color: rate-color(level.rate), width: 5cm), + )) + .flatten(), + ) + ] +] + +// ───────────────────────────────────────────────────────────────────────────── +// By lecture +// ───────────────────────────────────────────────────────────────────────────── + +#let lectures = cb-data.at("lectures", default: ()) + +#if show-lectures and lectures.len() > 0 [ + = By lecture, worst first + + #explain[ + Every question traces back to the lecture it was written from, so the item + statistics roll up into a view of the syllabus. This is the objective table + grouped into class meetings: one weak objective is a note, but three weak + objectives from the same lecture is a morning to reteach. + ] + + #table( + columns: (auto, 1fr, auto, auto, auto, 3.2cm), + stroke: none, + align: (left + horizon, left + horizon, right + horizon, right + horizon, right + horizon, left + horizon), + inset: (x: 6pt, y: 5pt), + fill: (_, row) => if calc.odd(row) { luma(250) } else { white }, + table.header(th[LEC], th[TITLE], th[Q], th[LO], th[CLASS], th[]), + ..lectures + .map(lecture => ( + text(size: size-small, fill: luma(110))[#lecture.lecture], + { + let parts = ([*#lecture.title*],) + let worst = lecture.at("worst-objective", default: none) + if worst != none { + parts.push(text(size: size-meta, fill: luma(110))[weakest: #markup(worst)]) + } + stack(dir: ttb, spacing: step * 0.4, ..parts) + }, + [#lecture.items], + { + let below = lecture.at("objectives-below", default: 0) + let total = lecture.at("objectives", default: 0) + if below > 0 { + text(fill: bad-color, weight: "bold")[#below/#total] + } else { + [#total] + } + }, + [#pct(lecture.rate)], + bar(lecture.rate, color: rate-color(lecture.rate), width: 3cm), + )) + .flatten(), + ) + + #v(0.4em) + #pad(right: prose-pad)[ + #text(size: size-micro, fill: luma(120))[ + LO counts the objectives the lecture's questions measured; a red fraction is + how many of them the class did not meet at #pct(threshold). + ] + ] +] + +// ───────────────────────────────────────────────────────────────────────────── +// By objective +// ───────────────────────────────────────────────────────────────────────────── + +#let objectives = cb-data.at("objectives", default: ()) + +#if objectives.len() > 0 [ + #pagebreak(weak: true) + = By objective, worst first + + #explain[ + The counts split the class into how many are meeting, developing, and not yet + on each objective. An objective where the class divides evenly is a different + teaching problem from one where nearly everyone is short. Where an objective + was measured by a single question, the split is thin evidence and the class + rate is the number to read. + ] + + #table( + columns: (1fr, auto, auto, auto, auto, auto, 2.2cm), + stroke: none, + align: ( + left + horizon, + right + horizon, + right + horizon, + right + horizon, + right + horizon, + right + horizon, + left + horizon, + ), + inset: (x: 6pt, y: 5pt), + fill: (_, row) => if calc.odd(row) { luma(250) } else { white }, + table.header(th[OBJECTIVE], th[Q], th[CLASS], th[MET], th[DEV], th[NOT], th[]), + ..objectives + .map(objective => ( + text(size: size-small)[#markup(objective.text)], + text(size: size-small, fill: luma(110))[#objective.items], + text(size: size-small, weight: "medium")[#pct(objective.rate)], + text(size: size-small)[#objective.meeting], + text(size: size-small)[#objective.developing], + text(size: size-small)[#objective.at("not-yet", default: 0)], + bar(objective.rate, color: rate-color(objective.rate), width: 2cm), + )) + .flatten(), + ) +] + +// ───────────────────────────────────────────────────────────────────────────── +// Item analysis +// ───────────────────────────────────────────────────────────────────────────── + +#if questions.len() > 0 [ + #pagebreak(weak: true) + = Item analysis + + #explain[ + One row per question, in the order they were numbered. Option letters are the + bank's, not any one form's. See the guide above for what p, r, and D mean; the + r column is colour-coded against the conventional cut points. + #{ + let dropped = cb-data.at("dropped-questions", default: ()) + if dropped.len() > 0 [ + #plural(dropped.len(), "Question", "Questions") + #dropped.map(d => str(d.number)).join(", ") + #plural(dropped.len(), "is", "are") + marked dropped in the assessment record, so + #plural(dropped.len(), "it is", "they are") + absent from this table and from every statistic above. + #{ + let credited = dropped.filter(d => d.at("full-credit", default: false)) + if credited.len() > 0 [ + #plural(credited.len(), "Question", "Questions") + #credited.map(d => str(d.number)).join(", ") + #plural(credited.len(), "was", "were") credited to every student, so + #plural(credited.len(), "it remains", "they remain") in the score + denominator and the means above match the grade of record. + ] + } + ] + } + ] + + #table( + columns: (auto, auto, auto, auto, auto, auto, auto, 1fr), + stroke: none, + align: ( + right + horizon, + right + horizon, + left + horizon, + right + horizon, + right + horizon, + right + horizon, + left + horizon, + left + horizon, + ), + inset: (x: 5pt, y: 4pt), + fill: (_, row) => if calc.odd(row) { luma(250) } else { white }, + table.header(th[Q], th[LVL], th[KEY], th[p], th[r], th[D], th[LEC], th[FLAGS]), + ..questions + .map(q => ( + [*#q.number*], + { + let level = q.at("level", default: none) + if level != none [#level] else [] + }, + [#q.at("key", default: ()).join("")], + text(fill: rate-color(q.p))[#num(q.p)], + { + let r = q.at("point-biserial", default: none) + if r == none { + text(fill: thin-color)[n/a] + } else { + text(fill: r-color(r), weight: if r < 0.2 { "bold" } else { "regular" })[#signed(r)] + } + }, + { + let d = q.at("discrimination", default: none) + if d != none [#signed(d)] else [—] + }, + { + // The lecture id only, which is all that fits and all that is needed to + // find the question in the bank. + let ids = q.at("lectures", default: ()) + if ids.len() > 0 { + text(size: size-tag, fill: luma(110))[#ids.join(", ")] + } else { [] } + }, + { + let flags = q.at("flags", default: ()) + let forms-gap = { + let by-form = q.at("by-form", default: (:)) + let values = by-form.values() + if values.len() > 1 and calc.max(..values) - calc.min(..values) >= 0.25 { + (badge("form gap " + pct(calc.max(..values) - calc.min(..values)), mid-color),) + } else { () } + } + stack(dir: ltr, spacing: 3pt, ..flags.map(flag-badge), ..forms-gap) + }, + )) + .flatten(), + ) +] + +// ───────────────────────────────────────────────────────────────────────────── +// Every question, with option tables +// ───────────────────────────────────────────────────────────────────────────── + +// Every scored question, not only the flagged ones. A question that raised no +// flag still has an option spread worth seeing: it is the reference for what a +// healthy item looks like on this exam, and the place to check a query about a +// question nothing was wrong with. The flagged ones are already called out by +// name in the triage lists above, so nothing is lost by putting them back among +// their neighbours here. +// Dropped items come from their own list, since they carry no statistics and +// belong in none of the tables above. They belong here: the option spread is +// the evidence that justified dropping them, and re-running the report after a +// drop should not erase it. +#let evidence = questions + cb-data.at("dropped-detail", default: ()) + +#if evidence.len() > 0 [ + #pagebreak(weak: true) + = The evidence, question by question + + #explain[ + The full breakdown for every question, flagged or not, including the ones + you dropped: what it asked, who chose what, and what any flags mean. A + dropped question is marked as such and carries no p, r, or D, because + dropping it overrode its credit. What students marked is untouched by the + drop, so its option spread still stands as the record of why it went. In question order, so you can find + one by its number; the lists above are the subset that needs a decision from + you, sorted by what they ask of you. The r column beside each option is the + correlation between choosing that option and scoring well on the rest of the + exam, which is how a defensible distractor announces itself. + + Each option shows the bank letter first, then what it was lettered on each + form (#raw("A:D") means the bank's option was printed as D on form A), so a + row here can be read against the paper a student is holding. + ] + + // Question order here, not triage order. The lists above are for deciding + // what to do; this section is for looking one question up while you do it, + // and a reader with a number in hand should not have to scan every entry. + #for q in evidence.sorted(key: q => q.number) [ + #block(breakable: false, above: entry-gap, width: 100%)[ + #{ + let parts = ( + { + let bits = (text(weight: "bold", fill: accent)[Question #q.number],) + let item = q.at("item", default: none) + if item != none { bits.push(text(size: size-meta, fill: luma(120))[#item]) } + if q.at("dropped", default: false) { + bits.push(text(size: size-micro, weight: "bold", fill: luma(110))[ + DROPPED#if q.at("dropped-full-credit", default: false) [ · FULL CREDIT] + ]) + } + let flags = q.at("flags", default: ()) + if flags.len() > 0 { + bits.push(stack(dir: ltr, spacing: 3pt, ..flags.map(flag-badge))) + } + bits.join(h(0.5em)) + }, + ) + + let context_ = question-context(q) + if context_ != none { parts.push(context_) } + + // The question itself. A row of option shares says a distractor drew + // 44% of the class; only the stem says whether that is a second + // defensible reading. Instructor copy only, which is the only place this + // section appears. + let stem = q.at("stem", default: none) + if stem != none { + parts.push(block( + width: 100%, + fill: luma(252), + stroke: (left: 2pt + luma(220)), + inset: (left: 8pt, rest: 6pt), + )[#text(size: size-small)[#markup(stem)]]) + } + + for note in q.at("notes", default: ()) { + parts.push(text(size: size-small)[— #markup(note)]) + } + + let predictions = q.at("prediction-notes", default: ()) + if predictions.len() > 0 { + parts.push(block( + width: 100%, + fill: luma(249), + radius: 3pt, + inset: (x: 7pt, y: 6pt), + )[ + #stack( + dir: ttb, + spacing: step * 0.4, + text(size: size-micro, weight: "bold", fill: luma(95), tracking: 0.04em)[ + #if q.at("calibrated", default: false) [AGAINST ITS CALIBRATION] else [AGAINST YOUR PREDICTION] + ], + ..predictions.map(note => text(size: size-meta)[#markup(note)]), + ) + ]) + } + + if show-options and q.at("options", default: ()).len() > 0 { + // The option column is two things: the bank letter every statistic is + // keyed by, and the letter the option actually carried on each paper. + // Without the second, a note about option D cannot be checked against + // the copy a student brings to office hours, since shuffling gives the + // same option a different letter on every form. + // + // The text goes last and takes the free column. It is the only cell + // whose length is not bounded, so anywhere else it either squeezes the + // numbers or wraps to three lines while they sit in a narrow gutter. + // Last, the numbers keep their natural widths and the text runs to the + // page edge. + parts.push(table( + columns: (auto, auto, auto, 2.6cm, auto, 1fr), + stroke: none, + align: ( + center + horizon, + right + horizon, + right + horizon, + left + horizon, + right + horizon, + left + top, + ), + inset: (x: 5pt, y: 3.5pt), + table.header(th[OPT], th[n], th[SHARE], th[], th[r], th[TEXT]), + ..q + .options + .map(option => ( + { + let printed = option.at("printed", default: ()) + let letter = if option.at("is-key", default: false) { + text(weight: "bold", fill: ok-color)[#option.letter] + } else { [#option.letter] } + if printed.len() == 0 { + letter + } else { + stack( + dir: ttb, + spacing: step * 0.25, + letter, + text(size: size-micro, fill: luma(125))[ + #printed.map(p => p.form + ":" + p.letter).join(" ") + ], + ) + } + }, + text(size: size-small)[#option.count], + text(size: size-small)[#pct(option.rate)], + bar( + option.rate, + color: if option.at("is-key", default: false) { ok-color } else { luma(160) }, + width: 2.2cm, + ), + { + let r = option.at("point-biserial", default: none) + if r == none { + text(fill: thin-color)[—] + } else if not option.at("is-key", default: false) and r > 0.0 { + text(size: size-small, fill: mid-color)[#signed(r)] + } else { + text(size: size-small)[#signed(r)] + } + }, + { + let body = option.at("text", default: none) + if body == none { + text(size: size-micro, fill: thin-color)[not in the bank] + } else if option.at("is-key", default: false) { + text(size: size-small, fill: ok-color.darken(25%))[#markup(body)] + } else { + text(size: size-small)[#markup(body)] + } + }, + )) + .flatten(), + )) + } + + stack(dir: ttb, spacing: step * 0.8, ..parts) + } + ] + ] +] + +// ───────────────────────────────────────────────────────────────────────────── +// How the predictions did +// ───────────────────────────────────────────────────────────────────────────── + +#let predictions = cb-data.at("predictions", default: (:)) + +#if show-predictions and predictions.at("predicted", default: 0) > 0 [ + #block(breakable: false)[ + = How your predictions did + + #let n = predictions.at("predicted", default: 0) + #let calibrated = predictions.at("calibrated", default: 0) + #let signed-error = predictions.at("mean-signed-error", default: none) + #let abs-error = predictions.at("mean-abs-error", default: none) + + #explain[ + #if calibrated == 0 [ + None of these #n #plural(n, "expectation", "expectations") rests on prior + data, so they are predictions rather than calibrations. A prediction that + misses is a fact about the prediction: it does not flag the item, and it is + summarised here instead of appearing #n times in the tables above. Once + `coursebank calibrate` has written statistics back into the bank, a + subsequent miss means the cohort or the teaching moved, and it will be + flagged. + ] else [ + #calibrated of #n #plural(n, "expectation", "expectations") rests on a + prior calibration. Those are the ones whose misses are flagged on the item, + because a calibrated item that moves is telling you about this cohort. The + rest are predictions, and a miss corrects the prediction. + ] + ] + + #grid( + columns: (1fr, 1fr, 1fr), + gutter: 9pt, + stat-card( + "difficulty bias", + if signed-error != none { + (if signed-error >= 0 { "+" } else { "" }) + str(calc.round(signed-error * 100)) + " pts" + } else { "n/a" }, + note: if signed-error != none and signed-error > 0 { + "items came out easier than you expected" + } else if signed-error != none { + "items came out harder than you expected" + } else { none }, + ), + stat-card( + "typical miss", + if abs-error != none { str(calc.round(abs-error * 100)) + " pts" } else { "n/a" }, + note: str(predictions.at("within", default: 0)) + " of " + str(n) + " inside tolerance", + ), + stat-card( + "discrimination band", + str(predictions.at("band-hit", default: 0)) + " / " + str(predictions.at("band", default: 0)), + note: "landed in the band you expected", + ), + ) + + #{ + let surprise = predictions.at("biggest-surprise", default: none) + if surprise != none { + block(above: entry-gap)[ + #pad(right: prose-pad)[ + #text(size: size-lead, fill: luma(95))[ + The largest single gap was question #surprise.number, predicted at + #pct(surprise.expected) and observed at #pct(surprise.observed). + ] + ] + ] + } + } + ] +] + +// ───────────────────────────────────────────────────────────────────────────── +// Profiles, blueprint, cautions +// ───────────────────────────────────────────────────────────────────────────── + +#let blueprint = cb-data.at("blueprint", default: ()) +#let warnings = cb-data.at("warnings", default: ()) +#let patterns = cb-data.at("patterns", default: ()) + +#if patterns.len() > 0 [ + = Response profiles + + #explain[ + Students grouped by the shape of their answers rather than by their totals. + Two students on the same percentage can need opposite things. + ] + + #for pattern in patterns [ + #block(above: step)[ + *#pattern.label* #text(size: size-meta, fill: luma(110))[ + #pattern.students #plural(pattern.students, "student", "students") + ] + ] + ] +] + +#if blueprint.len() > 0 [ + = Where the form differs from its blueprint + + #for line in blueprint [ + #block(above: step)[#text(size: size-small)[· #line]] + ] +] + +#if warnings.len() > 0 [ + = Cautions + + #for line in warnings [ + #block(above: step)[#pad(right: prose-pad)[#text(size: size-small)[· #line]]] + ] +] + +#v(1.2em) +#line(length: 100%, stroke: 0.5pt + luma(210)) +#v(0.4em) +#pad(right: prose-pad)[ + #text(size: size-micro, fill: luma(120))[ + Generated #cb-meta.generator.on by #cb-meta.generator.tool #cb-meta.generator.version. + With #cb-data.students students, a point-biserial carries a standard error + near #num(1.0 / calc.sqrt(calc.max(cb-data.students - 2, 1)), digits: 2), so + treat a single item's r as provisional and the pattern across items as real. + Pool several administrations before retiring an item. Instructor copy: the + option tables identify the key. + ] +] diff --git a/src/export/typst/templates/exam.typ b/src/export/typst/templates/exam.typ index bdad42a..1841f2c 100644 --- a/src/export/typst/templates/exam.typ +++ b/src/export/typst/templates/exam.typ @@ -24,6 +24,8 @@ // copy — export the `key` variant instead, or the day you forget an `if` is the // day the class gets the answers. +#import "@preview/mitex:0.2.7": mi + // ───────────────────────────────────────────────────────────────────────────── // Metadata // ───────────────────────────────────────────────────────────────────────────── @@ -52,12 +54,15 @@ // how a course changes the look without editing this file at all. #let extra = cb-meta.at("extra", default: (:)) #let accent = rgb(extra.at("accent", default: "#1f4e79")) -#let body-font = extra.at("font", default: "Libertinus Serif") +#let body-font = extra.at("font", default: "Roboto") #let body-size = eval(extra.at("font-size", default: "11pt")) #let paper = extra.at("paper", default: "us-letter") #let show-name-block = extra.at("name-block", default: true) #let show-points = extra.at("show-points", default: true) #let page-per-item = extra.at("page-per-item", default: false) +#let show-bubbles = extra.at("show-bubbles", default: true) +#let bubble-radius = eval(extra.at("bubble-radius", default: "0.5em")) +#let scratch-space = eval(extra.at("scratch-space", default: "1.5em")) #let form-note = if cb-meta.form.at("count", default: 1) > 1 { " · Form " + cb-meta.form.id @@ -72,7 +77,7 @@ #cb-meta.course.code · #cb-meta.assessment.title#form-note ], footer: context text(size: 0.85em)[ - #counter(page).display("Page 1 of 1", both: true) + Page #counter(page).display("1 of 1", both: true) ], ) #set text(font: body-font, size: body-size, lang: "en") @@ -120,6 +125,36 @@ } } +// An unfilled bubble a student marks by hand. `bubble-radius` is the same knob +// the standalone answer sheet reads, so the two stay visually consistent if you +// ever generate both. +#let bubble() = circle(radius: bubble-radius, stroke: 0.5pt) + +// A small colored badge for a question's cognitive level, in the same visual +// style as the tier badges on Exam 4. `level` counts up from 1; swap the order +// of `level-colors` if your taxonomy numbers complexity the other way. Nothing +// is drawn when a variant withholds `level` (`fields.level: false`), same as +// any other optional field. +#let level-colors = ( + (bg: rgb("#D9E8EE"), fg: rgb("#264653")), + (bg: rgb("#DCF6F3"), fg: rgb("#2a9d8f")), + (bg: rgb("#FAF2DD"), fg: rgb("#D19F1F")), + (bg: rgb("#FBE6E0"), fg: rgb("#E24E29")), + (bg: rgb("#F0E9F5"), fg: rgb("#9967B6")), +) + +#let level-tag(q) = { + let level = q.at("level", default: none) + if level != none { + let idx = calc.max(1, calc.min(level, level-colors.len())) - 1 + let colors = level-colors.at(idx) + let label = q.at("level-name", default: "Level " + str(level)) + box(fill: colors.bg, radius: 3pt, inset: (x: 6pt, y: 3pt))[ + #text(weight: "bold", size: 0.75em, fill: colors.fg)[#label] + ] + } +} + // ───────────────────────────────────────────────────────────────────────────── // The renderer // ───────────────────────────────────────────────────────────────────────────── @@ -137,30 +172,44 @@ // The number is the one recorded in the assessment record, not the position on // the page. Keep it that way: it is the join key to every grading export. - // - // Built in code rather than written as `*#q.number.*` because a field access - // followed by a literal period reads as the start of another field access. let number-label = str(q.number) + "." - block(above: 1.2em, below: 0.5em)[ - *#number-label* #points-tag(q) #markup(q.stem) - ] - - if q.at("multi-select", default: false) { - block(below: 0.4em)[ - #text(size: 0.9em, style: "italic")[Select all that apply.] - ] - } - - block(inset: (left: 1.2em))[ - #for opt in q.at("options", default: ()) { - grid( - columns: (1.4em, 1fr), - gutter: 0.2em, - [#(opt.letter + ".")], [#markup(opt.text)], + block(breakable: false)[ + #block(above: 1.2em, below: 0.3em)[ + #grid( + columns: (1fr, auto), + align: (left + horizon, right + horizon), + [*#number-label* #points-tag(q)], level-tag(q), ) - v(0.15em) + ] + #block(below: 1.5em)[#markup(q.stem)] + + #if q.at("multi-select", default: false) { + block(below: 0.4em)[ + #text(size: 0.9em, style: "italic")[Select all that apply.] + ] } + + #block(inset: (left: 1.2em))[ + #for opt in q.at("options", default: ()) { + if show-bubbles { + grid( + columns: (1.6em, 1.4em, 1fr), + gutter: 0.2em, + align(top)[#v(-bubble-radius / 2.1) #bubble()], [#(opt.letter + ".")], [#markup(opt.text)], + ) + } else { + grid( + columns: (1.4em, 1fr), + gutter: 0.2em, + [#(opt.letter + ".")], [#markup(opt.text)], + ) + } + v(0.15em) + } + ] + + #if scratch-space > 0pt { v(scratch-space) } ] if page-per-item { pagebreak(weak: true) } @@ -170,43 +219,95 @@ // The page // ───────────────────────────────────────────────────────────────────────────── +#set page(paper: paper, margin: 2cm) +#set text(font: body-font, size: body-size, lang: "en") +#set par(justify: false, leading: 0.7em) + +// ── Cover page: no header or footer, so it reads as its own sheet. ── +#set page(header: none, footer: none) + +#v(4em) #align(center)[ - #text(size: 1.4em, weight: "bold", fill: accent)[#cb-meta.assessment.title]\ - #text(size: 0.95em)[ - #cb-meta.course.code — #cb-meta.course.title · #cb-meta.assessment.term + #text(weight: "bold", size: 18pt, fill: accent)[ + #cb-meta.course.code — #cb-meta.course.title ]\ - #text(size: 0.9em)[ + #text(size: 16pt)[#cb-meta.assessment.title#form-note]\ + #text(size: 14pt)[ #cb-meta.assessment.at("date", default: "") #{ let m = cb-meta.assessment.at("minutes-allowed", default: none) if m != none { " · " + str(int(calc.round(m))) + " minutes" } } - #{ - let p = cb-meta.totals.at("points", default: none) - if p != none { " · " + fmt-points(p) } - } - ] + ]\ + #{ + let p = cb-meta.totals.at("points", default: none) + if p != none { text(size: 12pt)[#fmt-points(p)] } + } ] -#if show-name-block { - block(above: 1em, below: 1.5em)[ - #grid( - columns: (auto, 1fr, auto, 1fr), - gutter: 0.6em, - [*Name*], box(width: 100%, repeat[.]), [*Student ID*], box(width: 100%, repeat[.]), - ) - ] -} +#v(1.5em) #{ let instructions = cb-meta.assessment.at("instructions", default: none) - if instructions != none { - block(fill: luma(245), inset: 8pt, radius: 3pt, width: 100%)[ - #markup(instructions) - ] - } + if instructions != none [ + Please read the following instructions carefully before beginning your assessment. + + #v(1em) + + #markup(instructions) + ] } +#if show-name-block [ + #v(1.5em) + + I agree to follow the above instructions. I affirm that all work on this + assessment will be my own and that I will not give or receive any + unauthorized assistance. To have your assessment graded, you must write your + name, sign, and provide your student ID below. + + #v(1em) + + #grid( + columns: (50%, 50%), + rows: 6em, + [ + #v(1em) + #line(length: 18em) + + *Name* + ], + [ + #v(1em) + #line(length: 18em) + + *Signature* + ], + ) + + #line(length: 8em) + + *Student ID* +] + +#pagebreak() +// Intentionally blank. Pull this sheet and the cover above as one unit if you +// need the cover gone — nothing on the back of either one is a question. +#pagebreak() + +// ── The exam itself: header and footer turn on here. ── +#set page( + header: text(size: 0.85em)[ + #cb-meta.course.code · #cb-meta.assessment.title#form-note + ], + footer: context text(size: 0.85em)[ + Page #counter(page).display("1 of 1", both: true) + ], +) +// Restart the printed page count here too, so the footer reads "Page 1 of N" +// for the exam itself rather than counting the cover and the blank page. +#counter(page).update(1) + // coursebank:begin questions #render-question(( number: 1, diff --git a/src/export/typst/templates/key.typ b/src/export/typst/templates/key.typ index 00c6c26..2610199 100644 --- a/src/export/typst/templates/key.typ +++ b/src/export/typst/templates/key.typ @@ -13,6 +13,8 @@ // paper's config, and it is why these are two templates rather than one with a // flag. +#import "@preview/mitex:0.2.7": mi + // coursebank:begin data #let cb-data = ( course: (code: "COURSE 101", title: "Sample Course", term: "2026s"), @@ -39,11 +41,29 @@ ) // coursebank:end data +// ───────────────────────────────────────────────────────────────────────────── +// Settings +// ───────────────────────────────────────────────────────────────────────────── + +// Anything under `extra` in templates/typst.yaml arrives here untouched, which is +// how a course changes the look without editing this file at all. +#let extra = cb-data.at("extra", default: (:)) +#let accent = rgb(extra.at("accent", default: "#1f4e79")) +#let body-font = extra.at("font", default: "Roboto") +#let body-size = eval(extra.at("font-size", default: "11pt")) +#let paper = extra.at("paper", default: "us-letter") +#let show-name-block = extra.at("name-block", default: true) +#let show-points = extra.at("show-points", default: true) +#let page-per-item = extra.at("page-per-item", default: false) +#let show-bubbles = extra.at("show-bubbles", default: true) +#let bubble-radius = eval(extra.at("bubble-radius", default: "0.5em")) +#let scratch-space = eval(extra.at("scratch-space", default: "1.5em")) + #let extra = cb-data.at("extra", default: (:)) #let accent = rgb(extra.at("accent", default: "#1f4e79")) #set page(paper: extra.at("paper", default: "us-letter"), margin: 2cm) -#set text(size: 10pt) +#set text(font: body-font, size: body-size, lang: "en") #let markup(v) = if type(v) == str { eval(v, mode: "markup") } else { v } #let fmt-points(p) = if p == calc.trunc(p) { str(calc.trunc(p)) } else { str(p) } diff --git a/src/export/typst/templates/student-report.typ b/src/export/typst/templates/student-report.typ new file mode 100644 index 0000000..da43647 --- /dev/null +++ b/src/export/typst/templates/student-report.typ @@ -0,0 +1,994 @@ +// coursebank — individual diagnostic template +// +// This file is a template, not generated output. `coursebank report students` +// replaces only the marked regions below and leaves every other line exactly as +// you wrote it, so this is where layout decisions belong. +// +// coursebank template dump --variant student-report +// typst watch templates/student-report.typ +// +// Two markers are in play: +// +// // coursebank:begin meta course, assessment, how many sat it, policy +// // coursebank:end meta +// // coursebank:begin data one student's diagnostic +// // coursebank:end data +// +// Both regions ship with sample values, so `typst watch` works before any report +// has been generated. +// +// What is not in the payload: the questions. There is no stem field and no option +// text field, on any question, in any configuration. A student report is handed +// back before the makeup exam is given, and a report that reproduces the paper +// cannot be. If you find yourself wanting to print the question, print its number +// and let the student read it off their own copy. +// +// Two fields appear only when you ask for them at the command line, because both +// trade a student's understanding against reusing the question: `misconception` +// (`--misconceptions`) and `worked` (`--solutions`). The blocks that print them +// are below and cost nothing when the fields are absent. +// +// The prose in this file is addressed to a nineteen-year-old reading their own +// result, alone, possibly disappointed. It explains every number before showing +// it. If you change one thing here, keep that. + +#import "@preview/mitex:0.2.7": mi +#import "@preview/whalogen:0.3.0": ce + +// ───────────────────────────────────────────────────────────────────────────── +// Data +// ───────────────────────────────────────────────────────────────────────────── + +// coursebank:begin meta +#let cb-meta = ( + course: (code: "COURSE 101", title: "Sample Course", term: "2026f"), + assessment: (id: "sample", title: "Sample assessment", date: "2026-01-01"), + generator: (tool: "coursebank", version: "0.0.0", on: "2026-01-02"), + policy: (mastery-threshold: 0.75, min-items-for-mastery: 2), + students-tested: 24, + class-size: 24, + extra: (:), +) +// coursebank:end meta + +// coursebank:begin data +#let cb-data = ( + student-key: "s-000000000000", + name: "Sample Student", + email: "sample@example.edu", + form: "A", + score: (points: 27.0, possible: 36.0, percent: 75.0, bonus: 1.0, correct: 27, items: 36), + standing: (class-mean: 71.2, class-sd: 11.4, band: "upper half"), + levels: ( + ( + level: 1, + name: "Remember", + blurb: "recalling terms, facts, and definitions", + items: 6, + rate: 1.0, + class-rate: 0.91, + comparison: "above the class", + ), + ( + level: 3, + name: "Apply", + blurb: "using a procedure in a new situation", + items: 9, + rate: 0.56, + class-rate: 0.64, + comparison: "below the class", + ), + ), + objectives: ( + ( + id: "lo-sample-met", + text: [A sample objective this student met.], + items: 3, + credit: 3.0, + rate: 1.0, + lower: 0.44, + upper: 1.0, + class-rate: 0.81, + status: "meeting", + symbol: "✓", + confident: false, + thin-evidence: false, + levels: (1, 2), + ), + ( + id: "lo-sample-gap", + text: [A sample objective to work on.], + items: 3, + credit: 1.0, + rate: 0.33, + lower: 0.06, + upper: 0.79, + class-rate: 0.58, + status: "not yet", + symbol: "✗", + confident: false, + thin-evidence: false, + levels: (3,), + ), + ), + strengths: ((id: "lo-sample-met", text: [A sample objective this student met.], rate: 1.0, items: 3),), + focus: ((id: "lo-sample-gap", text: [A sample objective to work on.], rate: 0.33, items: 3),), + questions: ( + ( + number: 1, + level: 1, + targets: ("t-sample-met",), + measured: (), + correct: true, + credit: 1.0, + bonus: false, + dropped: false, + blank: false, + class-rate: 0.91, + taught-in: (), + review: (), + ), + ( + number: 2, + level: 3, + targets: ("t-sample-gap",), + measured: (( + objective: [A sample objective to work on.], + target: [A sample learning target under it.], + ),), + correct: false, + credit: 0.0, + bonus: false, + blank: false, + class-rate: 0.58, + feedback: [This is the note written for the option that was chosen.], + hint: [This is the question you would ask someone reconsidering that option.], + taught-in: ("Enthalpy (L1.1), slides 12, 13",), + review: ((citation: "KKW §6.2", title: "Molecules and Medicine", url: "https://example.edu/6/2"),), + ), + ), + dropped-questions: ((number: 35, full-credit: true),), + review-lectures: ( + ( + lecture: "L1.1", + title: "Enthalpy", + targets-missed: 2, + questions-missed: 3, + questions: (2, 14, 15), + slides: (12, 13), + targets: ([A sample learning target to work on.], [A second one from the same lecture.]), + ), + ), + study: ( + ( + objective: "lo-sample-gap", + text: [A sample objective to work on.], + rate: 0.33, + readings: ( + ( + citation: "KKW §6.2", + lecture: "L1.1", + lecture-title: "Enthalpy", + focus: [What to take from this section.], + supplemental: false, + ), + ), + ), + ), +) +// coursebank:end data + +// ───────────────────────────────────────────────────────────────────────────── +// Settings +// ───────────────────────────────────────────────────────────────────────────── + +#let extra = cb-meta.at("extra", default: (:)) +#let accent = rgb(extra.at("accent", default: "#1f4e79")) +#let body-font = extra.at("font", default: "Roboto") +#let body-size = eval(extra.at("font-size", default: "10pt")) +#let paper = extra.at("paper", default: "us-letter") + +// A five-step type scale. Sizes are picked from this list rather than invented at +// the call site, which is what stops a document from drifting into a dozen +// slightly different smalls. +#let size-tag = 0.6em // the level tag inside a question box +#let size-micro = 0.72em // column heads, badges, legends +#let size-meta = 0.9em // provenance: lecture ids, citations, dates +#let size-small = 0.88em // table cells, objective bullets, callouts +#let size-lead = 0.94em // the explanation under each heading +#let size-name = 1.1em // the student's name +#let size-h2 = 1.02em +#let size-h1 = 1.15em +#let size-display = 1.4em // the three numbers at the top +#let size-title = 1.5em + +// One vertical step. Every gap inside a list entry is this or a stated multiple +// of it, and the entries themselves are separated by `entry-gap`. Uneven rhythm +// in a document like this comes from mixing `v()`, linebreaks, and Typst's +// default paragraph spacing in the same block; a stack plus one step avoids all +// three. +#let step = 1.2em +#let entry-gap = 1.0em + +// Prose is held to a readable measure instead of spanning the full text block. +// At 10pt across 17.8cm a line runs to about a hundred characters, which is +// roughly a third too long to track comfortably. Tables and the question grid +// still use the whole width, which is what they are for. +#let prose-pad = 1.5cm + +// The same right edge for prose that already sits in a gutter, so a note body +// lines up with the explanation above it. One centimetre is the 2.8em of number +// column plus gutter at the default body size. +#let prose-pad-inset = prose-pad - 1cm + +// Turn sections off from templates/typst.yaml rather than by deleting code, so +// a course that does not want the question map keeps the rest of this file. +#let show-question-map = extra.at("question-map", default: true) +#let show-feedback = extra.at("feedback", default: true) +#let show-study = extra.at("study-plan", default: true) +#let show-comparison = extra.at("comparison", default: true) +#let show-lecture-plan = extra.at("lecture-plan", default: true) +#let show-item-readings = extra.at("item-readings", default: true) +#let show-intro = extra.at("intro", default: true) + +#let threshold = cb-meta.policy.at("mastery-threshold", default: 0.75) +#let n-tested = cb-meta.at("students-tested", default: cb-meta.at("class-size", default: 0)) + +#let ok-color = rgb("#2a9d8f") +#let mid-color = rgb("#D19F1F") +#let bad-color = rgb("#E24E29") +#let thin-color = luma(150) + +#set page( + paper: paper, + margin: (x: 1.9cm, y: 2.1cm), + header: text(size: size-meta, fill: luma(110))[ + #cb-meta.course.code · #cb-meta.assessment.title · individual diagnostic + ], + footer: context text(size: size-meta, fill: luma(110))[ + #h(1fr) + Page #counter(page).display("1 of 1", both: true) + ], +) +#set text(font: body-font, size: body-size, lang: "en") +#set par(justify: false, leading: 0.62em) + +// Figures in the tables are columns of numbers, so they get tabular widths and +// line up under one another. +#show table: set text(number-width: "tabular") + +#show heading.where(level: 1): it => block(above: 1.5em, below: entry-gap)[ + #text(size: size-h1, weight: "bold", fill: accent)[#it.body] + #v(-0.45em) + #line(length: 100%, stroke: 0.6pt + accent.lighten(55%)) +] +#show heading.where(level: 2): it => block(above: 1em, below: 0.45em)[ + #text(size: size-h2, weight: "bold")[#it.body] +] + +// ───────────────────────────────────────────────────────────────────────────── +// Helpers +// ───────────────────────────────────────────────────────────────────────────── + +#let markup(v) = if type(v) == str { eval(v, mode: "markup") } else { v } + +// A rate is a fraction in 0..1; a percent is already out of 100. Keeping the two +// straight is the only arithmetic this template does. +#let pct(rate) = str(calc.round(rate * 100)) + "%" +#let pct1(value) = str(calc.round(value, digits: 0)) + "%" + +#let plural(n, one, many) = if n == 1 { one } else { many } + +#let status-color(status) = { + if status == "meeting" { ok-color } else if status == "developing" { mid-color } else if status == "not yet" { + bad-color + } else { thin-color } +} + +#let rate-color(rate) = { + if rate >= threshold { ok-color } else if rate >= threshold * 0.6 { mid-color } else { bad-color } +} + +#let badge(label, color) = box( + fill: color.lighten(82%), + radius: 3pt, + inset: (x: 5pt, y: 2.5pt), +)[#text(size: size-micro, weight: "bold", fill: color.darken(12%))[#label]] + +// A column head. One definition rather than the same three arguments repeated at +// every header cell, so the heads cannot drift apart. +#let th(body) = text(size: size-micro, fill: luma(95), tracking: 0.04em)[#body] + +// A numbered disc for a ranked list. Fixed width, so the text of every entry +// starts at the same place no matter whether the rank is 1 or 11. +#let rank(n, color) = box( + width: 1.35em, + height: 1.35em, + radius: 50%, + fill: color, +)[#align(center + horizon)[#text(size: size-micro, weight: "bold", fill: white)[#str(n)]]] + +// An aside with a rule down its left edge: the action to take, set apart from the +// explanation above it without another box or another tint. +#let callout(name, body) = block( + inset: (left: 0.65em), + stroke: (left: 1.5pt + accent.lighten(55%)), +)[ + #text(size: size-small)[#text(weight: "bold", fill: luma(75))[#name:] #body] +] + +// A provenance line: where something was taught, or what to read. +#let meta-pair(name, body) = text(size: size-meta, fill: luma(115))[ + #text(weight: "bold")[#name:] #body +] + +// Two boxes side by side rather than an overlay: no `place`, no coordinate +// arithmetic, and it degrades to something sensible at any width. +#let bar(rate, color: accent, width: 3.6cm) = { + let r = calc.max(0.0, calc.min(1.0, rate)) + box(baseline: 0.15em)[ + #stack( + dir: ltr, + box(width: width * r, height: 0.62em, fill: color, radius: (left: 2pt)), + box(width: width * (1.0 - r), height: 0.62em, fill: luma(232), radius: (right: 2pt)), + ) + ] +} + +// The three cards at the top. The label sits above the number and the gloss +// below it, so the eye lands on the figure and can then read outwards. +#let stat-card(label, value, note: none) = { + let parts = ( + text(size: size-micro, fill: luma(95), tracking: 0.06em)[#upper(label)], + text(size: size-display, weight: "bold", fill: accent)[#value], + ) + if note != none { + parts.push(text(size: size-meta, fill: luma(95))[#note]) + } + block( + height: 9em, + width: 100%, + fill: luma(247), + radius: 4pt, + inset: (x: 10pt, y: 9pt), + )[#stack(dir: ttb, spacing: step * 1.0, ..parts)] +} + +// The explanation under a heading. Every section has one, because a number a +// student cannot interpret is worse than no number at all. +#let explain(body) = block(below: entry-gap)[ + #pad(right: prose-pad)[#text(size: size-lead, fill: luma(95))[#body]] +] + +// ───────────────────────────────────────────────────────────────────────────── +// Heading +// ───────────────────────────────────────────────────────────────────────────── + +#block(below: 0.35em)[ + #text(size: size-title, weight: "bold")[#cb-meta.assessment.title] + #h(0.6em) + #text(fill: luma(110))[ + #cb-meta.course.code · #cb-meta.course.title + ] +] + +#let student-name = cb-data.at("name", default: none) +#let student-email = cb-data.at("email", default: none) +#let student-sid = cb-data.at("sid", default: none) + +#block(below: entry-gap)[ + #stack( + dir: ttb, + spacing: step * 0.5, + text(size: size-name, weight: "bold")[ + #if student-name != none { student-name } else { cb-data.student-key } + ], + { + // Identity on its own line, and only what the export actually carried. A + // label with nothing after it reads like a mistake. + let parts = () + if student-email != none { parts.push(student-email) } + if student-sid != none { parts.push(student-sid) } + let form = cb-data.at("form", default: none) + if form != none { parts.push("Form " + form) } + let date = cb-meta.assessment.at("date", default: none) + if date != none { parts.push(date) } + if parts.len() > 0 { + text(size: size-meta, fill: luma(110))[#parts.join(" · ")] + } + }, + ) +] + +#if show-intro [ + #block( + width: 100%, + fill: accent.lighten(95%), + radius: 4pt, + inset: (x: 11pt, y: 10pt), + below: 1.2em, + )[ + #pad(right: prose-pad - 1.1cm)[ + #text(size: size-lead)[ + This map explains what the exam covered and what your results mean, so you can use the information. Your score appears first because that's what you probably want to see. After that, you'll find more helpful details: which types of questions you did well on, which topics you might want to review, and what to read for each area. The last two sections are the most practical, so if you only read part of this, focus on those. + + This report does not judge your abilities, and one question alone does not say much. If a score is based on just one or two questions, the report will point that out so you don't read too much into it. + ] + ] + ] +] + +// ───────────────────────────────────────────────────────────────────────────── +// Score +// ───────────────────────────────────────────────────────────────────────────── + +#let score = cb-data.score +#let standing = cb-data.at("standing", default: none) + +#grid( + columns: (1fr, 1fr, 1fr), + gutter: 10pt, + stat-card( + "your score", + pct1(score.percent), + note: str(score.points) + + " of " + + str(score.possible) + + " points" + + (if score.at("bonus", default: 0.0) > 0.0 { " (+" + str(score.bonus) + " bonus)" } else { "" }), + ), + stat-card( + "questions right", + str(score.correct) + " / " + str(score.items), + note: "out of " + str(score.items) + " " + plural(score.items, "question", "questions"), + ), + if standing != none and show-comparison { + stat-card( + "class average", + pct1(standing.class-mean), + note: "across the " + str(n-tested) + " students who took this exam · you are in the " + standing.band, + ) + } else { + stat-card( + "students tested", + str(n-tested), + note: "took this exam", + ) + }, +) + +#let dropped-questions = cb-data.at("dropped-questions", default: ()) + +#if dropped-questions.len() > 0 [ + #block(above: entry-gap)[ + #pad(right: prose-pad)[ + #text(size: size-lead)[ + #{ + // Two kinds of drop, and they need different sentences. A credited + // question is still in the denominator, so telling a student it was + // removed would not match the arithmetic they can do themselves. + let credited = dropped-questions.filter(d => d.at("full-credit", default: false)) + let removed = dropped-questions.filter(d => not d.at("full-credit", default: false)) + let numbers = list => list.map(d => str(d.number)).join(", ") + + if credited.len() > 0 [ + #plural(credited.len(), "Question", "Questions") #numbers(credited) + #plural(credited.len(), "was", "were") thrown out after the exam. + Everyone received full credit for + #plural(credited.len(), "it", "them"), so + #plural(credited.len(), "it is", "they are") still counted in the + score above and whatever you chose made no difference. + ] + if removed.len() > 0 [ + #plural(removed.len(), "Question", "Questions") #numbers(removed) + #plural(removed.len(), "was", "were") thrown out and removed from + scoring, so your percentage is out of the remaining questions. + Nothing you wrote on #plural(removed.len(), "it", "them") counted + either way. + ] + } + ] + ] + ] +] + +// ───────────────────────────────────────────────────────────────────────────── +// Levels +// ───────────────────────────────────────────────────────────────────────────── + +#if cb-data.levels.len() > 0 [ + = How you did, by kind of thinking + + #explain[ + Each question on this exam was designed to test a specific type of thinking, from recalling definitions to analyzing situations. The table below breaks down your results by these types. The number in brackets shows how many questions of each type you answered. 'You' shows the percentage you got right, and 'class' shows the average for everyone who took the exam. + + Focus on the overall pattern, not just the specific numbers. If you did well on recall but struggled with applied questions, try practicing more problems. This is different from just reviewing the material. If your results are similar across all levels, it may be helpful to review the material itself. + ] + + // The figure comes before its bar in both tables on this page, so the numbers + // read as a column and the bars all start from the same left edge. + #table( + columns: (auto, 1fr, auto, 2.7cm, auto), + stroke: none, + align: (left + horizon, left + horizon, right + horizon, left + horizon, right + horizon), + inset: (x: 5pt, y: 6pt), + fill: (_, row) => if calc.odd(row) { luma(250) } else { white }, + table.header(th[LEVEL], th[WHAT IT ASKS FOR], th[YOU], th[], th[CLASS]), + ..cb-data + .levels + .map(level => ( + [*#level.name* #text(size: size-meta, fill: luma(120))[(#level.items)]], + text(size: size-small, fill: luma(80))[#level.blurb], + text(weight: "medium")[#pct(level.rate)], + bar(level.rate, color: rate-color(level.rate), width: 2.4cm), + { + let class-rate = level.at("class-rate", default: none) + if class-rate != none and show-comparison { + text(size: size-small, fill: luma(100))[#pct(class-rate)] + } else { [] } + }, + )) + .flatten(), + ) +] + +// ───────────────────────────────────────────────────────────────────────────── +// Objectives +// ───────────────────────────────────────────────────────────────────────────── + +#if cb-data.objectives.len() > 0 [ + = What the exam measured, objective by objective + + #explain[ + Each line is one learning objective, just as it appears in the syllabus. *Q* counts every question on this exam that measured any part of it, so a line usually rests on several questions rather than one. *You* shows the share you got right, and *class* shows the same for everyone else. + + The symbol in the first column is the summary. A #text(fill: thin-color, weight: "bold")[?] means this exam did not ask enough about that objective to say anything either way, which is a fact about the exam and not about you. Nothing on this page is broken down question by question; for that, see which lectures to go back to, and the notes on the ones you missed. + ] + + // The objective text is the only thing in this table that wants width, so it + // takes the free column and everything else is sized to its content. The + // thin-evidence badge that used to sit inline is gone: it repeated on every + // row of a three-page table, wrapped the text, and said no more than the `?` + // in the first column already says. + #table( + columns: (auto, 1fr, auto, auto, 2.4cm, auto), + stroke: none, + align: (center + horizon, left + horizon, right + horizon, right + horizon, left + horizon, right + horizon), + inset: (x: 5pt, y: 6pt), + fill: (_, row) => if calc.odd(row) { luma(250) } else { white }, + table.header(th[], th[OBJECTIVE], th[Q], th[YOU], th[], th[CLASS]), + ..cb-data + .objectives + .map(objective => ( + text(fill: status-color(objective.status), weight: "bold")[#objective.symbol], + text(size: size-small)[#markup(objective.text)], + text(size: size-small, fill: luma(110))[#objective.items], + text(size: size-small, weight: "medium")[#pct(objective.rate)], + bar(objective.rate, color: status-color(objective.status), width: 2.1cm), + { + let class-rate = objective.at("class-rate", default: none) + if class-rate != none and show-comparison { + text(size: size-small, fill: luma(100))[#pct(class-rate)] + } else { [] } + }, + )) + .flatten(), + ) + + #v(0.5em) + #text(size: size-micro, fill: luma(110))[ + #text(fill: ok-color, weight: "bold")[✓] you have this · + #text(fill: mid-color, weight: "bold")[~] getting there · + #text(fill: bad-color, weight: "bold")[✗] not yet · + #text(fill: thin-color, weight: "bold")[?] too few questions to say, which is + every line measured by one question. + A line counts as solid at #pct(threshold) or better. + ] +] + +// ───────────────────────────────────────────────────────────────────────────── +// Strengths and focus +// ───────────────────────────────────────────────────────────────────────────── + +// The two panels are the same object twice, so they are one function. `rows` is +// a list of (text, trailing) pairs; the trailing part is the rate, which only +// the focus panel carries. +#let panel(title, color, note, rows) = block( + width: 100%, + fill: color.lighten(93%), + radius: 4pt, + inset: 10pt, +)[ + #stack( + dir: ttb, + spacing: step, + text(weight: "bold", fill: color.darken(15%))[#title], + text(size: size-small, fill: luma(95))[#note], + { + set text(size: size-small) + list( + indent: 0pt, + body-indent: 0.45em, + spacing: step * 0.7, + marker: text(fill: color.darken(5%))[·], + ..rows, + ) + }, + ) +] +// ───────────────────────────────────────────────────────────────────────────── + +#let strengths = cb-data.at("strengths", default: ()) +#let focus = cb-data.at("focus", default: ()) + +#if strengths.len() > 0 or focus.len() > 0 [ + = Where your time will go furthest + + #explain[ + These two lists are based on the table above. I left out the single-question lines, so what is left has enough evidence to support action. + ] + + #grid( + columns: (1fr, 1fr), + gutter: 12pt, + if focus.len() > 0 { + panel( + "Start here", + bad-color, + [Start by reviewing the material that is likely to change the most and is the hardest for you.], + focus.map(objective => [ + #markup(objective.text) + #text(size: size-meta, fill: luma(110))[(#pct(objective.rate) of #objective.items)] + ]), + ) + } else { [] }, + if strengths.len() > 0 { + panel( + "Already solid", + ok-color, + [You showed these. Spend your review time elsewhere.], + strengths.map(objective => markup(objective.text)), + ) + } else { [] }, + ) +] + +// ───────────────────────────────────────────────────────────────────────────── +// Which lecture to go back to +// ───────────────────────────────────────────────────────────────────────────── + +// One ranked lecture. Three fixed columns: the rank, the body, the count. The +// body is a stack, so the title, the provenance line, and the objectives are one +// step apart and no paragraph contributes spacing of its own. +#let lecture-entry(index, lecture) = { + let count = lecture.at("targets-missed", default: 0) + let numbers = lecture.at("questions", default: ()) + let slides = lecture.at("slides", default: ()) + let targets = lecture.at("targets", default: ()) + + let parts = () + + parts.push({ + let url = lecture.at("url", default: none) + let title = text(weight: "bold")[#lecture.title] + [ + #(if url != none { link(url)[#title] } else { title }) + #text(size: size-meta, fill: luma(115))[(#lecture.lecture)] + ] + }) + + // Slides and question numbers were two separate lines, one of them reached by + // a linebreak and one by a block. Together on one line they are easier to skim + // and the entry loses a ragged gap. + let trail = () + if slides.len() > 0 { + trail.push(plural(slides.len(), "slide ", "slides ") + slides.map(str).join(", ")) + } + if numbers.len() > 0 { + trail.push( + "you missed " + plural(numbers.len(), "question ", "questions ") + numbers.map(str).join(", "), + ) + } + if trail.len() > 0 { + parts.push(text(size: size-meta, fill: luma(115))[#trail.join(" · ")]) + } + + // The learning targets, not the objectives they belong to. This section + // answers "what do I go and restudy", and a target is the grain that can be + // acted on: it names one performance rather than a whole claim. + // + // A real list rather than a middot glued to the front of a paragraph, so the + // second line of a long target indents under the first instead of running + // back to the margin. + if targets.len() > 0 { + parts.push({ + set text(size: size-small, fill: luma(80)) + list( + indent: 0pt, + body-indent: 0.45em, + spacing: step * 0.7, + marker: text(fill: luma(165))[·], + ..targets.map(target => markup(target)), + ) + }) + } + + block(breakable: false, above: entry-gap, width: 100%)[ + #grid( + columns: (1.35em, 1fr, 5.2em), + column-gutter: 0.7em, + align: (left + top, left + top, right + top), + rank(index + 1, if index == 0 { bad-color } else if count > 1 { mid-color } else { accent }), + pad(right: prose-pad-inset)[#stack(dir: ttb, spacing: step, ..parts)], + text(size: size-micro, fill: luma(105))[ + #count #plural(count, "target", "targets") + ], + ) + ] +} +// ───────────────────────────────────────────────────────────────────────────── + +#let review-lectures = cb-data.at("review-lectures", default: ()) + +#if show-lecture-plan and review-lectures.len() > 0 [ + = Which lectures to go back to + + #explain[ + Each question connects to the lecture it came from. Sorting the lectures by how many separate things went wrong in each gives you an order to work through, hardest first. Under each lecture are the specific skills the questions were testing, so you can go to the part of it you need rather than rewatching the whole thing. Several from one lecture usually means an early idea did not land, and fixing that one is the cheapest repair. + ] + + #for (index, lecture) in review-lectures.enumerate() [ + #lecture-entry(index, lecture) + ] +] + +// ───────────────────────────────────────────────────────────────────────────── +// Study plan +// ───────────────────────────────────────────────────────────────────────────── + +#let study = cb-data.at("study", default: ()) + +#if show-study and study.len() > 0 [ + = What to read + + #explain[ + Here are the sections from the course reading list that match the objectives above. Under each section, you'll find a brief sentence on what to focus on, so you don't need to reread the entire section. + ] + + #for group in study [ + #block(breakable: false, above: entry-gap)[ + === #markup(group.text) + + #for reading in group.readings [ + #block(inset: (left: 0.8em), above: step)[ + #{ + let url = reading.at("url", default: none) + let cite = text(weight: "bold")[#reading.citation] + let parts = ( + [ + #(if url != none { link(url)[#cite] } else { cite }) + #text(size: size-meta, fill: luma(110))[ + · #reading.lecture-title (#reading.lecture)#{ + if reading.at("supplemental", default: false) { ", optional" } + } + ] + ], + ) + let focus-note = reading.at("focus", default: none) + if focus-note != none { + parts.push(pad(right: prose-pad-inset)[ + #text(size: size-small)[#markup(focus-note)] + ]) + } + stack(dir: ttb, spacing: step * 0.6, ..parts) + } + ] + ] + ] + ] +] + +// ───────────────────────────────────────────────────────────────────────────── +// Question map +// ───────────────────────────────────────────────────────────────────────────── + +#let questions = cb-data.at("questions", default: ()) + +#let question-box(q) = { + let dropped = q.at("dropped", default: false) + let color = if dropped { luma(130) } else if q.at("blank", default: false) { + luma(160) + } else if q.at("correct", default: false) == true { + ok-color + } else if q.at("credit", default: 0.0) > 0.0 { mid-color } else { bad-color } + box( + width: 100%, + fill: color.lighten(85%), + stroke: 0.5pt + color.lighten(45%), + radius: 3pt, + inset: (x: 2pt, y: 4pt), + )[ + #align(center)[ + #text(size: size-small, weight: "bold", fill: color.darken(18%))[#q.number] + #{ + if dropped { + linebreak() + text(size: size-tag, fill: luma(110))[out] + } else { + let level = q.at("level", default: none) + if level != none { + linebreak() + text(size: size-tag, fill: luma(120))[L#level] + } + } + } + ] + ] +} + +#if show-question-map and questions.len() > 0 [ + #block(breakable: false)[ + = Question by question + + #explain[ + There is one box for each question, in the same order as on your exam paper. Each box is colored to show your result. The small *L* below each box shows the type of thinking required, matching the first table. The questions are not shown here, so use these numbers during office hours. + ] + + #grid( + columns: (1fr,) * 10, + column-gutter: 4pt, + row-gutter: 5pt, + ..questions.map(question-box), + ) + + #v(0.6em) + #text(size: size-micro, fill: luma(110))[ + #box(width: 0.7em, height: 0.7em, fill: ok-color.lighten(70%), radius: 2pt) right · + #box(width: 0.7em, height: 0.7em, fill: mid-color.lighten(70%), radius: 2pt) part marks · + #box(width: 0.7em, height: 0.7em, fill: bad-color.lighten(70%), radius: 2pt) not right · + #box(width: 0.7em, height: 0.7em, fill: luma(210), radius: 2pt) left blank · + #box(width: 0.7em, height: 0.7em, fill: luma(150), radius: 2pt) dropped, not scored + ] + ] +] + +// ───────────────────────────────────────────────────────────────────────────── +// Notes on what was missed +// ───────────────────────────────────────────────────────────────────────────── + +// One note. Four layers, in the order a student needs them: what the question +// was measuring, what went wrong, what to ask next time, and where to look it +// up. Each is a stack child, so every gap is one step and the provenance lines +// at the end sit tight together as a single footer. +#let note-entry(q) = { + let parts = () + + // Both tiers. The target is what this question actually asked, which is the + // thing to practise; the objective is the row it counted toward in the table + // above, which is how a student tells whether one slip cost them a claim. The + // objective is absent when the two would be the same sentence. + let measured = q.at("measured", default: ()) + if measured.len() > 0 { + parts.push({ + set text(fill: luma(21.57%)) + stack( + dir: ttb, + spacing: step * 0.45, + ..measured.map(m => { + let objective = m.at("objective", default: none) + let target = [ + #text(weight: "bold")[This question asked you to:] #markup(m.target) + ] + if objective == none { + target + } else { + stack( + dir: ttb, + spacing: step * 0.3, + target, + text(size: size-small, fill: luma(110))[ + Counts toward: #markup(objective) + ], + ) + } + }), + ) + }) + } + + // The diagnosis carries full body size. It is the sentence worth reading + // twice, and it was previously the same weight as the objective above it. + let feedback = q.at("feedback", default: none) + if feedback != none { parts.push(markup(feedback)) } + + let hint = q.at("hint", default: none) + if hint != none { parts.push(callout("Try this", markup(hint))) } + + // Present only with `--misconceptions`. This one is written to the instructor + // about the answer, which is why it reads differently. + let misconception = q.at("misconception", default: none) + if misconception != none { + parts.push(callout("The idea this option tests for", markup(misconception))) + } + + // Present only with `--solutions`. + let worked = q.at("worked", default: none) + if worked != none { + parts.push(block( + width: 100%, + fill: luma(249), + radius: 3pt, + inset: (x: 8pt, y: 7pt), + )[ + #stack( + dir: ttb, + spacing: step * 0.5, + text(size: size-meta, weight: "bold", fill: luma(85))[HOW IT WORKS OUT], + text(size: size-small)[#markup(worked)], + ) + ]) + } + + let trail = () + let taught = q.at("taught-in", default: ()) + if taught.len() > 0 { trail.push(meta-pair("Taught in", taught.join("; "))) } + + let review = q.at("review", default: ()) + if show-item-readings and review.len() > 0 { + let cites = review.map(reading => { + let url = reading.at("url", default: none) + let cite = reading.citation + if url != none { link(url)[#cite] } else { [#cite] } + }) + trail.push(meta-pair("Read again", cites.join([ · ]))) + } + if trail.len() > 0 { + parts.push(stack(dir: ttb, spacing: step * 0.4, ..trail)) + } + + block(breakable: false, above: entry-gap, width: 100%)[ + #grid( + columns: (1.8em, 1fr), + column-gutter: 0.7em, + align: (right + top, left + top), + text(weight: "bold", fill: accent)[#str(q.number)], + pad(right: prose-pad-inset)[#stack(dir: ttb, spacing: step, ..parts)], + ) + ] +} +// ───────────────────────────────────────────────────────────────────────────── + +#let missed = questions.filter(q => ( + q.at("credit", default: 0.0) < 0.999 + and ( + q.at("feedback", default: none) != none + or q.at("hint", default: none) != none + or q.at("worked", default: none) != none + or q.at("review", default: ()).len() > 0 + ) +)) + +#if show-feedback and missed.len() > 0 [ + = A closer look at the ones you missed + + #explain[ + Each note below explains the reasoning behind your answer instead of just giving the correct one. It helps you see where things went off track so you can avoid the same mistake next time. You can also use your paper to compare your answer with the question. + + If you see a line starting with *try this*, it suggests a question to ask yourself as you review the problem. If a reading is listed, it shows which section it came from. + ] + + #for q in missed [ + #note-entry(q) + ] +] + +// ───────────────────────────────────────────────────────────────────────────── +// Footer note +// ───────────────────────────────────────────────────────────────────────────── + +#v(1.3em) +#line(length: 100%, stroke: 0.5pt + luma(210)) +#v(0.45em) +#pad(right: prose-pad)[#text(size: size-micro, fill: luma(120))[ + Built on #cb-meta.generator.on by #cb-meta.generator.tool #cb-meta.generator.version, from the #str(n-tested) #plural(n-tested, "student", "students") who took this exam. Because a percentage based on one or two questions can be highly uncertain, bring anything here that looks wrong or surprising to office hours. That is the best use you can make of this page. +]] diff --git a/src/guide.rs b/src/guide.rs index b0b99a8..490a0aa 100644 --- a/src/guide.rs +++ b/src/guide.rs @@ -21,7 +21,9 @@ //! 3. [`first_exam`] runs one exam end to end: assemble, export, administer, //! ingest, analyze, report. //! 4. [`typst_export`] covers printed output, template markers, and render config. -//! 5. [`recipes`] holds short answers to specific questions, for when you already +//! 5. [`assignment`] publishes a homework to the course website, with solutions +//! gated behind a password. +//! 6. [`recipes`] holds short answers to specific questions, for when you already //! know the shape of the tool. //! //! ## Why the tutorials are in here rather than a wiki @@ -55,6 +57,10 @@ pub mod first_exam {} #[doc = include_str!("../docs/TYPST.md")] pub mod typst_export {} +/// Publishing an assignment to the website, with solutions gated behind a password. +#[doc = include_str!("../docs/guide/assignment.md")] +pub mod assignment {} + /// Short answers to specific questions. #[doc = include_str!("../docs/guide/recipes.md")] pub mod recipes {} diff --git a/src/lib.rs b/src/lib.rs index c1ff233..f8a485b 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -18,12 +18,19 @@ //! //! | File | Holds | Written by | //! |:--|:--|:--| -//! | `course.yaml` | identity, policy, objectives, lectures | you | +//! | `course.yaml` | identity, policy, units | you | +//! | `references.yaml` | the works the course cites | you | +//! | `lectures/*.yaml` | one lecture: readings, and what it teaches | you | +//! | `objectives/*.yaml` | one objective and its learning targets | you | //! | `banks/*.yaml` | items, with design intent and pooled statistics | you, then `calibrate` | //! | `assessments/*.yaml` | what was given, to whom, when | `assemble`, then you | //! | `data/*.parquet` | one row per student per item | `ingest` | //! -//! Three of the four are hand-editable YAML meant to be reviewed in a pull request. +//! The first four are one course file split by subject; `coursebank course build` +//! prints the merged result, and a course that keeps everything in `course.yaml` +//! still loads unchanged. See [`course::fragment`]. +//! +//! Most of these are hand-editable YAML meant to be reviewed in a pull request. //! Only the response data is machine-only, and it is stored in an open columnar //! format so pandas, polars, R, and DuckDB can all read it without this tool. //! @@ -95,22 +102,27 @@ pub mod data; pub mod error; pub mod export; pub mod guide; +pub mod migrate; pub mod model; pub mod util; -pub use util::{date, hash, markup, rng, yaml, zipfile}; +pub use util::{citation, date, hash, markup, rng, yaml, zipfile}; -pub use model::{assessment, bank, catalog, course, history, item, layout, taxonomy}; +pub use model::{ + assessment, bank, calibration, catalog, course, history, item, layout, seal, taxonomy, +}; + +pub use course::fragment; pub use authoring::{jsonschema, lint, select}; -#[cfg(feature = "parquet")] pub use data::store_parquet; -pub use data::{canvas, gradescope, responses, store}; +pub use data::{canvas, decode, gradescope, intake, responses, store}; -pub use analysis::{calibrate, classical, irt, students}; +pub use analysis::{calibrate, classical, diagnostic, irt, students}; -pub use export::{qti, report, typst}; +pub use export::site; +pub use export::{lecture, practice, qti, references, report, typst}; pub use catalog::Catalog; pub use course::{CourseFile, SCHEMA_VERSION}; diff --git a/src/migrate.rs b/src/migrate.rs new file mode 100644 index 0000000..4b901fd --- /dev/null +++ b/src/migrate.rs @@ -0,0 +1,2590 @@ +// SPDX-License-Identifier: Prosperity-3.0.0 +// Copyright Scientific Computing Studio +// Source: https://git.scient.ing/education/coursebank + +//! One-time conversions from one layout to the next. +//! +//! Everything here is a command you run once and then delete from your shell +//! history. It lives in the binary anyway, because the alternative is a script +//! in a gist that nobody can find in two years when a colleague clones the +//! course. +//! +//! # What there is +//! +//! | Function | Rewrites | Because | +//! |:--|:--|:--| +//! | [`split_plan`] | `course.yaml` into fragments | one file per lecture beats one file | +//! | [`qualified_ids`] | `item:` in records and seals | an id names the item, not its bank | +//! | [`store_ids`] | `item_ref` in the response store | so other readers see one id per item | +//! | [`options_plan`] and [`apply_options`] | option letters into names | a letter is a position, not an identity | +//! | [`stems`] | drops `version:` and `history:` | a stem's text is its identity | +//! | [`store_variants`] | fills the stored `variant` column | so other readers can group by option set | +//! | [`references`] | prose notes into citation fields | a DOI in a note is a link nobody can follow | +//! | [`order`] | removes the `order:` integers | objective order is derived, and targets have none | +//! | [`counters`] | drops the `-001` from item ids | a per-bank sequence number is not part of a name | +//! +//! # What the option migration does not touch +//! +//! Seals. A seal's digest covers the option ids it froze, so renaming them +//! would either invalidate the digest or require recomputing it — and a +//! tamper-evident record of an administration that has been quietly recomputed +//! is worth less than one that plainly says it predates the rename. A seal +//! keeps its letters, and [`crate::decode`] reads it as it stands. +//! +//! # Splitting the course file +//! +//! [`split_plan`] turns one `course.yaml` into a directory of fragments: the +//! bibliography into `references.yaml`, each lecture into `lectures/l-1-2.yaml`, +//! and each objective with its targets into `objectives/lo-....yaml`. +//! [`fragment::assemble`](crate::course::fragment::assemble) merges them back. +//! +//! ## Why this works on text +//! +//! The obvious implementation parses the course file, partitions the model, and +//! serializes each part. It is also the wrong one. A hand-written course file +//! carries section comments, folded block scalars, and deliberate quoting, and +//! `serde` round-tripping discards all three: five hundred lines of reading +//! prose come back as reflowed one-line scalars, and `# === L1.2` is simply +//! gone. The diff would be unreviewable, which for a migration is the whole +//! game. +//! +//! So the split copies *lines*. Fragments use the same section keys at the same +//! nesting depth as the file they came from, which means moving a lecture out is +//! a verbatim copy with no re-indentation, and every comment and scalar style +//! survives. Two edits are made to the copied text, both line-oriented: a +//! `teaches:` list is inserted into each lecture, and the `lectures:` lines that +//! [`fragment::assemble`](crate::course::fragment::assemble) now derives are +//! dropped. +//! +//! The scanner underneath is not a YAML parser and does not need to be. It finds +//! keys at an exact indentation, and a line at that indentation cannot be inside +//! a block scalar, because scalar content is indented deeper than the key that +//! introduces it. What it produces is checked the only way worth checking: the +//! written fragments are loaded back and compared against the original model by +//! [`differences`]. + +use std::collections::{BTreeMap, BTreeSet}; +use std::path::{Path, PathBuf}; + +use crate::course::fragment::REFERENCES_FILE; +use crate::course::{COURSE_FILE, CourseFile}; +use crate::error::{Error, Result}; +use crate::item::{Choice, Item}; +use crate::layout::Layout; +use crate::store::{self, Store}; +use crate::yaml; + +/// The width past which a flow sequence is written as a block list instead. +const FLOW_WIDTH: usize = 96; + +/// One file a split would write. +#[derive(Debug, Clone)] +pub struct PlannedFile { + /// Where it goes, relative to the course root. + pub path: PathBuf, + /// Its full contents. + pub text: String, + /// What it holds, for the command's output. + pub summary: String, +} + +/// What a split would do. +#[derive(Debug, Clone, Default)] +pub struct Plan { + /// The files to write, `course.yaml` first. + pub files: Vec, + /// Things worth saying out loud before writing. + pub notes: Vec, +} + +impl Plan { + /// The number of lines each planned file holds. + pub fn lines(&self) -> Report { + self.files + .iter() + .map(|f| (f.path.clone(), f.text.lines().count())) + .collect() + } +} + +/// Plans the split of a course directory's `course.yaml`. +/// +/// Nothing is written. The plan holds complete file contents, so the caller can +/// print it, write it, or throw it away. +/// +/// # Arguments +/// +/// * `root` - the course directory. +/// +/// # Returns +/// +/// The files a split would write. +/// +/// # Errors +/// +/// Returns [`Error::Io`] or [`Error::Yaml`] if `course.yaml` cannot be read or +/// parsed, and [`Error::Usage`] when the course is already split. +pub fn split_plan(root: &Path) -> Result { + let layout = Layout::new(root); + let course_path = layout.course_file(); + let course = CourseFile::load(&course_path)?; + + for dir in [layout.lectures(), layout.objectives()] { + if !yaml::list_yaml(&dir)?.is_empty() { + return Err(Error::usage(format!( + "{} already holds YAML files, so this course is already split", + dir.display() + ))); + } + } + if layout.references_file().exists() { + return Err(Error::usage(format!( + "{} already exists", + layout.references_file().display() + ))); + } + + let text = std::fs::read_to_string(&course_path).map_err(|e| Error::io(&course_path, e))?; + let lines: Vec = text.lines().map(str::to_string).collect(); + let sections = blocks(&lines, 0); + + let mut plan = Plan::default(); + + // Which objectives each lecture develops, and which targets each objective + // decomposes into. Both come from the parsed model rather than the text, + // because both are relations rather than syntax. + let mut teaches: BTreeMap<&str, Vec<&str>> = BTreeMap::new(); + for (id, objective) in &course.learning_objectives { + for lecture in &objective.lectures { + teaches + .entry(lecture.as_str()) + .or_default() + .push(id.as_str()); + } + } + let mut targets_of: BTreeMap<&str, Vec<&str>> = BTreeMap::new(); + for (id, target) in &course.learning_targets { + targets_of + .entry(target.objective.as_str()) + .or_default() + .push(id.as_str()); + } + + // 1. The root keeps identity, policy, units, and anything not moved. + let moved = [ + "references", + "lectures", + "learning_objectives", + "learning_targets", + ]; + let mut root_text = String::new(); + for section in §ions { + if moved.contains(§ion.key.as_str()) { + continue; + } + root_text.push_str(§ion.text()); + root_text.push_str("\n\n"); + } + plan.files.push(PlannedFile { + path: PathBuf::from(COURSE_FILE), + text: tidy(&root_text), + summary: "identity, policy, units".to_string(), + }); + + // 2. The bibliography. + if let Some(section) = sections.iter().find(|s| s.key == "references") { + plan.files.push(PlannedFile { + path: PathBuf::from(REFERENCES_FILE), + text: tidy(§ion.text()), + summary: format!("{} reference(s)", course.references.len()), + }); + } + + // 3. One file per lecture, with a `teaches:` list inserted. + if let Some(section) = sections.iter().find(|s| s.key == "lectures") { + for block in blocks(§ion.body(), 2) { + let id = block.key.clone(); + let listed = teaches.get(id.as_str()).cloned().unwrap_or_default(); + let mut body = block.lines.clone(); + if !listed.is_empty() { + body = with_teaches(&body, &listed); + } + let readings = course + .lectures + .get(&id) + .map(|l| l.readings.len()) + .unwrap_or(0); + plan.files.push(PlannedFile { + path: layout_relative("lectures", &lecture_slug(&id)), + text: tidy(&format!( + "lectures:\n{}\n", + join(&strip_group_header(&block.lead, &[id.as_str()]), &body) + )), + summary: format!("{id}: {} objective(s), {readings} reading(s)", listed.len()), + }); + } + } + + // 4. One file per objective, holding its targets. + let objective_blocks: BTreeMap = sections + .iter() + .find(|s| s.key == "learning_objectives") + .map(|s| { + blocks(&s.body(), 2) + .into_iter() + .map(|b| (b.key.clone(), b)) + .collect() + }) + .unwrap_or_default(); + let target_blocks: Vec = sections + .iter() + .find(|s| s.key == "learning_targets") + .map(|s| blocks(&s.body(), 2)) + .unwrap_or_default(); + + for (id, objective) in &course.learning_objectives { + let Some(block) = objective_blocks.get(id) else { + plan.notes.push(format!( + "objective `{id}` was not found in the text; skipped" + )); + continue; + }; + let lecture_labels: Vec<&str> = objective.lectures.iter().map(String::as_str).collect(); + let mut labels = lecture_labels.clone(); + labels.push(id.as_str()); + + // `lectures:` on the objective is derived from each lecture's `teaches:`, + // so carrying it here as well would be the duplication this move exists + // to remove. + let body = without_key(&block.lines, 4, "lectures"); + let mut text = format!( + "learning_objectives:\n{}\n", + join(&strip_group_header(&block.lead, &labels), &body) + ); + + // Taken from the text in the order it was written, not from the model in + // id order: a target's `order:` follows the sequence it was authored in, + // and alphabetizing 165 of them would scramble every file. + let mine: Vec<&Block> = target_blocks + .iter() + .filter(|b| { + course + .learning_targets + .get(&b.key) + .is_some_and(|t| t.objective == *id) + }) + .collect(); + + if !mine.is_empty() { + text.push_str("\nlearning_targets:\n"); + let objective_lectures: BTreeSet<&str> = + objective.lectures.iter().map(String::as_str).collect(); + for target_block in &mine { + let target = &course.learning_targets[target_block.key.as_str()]; + let own: BTreeSet<&str> = target.lectures.iter().map(String::as_str).collect(); + // A target inherits its objective's lectures, so an identical + // list is noise. A narrower one is a real claim and stays. + let body = if own == objective_lectures { + without_key(&target_block.lines, 4, "lectures") + } else { + target_block.lines.clone() + }; + text.push_str(&join( + &strip_group_header(&target_block.lead, &[id.as_str()]), + &body, + )); + text.push('\n'); + } + } + + let expected = targets_of.get(id.as_str()).map(Vec::len).unwrap_or(0); + if mine.len() != expected { + plan.notes.push(format!( + "objective `{id}`: {} of its {expected} target(s) were found in the text", + mine.len() + )); + } + + plan.files.push(PlannedFile { + path: layout_relative("objectives", id), + text: tidy(&text), + summary: format!("{id}: {} target(s)", mine.len()), + }); + } + + if !course.stimuli.is_empty() { + plan.notes.push(format!( + "{} stimulus/stimuli left in {COURSE_FILE}: which objective owns one is an \ + item-level question, so moving them waits for the per-objective banks", + course.stimuli.len() + )); + } + plan.notes.push( + "comments between two entries are attached to the entry below them, which is where a \ + `# === ...` group header belongs. Check any comment that trailed an entry." + .to_string(), + ); + + Ok(plan) +} + +/// Writes a plan, backing up the file it replaces. +/// +/// # Arguments +/// +/// * `root` - the course directory. +/// * `plan` - the plan to write. +/// +/// # Returns +/// +/// The paths written, plus the backup. +/// +/// # Errors +/// +/// Returns [`Error::Usage`] if a backup already exists, or [`Error::Io`] on a +/// write failure. +pub fn apply(root: &Path, plan: &Plan) -> Result> { + let backup = root.join(format!("{COURSE_FILE}.bak")); + if backup.exists() { + return Err(Error::usage(format!( + "{} already exists; move it aside first so a second run cannot overwrite the only \ + copy of the original", + backup.display() + ))); + } + std::fs::copy(root.join(COURSE_FILE), &backup).map_err(|e| Error::io(&backup, e))?; + + let mut written = vec![backup]; + for file in &plan.files { + let path = root.join(&file.path); + yaml::write_text(&path, &file.text)?; + written.push(path); + } + Ok(written) +} + +/// Compares two courses for everything a split is supposed to preserve. +/// +/// Lecture lists are compared as sets, since a derived list is ordered by the +/// files it was derived from, and `teaches` is ignored on the way in because +/// nothing declared it before the split. +/// +/// # Arguments +/// +/// * `before` - the course as it was. +/// * `after` - the course as reassembled from fragments. +/// +/// # Returns +/// +/// One message per difference, empty when the two agree. +/// +/// # Errors +/// +/// Returns [`Error::Other`] if either model cannot be serialized. +pub fn differences(before: &CourseFile, after: &CourseFile) -> Result> { + let old = projection(before)?; + let new = projection(after)?; + let mut out = Vec::new(); + + for (key, value) in &old { + match new.get(key) { + None => out.push(format!("{key} was lost")), + Some(other) if other != value => out.push(format!("{key} changed")), + Some(_) => {} + } + } + for key in new.keys() { + if !old.contains_key(key) { + out.push(format!("{key} appeared")); + } + } + Ok(out) +} + +/// A comparable, order-insensitive view of a course. +fn projection(course: &CourseFile) -> Result> { + let mut out = BTreeMap::new(); + out.insert("course".to_string(), yaml::to_string(&course.course)?); + out.insert("policy".to_string(), yaml::to_string(&course.policy)?); + out.insert("units".to_string(), yaml::to_string(&course.units)?); + out.insert("stimuli".to_string(), yaml::to_string(&course.stimuli)?); + + for (id, reference) in &course.references { + out.insert(format!("reference `{id}`"), yaml::to_string(reference)?); + } + for (id, lecture) in &course.lectures { + let mut lecture = lecture.clone(); + lecture.teaches.clear(); + out.insert(format!("lecture `{id}`"), yaml::to_string(&lecture)?); + } + for (id, objective) in &course.learning_objectives { + let mut objective = objective.clone(); + objective.lectures.sort(); + out.insert(format!("objective `{id}`"), yaml::to_string(&objective)?); + } + for (id, target) in &course.learning_targets { + let mut target = target.clone(); + target.lectures.sort(); + out.insert(format!("target `{id}`"), yaml::to_string(&target)?); + } + Ok(out) +} + +/// A file name for one fragment, relative to the course root. +fn layout_relative(dir: &str, stem: &str) -> PathBuf { + PathBuf::from(dir).join(format!("{stem}.yaml")) +} + +/// The file stem for a lecture id: `L1.2` becomes `l-1-2`. +/// +/// Matches how the bank and assessment files are already named, so a directory +/// listing sorts the way the course runs. +fn lecture_slug(id: &str) -> String { + let mut out = String::new(); + let mut previous: Option = None; + for ch in id.chars() { + if ch.is_ascii_alphanumeric() { + let boundary = + matches!(previous, Some(p) if p.is_ascii_alphabetic() != ch.is_ascii_alphabetic()); + if boundary && !out.ends_with('-') { + out.push('-'); + } + out.extend(ch.to_lowercase()); + } else if !out.is_empty() && !out.ends_with('-') { + out.push('-'); + } + previous = Some(ch); + } + out.trim_matches('-').to_string() +} + +/// Rewrites pre-2.0 bank-qualified item ids in the files that carry them. +/// +/// Textual, for the same reason [`split_plan`] is: an assessment record is +/// hand-edited and carries comments, and a serde round trip would drop them to +/// change one field. Only the value of an `item:` key is touched, so a +/// fingerprint or a note that happens to contain `::` is left alone. +/// +/// # Arguments +/// +/// * `root` - the course directory. +/// +/// # Returns +/// +/// One entry per file that would change, with how many ids it holds. +/// +/// # Errors +/// +/// Returns [`Error::Io`] when a directory or file cannot be read. +pub fn qualified_ids(root: &Path) -> Result> { + let layout = Layout::new(root); + let mut out = Vec::new(); + + for dir in [layout.assessments(), layout.seals()] { + for path in yaml::list_yaml(&dir)? { + let text = std::fs::read_to_string(&path).map_err(|e| Error::io(&path, e))?; + let rewritten = unqualify(&text); + if rewritten != text { + let n = text + .lines() + .zip(rewritten.lines()) + .filter(|(a, b)| a != b) + .count(); + let shown = path.strip_prefix(root).unwrap_or(&path).to_path_buf(); + out.push((shown, n, rewritten)); + } + } + } + Ok(out) +} + +/// Writes the rewritten files a call to [`qualified_ids`] produced. +/// +/// # Arguments +/// +/// * `root` - the course directory. +/// * `changes` - the plan. +/// +/// # Returns +/// +/// The paths written. +/// +/// # Errors +/// +/// Returns [`Error::Io`] on a write failure. +pub fn apply_ids(root: &Path, changes: &[(PathBuf, usize, String)]) -> Result> { + let mut written = Vec::new(); + for (relative, _, text) in changes { + let path = root.join(relative); + yaml::write_text(&path, text)?; + written.push(path); + } + Ok(written) +} + +/// Strips the bank qualifier from every `item:` value in a YAML document. +fn unqualify(text: &str) -> String { + let mut out: Vec = Vec::new(); + for line in text.lines() { + out.push(unqualify_line(line)); + } + let mut joined = out.join("\n"); + if text.ends_with('\n') { + joined.push('\n'); + } + joined +} + +/// Strips the bank qualifier from an `item:` value on one line. +/// +/// Handles both the block form and the flow form the records use: +/// +/// ```text +/// item: b-1-2::q-fastq-line -> item: q-fastq-line +/// { number: 1, item: "b1::q-a-001" } -> { number: 1, item: "q-a-001" } +/// ``` +fn unqualify_line(line: &str) -> String { + let Some(at) = line.find("item:") else { + return line.to_string(); + }; + // `learning_targets:` and `dropped_before_printing:` do not end in `item:`, + // but `item:` must be a whole key rather than a suffix of one. + let before = &line[..at]; + if before + .chars() + .next_back() + .is_some_and(|c| c.is_ascii_alphanumeric() || c == '_' || c == '-') + { + return line.to_string(); + } + + let rest = &line[at + "item:".len()..]; + let value_at = rest.len() - rest.trim_start().len(); + let value = rest.trim_start(); + let quote = value.starts_with(['"', '\'']); + let inner = if quote { &value[1..] } else { value }; + let end = inner + .find(|c: char| c == '"' || c == '\'' || c == ',' || c == '}' || c.is_whitespace()) + .unwrap_or(inner.len()); + let (id, tail) = inner.split_at(end); + match id.split_once(crate::item::LEGACY_QUALIFIER) { + None => line.to_string(), + Some((_, bare)) => format!( + "{}{}{}{}{}", + &line[..at + "item:".len()], + &rest[..value_at], + if quote { &value[..1] } else { "" }, + bare, + tail + ), + } +} + +/// Rewrites the item ids in the response store. +/// +/// The reader already canonicalizes ids in memory, so nothing depends on this +/// having been run. What it is for is the other readers: the store is Parquet +/// precisely so that pandas, DuckDB, and R can use it without this tool, and +/// those readers see whatever is actually in the column. A store holding both +/// `b-1-2::q-x` and `q-x` groups one question into two. +/// +/// Rows are rewritten flat, without a round trip through +/// [`crate::responses::Response`], so a column this migration is not about +/// cannot be reshaped on the way through. +/// +/// # Arguments +/// +/// * `root` - the course directory. +/// * `write` - whether to write, or only to count. +/// +/// # Returns +/// +/// One entry per file holding qualified ids, with how many rows carry them. +/// +/// # Errors +/// +/// Propagates read and write failures, including +/// [`Error::FeatureDisabled`] when the store is Parquet and the feature is off. +pub fn store_ids(root: &Path, write: bool) -> Result { + let layout = Layout::new(root); + let store = Store::open(layout.data())?; + let mut out = Vec::new(); + + for path in store.files()? { + let mut rows = store::read_flat(&path)?; + let mut touched = 0; + for row in &mut rows { + let canonical = crate::item::canonical_id(&row.item_ref); + if canonical != row.item_ref { + row.item_ref = canonical.to_string(); + touched += 1; + } + } + if touched == 0 { + continue; + } + if write { + store::write_flat(&path, &rows)?; + } + let shown = path.strip_prefix(root).unwrap_or(&path).to_path_buf(); + out.push((shown, touched)); + } + Ok(out) +} + +/// One file a migration changed, and how many things in it changed. +pub type Changed = (PathBuf, usize); + +/// What a migration changed, file by file. +pub type Report = Vec; + +/// Old id to new id, for the migrations that rename things. +pub type Renames = BTreeMap; + +/// The per-item map from a pre-2.0 option letter to the name that replaces it. +pub type OptionMap = BTreeMap>; + +/// Words an option id gains nothing from carrying. +const FUNCTION_WORDS: [&str; 40] = [ + "the", "a", "an", "is", "are", "was", "were", "of", "to", "for", "from", "in", "on", "at", + "and", "or", "but", "while", "that", "which", "it", "its", "they", "their", "this", "these", + "those", "be", "been", "with", "by", "as", "so", "than", "then", "each", "every", "one", + "about", "into", +]; + +/// How many words an option id takes from the option's text. +const SLUG_WORDS: usize = 5; + +/// How many characters of words an option id takes. +const SLUG_BUDGET: usize = 34; + +/// Derives an option id for every option of one item. +/// +/// The ids have to be readable, because they end up in a student report's +/// diagnostics, in `credit_overrides`, and in the `selected` column of every +/// response row — and they have to be unique within the item, because that is +/// what makes them an identity at all. +/// +/// Leading words give a readable id for most options. Where two of them come +/// out the same, each is extended with the first word that tells it apart from +/// the others it collided with, preferring a content word: three options that +/// all begin "its error probability falls to about one" end in `-half`, +/// `-tenth`, and `-hundredth`. +/// +/// # Arguments +/// +/// * `texts` - the option texts, in the order the item declares them. +/// +/// # Returns +/// +/// One id per option, in the same order. Empty when two options are so alike +/// that no word distinguishes them, which is a signal to look at the item +/// rather than to suffix a number onto it. +pub fn option_ids(texts: &[String]) -> Vec { + let words: Vec> = texts.iter().map(|t| slug_words(t)).collect(); + let mut stems: Vec = words.iter().map(|w| stem(w)).collect(); + + // Groups are taken before anything is extended: extending as we go would + // leave the last member of a group looking like the bare prefix its + // siblings were derived from. + let mut groups: BTreeMap> = BTreeMap::new(); + for (i, s) in stems.iter().enumerate() { + groups.entry(s.clone()).or_default().push(i); + } + for (_, group) in groups.iter().filter(|(_, g)| g.len() > 1) { + for &i in group { + let others: Vec<&Vec> = group + .iter() + .filter(|&&j| j != i) + .map(|&j| &words[j]) + .collect(); + if let Some(word) = distinguishing(&words[i], &others) { + stems[i] = format!("{}-{word}", stems[i]); + } + } + } + + if stems.iter().collect::>().len() != stems.len() { + return Vec::new(); + } + stems.iter().map(|s| format!("o-{s}")).collect() +} + +/// The words of an option's text, with markup and math flattened. +/// +/// Math is kept rather than stripped: two options that read `score $5$` and +/// `score $-1$` differ only inside the delimiters, and dropping them would make +/// the two ids identical. +fn slug_words(text: &str) -> Vec { + let mut flat = String::with_capacity(text.len()); + let mut in_math = false; + for ch in text.chars() { + match ch { + '$' => { + in_math = !in_math; + flat.push(' '); + } + '-' if in_math => flat.push_str(" minus "), + '`' | '*' | '_' => flat.push(' '), + c => flat.push(c.to_ascii_lowercase()), + } + } + flat.split(|c: char| !c.is_ascii_alphanumeric()) + .filter(|w| !w.is_empty()) + .map(str::to_string) + .collect() +} + +/// The leading words of an option, within the word and character budgets. +fn stem(words: &[String]) -> String { + let mut kept: Vec<&String> = Vec::new(); + let mut total = 0; + for word in words + .iter() + .skip_while(|w| matches!(w.as_str(), "the" | "a" | "an")) + { + if kept.len() == SLUG_WORDS || (total + word.len() > SLUG_BUDGET && !kept.is_empty()) { + break; + } + total += word.len() + 1; + kept.push(word); + } + // An id ending on `for` or `the` reads like a truncation, which it is. + while kept.len() > 1 && FUNCTION_WORDS.contains(&kept[kept.len() - 1].as_str()) { + kept.pop(); + } + if kept.is_empty() { + return "option".to_string(); + } + kept.iter() + .map(|w| w.as_str()) + .collect::>() + .join("-") +} + +/// The first word of `mine` that appears at that position in none of `others`. +fn distinguishing(mine: &[String], others: &[&Vec]) -> Option { + let differs = |at: usize, word: &str| { + others + .iter() + .all(|other| other.get(at).map(String::as_str) != Some(word)) + }; + mine.iter() + .enumerate() + .find(|(at, word)| !FUNCTION_WORDS.contains(&word.as_str()) && differs(*at, word)) + .or_else(|| { + mine.iter() + .enumerate() + .find(|(at, word)| differs(*at, word)) + }) + .map(|(_, word)| word.clone()) +} + +/// Derives the option ids for every item in every bank. +/// +/// # Arguments +/// +/// * `root` - the course directory. +/// +/// # Returns +/// +/// The map, and one message per item whose options could not be told apart. +/// +/// # Errors +/// +/// Propagates bank load failures. +pub fn options_plan(root: &Path) -> Result<(OptionMap, Vec)> { + let layout = Layout::new(root); + let mut map: OptionMap = BTreeMap::new(); + let mut problems = Vec::new(); + + for path in yaml::list_yaml(&layout.banks())? { + let bank: crate::bank::BankFile = yaml::read(&path)?; + for item in &bank.items { + let legacy: Vec<&Choice> = item + .options + .iter() + .filter(|o| Item::is_legacy_option_id(&o.id)) + .collect(); + if legacy.is_empty() { + continue; + } + if legacy.len() != item.options.len() { + problems.push(format!( + "{}: `{}` has both named and lettered options; name the rest by hand", + path.display(), + item.id + )); + continue; + } + let texts: Vec = item.options.iter().map(|o| o.text.clone()).collect(); + let ids = option_ids(&texts); + if ids.is_empty() { + problems.push(format!( + "{}: `{}` has options no word tells apart, so no id can be derived. Name \\ + them by hand, or look again at whether they are two readings of one option.", + path.display(), + item.id + )); + continue; + } + let entry: BTreeMap = + item.options.iter().map(|o| o.id.clone()).zip(ids).collect(); + map.insert(item.id.clone(), entry); + } + } + Ok((map, problems)) +} + +/// Rewrites option letters as names, in every file that carries one. +/// +/// Banks and assessment records are rewritten textually, for the reason +/// [`split_plan`] is: both are hand-edited and full of prose. The response +/// store is rewritten flat. Seals are *not* rewritten — see the note in the +/// module documentation. +/// +/// # Arguments +/// +/// * `root` - the course directory. +/// * `map` - the plan from [`options_plan`]. +/// * `write` - whether to write, or only to count. +/// +/// # Returns +/// +/// One entry per file that changes, with how many substitutions it holds. +/// +/// # Errors +/// +/// Propagates read and write failures. +pub fn apply_options(root: &Path, map: &OptionMap, write: bool) -> Result { + let layout = Layout::new(root); + let mut out = Vec::new(); + + for path in yaml::list_yaml(&layout.banks())? { + let text = std::fs::read_to_string(&path).map_err(|e| Error::io(&path, e))?; + let (rewritten, n) = rename_in_bank(&text, map); + if n > 0 { + if write { + yaml::write_text(&path, &rewritten)?; + } + out.push((relative(root, &path), n)); + } + } + + for path in yaml::list_yaml(&layout.assessments())? { + let text = std::fs::read_to_string(&path).map_err(|e| Error::io(&path, e))?; + let (rewritten, n) = rename_in_record(&text, map); + if n > 0 { + if write { + yaml::write_text(&path, &rewritten)?; + } + out.push((relative(root, &path), n)); + } + } + + let store = Store::open(layout.data())?; + for path in store.files()? { + let mut rows = store::read_flat(&path)?; + let mut touched = 0; + for row in &mut rows { + let Some(item) = map.get(crate::item::canonical_id(&row.item_ref)) else { + continue; + }; + for column in [&mut row.selected, &mut row.eliminated] { + let renamed = rename_list(column, item); + if renamed != *column { + *column = renamed; + touched += 1; + } + } + } + if touched > 0 { + if write { + store::write_flat(&path, &rows)?; + } + out.push((relative(root, &path), touched)); + } + } + + Ok(out) +} + +/// A path shown relative to the course root. +fn relative(root: &Path, path: &Path) -> PathBuf { + path.strip_prefix(root).unwrap_or(path).to_path_buf() +} + +/// Renames the semicolon-joined option ids a stored response column holds. +fn rename_list(column: &str, item: &BTreeMap) -> String { + column + .split(';') + .map(|part| match item.get(part.trim()) { + Some(renamed) => renamed.as_str(), + None => part, + }) + .collect::>() + .join(";") +} + +/// Rewrites the `- id:` lines of every option in a bank file. +/// +/// Which `- id:` lines those are is decided by the plan rather than by +/// indentation: a value that names an item in the plan sets the item, and a +/// value the current item maps is an option of it. So a bank indented +/// unusually still migrates, and nothing else called `id` can be caught by +/// accident. +fn rename_in_bank(text: &str, map: &OptionMap) -> (String, usize) { + let mut out: Vec = Vec::new(); + let mut current: Option<&BTreeMap> = None; + let mut n = 0; + + for line in text.lines() { + let value = list_id(line); + if let Some(value) = value { + if let Some(item) = map.get(value) { + current = Some(item); + out.push(line.to_string()); + continue; + } + if let Some(renamed) = current.and_then(|item| item.get(value)) { + out.push(line.replace(value, renamed)); + n += 1; + continue; + } + } + out.push(line.to_string()); + } + + let mut joined = out.join("\n"); + if text.ends_with('\n') { + joined.push('\n'); + } + (joined, n) +} + +/// The value of a `- id: value` line, if the line is one. +fn list_id(line: &str) -> Option<&str> { + let trimmed = line.trim_start(); + let rest = trimmed.strip_prefix("- id:")?; + let value = rest.trim().trim_matches(['\'', '"']); + if value.is_empty() { None } else { Some(value) } +} + +/// Rewrites `key:` and `credit_overrides:` in an assessment record. +/// +/// Both belong to the placement whose `item:` was seen most recently, which is +/// how the records are written: the item and its key sit in one block, or on +/// one line. +fn rename_in_record(text: &str, map: &OptionMap) -> (String, usize) { + let mut out: Vec = Vec::new(); + let mut current: Option<&BTreeMap> = None; + let mut n = 0; + + for line in text.lines() { + if let Some(item) = item_on_line(line) { + current = map.get(crate::item::canonical_id(item)); + } + let Some(item) = current else { + out.push(line.to_string()); + continue; + }; + let (rewritten, count) = rename_bracketed(line, "key:", item); + let (rewritten, more) = rename_braced(&rewritten, "credit_overrides:", item); + n += count + more; + out.push(rewritten); + } + + let mut joined = out.join("\n"); + if text.ends_with('\n') { + joined.push('\n'); + } + (joined, n) +} + +/// The `item:` value on a line, if it carries one. +fn item_on_line(line: &str) -> Option<&str> { + let at = line.find("item:")?; + let before = &line[..at]; + if before + .chars() + .next_back() + .is_some_and(|c| c.is_ascii_alphanumeric() || c == '_' || c == '-') + { + return None; + } + let rest = line[at + "item:".len()..].trim_start(); + let quote = rest.starts_with(['"', '\'']); + let inner = if quote { &rest[1..] } else { rest }; + let end = inner + .find(|c: char| c == '"' || c == '\'' || c == ',' || c == '}' || c.is_whitespace()) + .unwrap_or(inner.len()); + Some(&inner[..end]).filter(|v| !v.is_empty()) +} + +/// Renames the members of a `key: [A, B]` list. +fn rename_bracketed(line: &str, key: &str, item: &BTreeMap) -> (String, usize) { + let Some(at) = key_at_word(line, key) else { + return (line.to_string(), 0); + }; + let rest = &line[at + key.len()..]; + let Some(open) = rest.find('[') else { + return (line.to_string(), 0); + }; + let Some(close) = rest[open..].find(']') else { + return (line.to_string(), 0); + }; + let inside = &rest[open + 1..open + close]; + let mut n = 0; + let renamed: Vec = inside + .split(',') + .map(|part| { + let trimmed = part.trim().trim_matches(['\'', '"']); + match item.get(trimmed) { + Some(new) => { + n += 1; + new.clone() + } + None => part.trim().to_string(), + } + }) + .collect(); + if n == 0 { + return (line.to_string(), 0); + } + ( + format!( + "{}[{}]{}", + &line[..at + key.len() + open], + renamed.join(", "), + &rest[open + close + 1..] + ), + n, + ) +} + +/// Renames the keys of a `credit_overrides: { A: 0.5 }` mapping. +fn rename_braced(line: &str, key: &str, item: &BTreeMap) -> (String, usize) { + let Some(at) = key_at_word(line, key) else { + return (line.to_string(), 0); + }; + let rest = &line[at + key.len()..]; + let Some(open) = rest.find('{') else { + return (line.to_string(), 0); + }; + let Some(close) = rest[open..].find('}') else { + return (line.to_string(), 0); + }; + let inside = &rest[open + 1..open + close]; + let mut n = 0; + let renamed: Vec = inside + .split(',') + .map(|pair| match pair.split_once(':') { + Some((name, value)) => { + let trimmed = name.trim().trim_matches(['\'', '"']); + match item.get(trimmed) { + Some(new) => { + n += 1; + format!("{new}: {}", value.trim()) + } + None => pair.trim().to_string(), + } + } + None => pair.trim().to_string(), + }) + .collect(); + if n == 0 { + return (line.to_string(), 0); + } + ( + format!( + "{}{{ {} }}{}", + &line[..at + key.len() + open], + renamed.join(", "), + &rest[open + close + 1..] + ), + n, + ) +} + +/// Where a key appears as a whole key rather than as the tail of a longer one. +fn key_at_word(line: &str, key: &str) -> Option { + let at = line.find(key)?; + let before = &line[..at]; + if before + .chars() + .next_back() + .is_some_and(|c| c.is_ascii_alphanumeric() || c == '_' || c == '-') + { + return None; + } + Some(at) +} + +/// Drops the fields that a version number used to carry. +/// +/// `version:` on an item, `version:` on a placement, and the whole `history:` +/// block. All three still load and are ignored, so this is tidying rather than +/// repair — but a field that is read and thrown away is a field the next person +/// will keep maintaining. +/// +/// Textual, and deliberately narrow: only a `version:` key at the depth an item +/// or a placement declares one, and only a `history:` block and the lines +/// indented under it. A `version` inside a stem, a note, or an option's text is +/// not a key and is left alone. +/// +/// # Arguments +/// +/// * `root` - the course directory. +/// * `write` - whether to write, or only to count. +/// +/// # Returns +/// +/// One entry per file that changes, with how many lines it drops. +/// +/// # Errors +/// +/// Propagates read and write failures. +pub fn stems(root: &Path, write: bool) -> Result { + let layout = Layout::new(root); + let mut out = Vec::new(); + + for dir in [layout.banks(), layout.assessments(), layout.seals()] { + for path in yaml::list_yaml(&dir)? { + let text = std::fs::read_to_string(&path).map_err(|e| Error::io(&path, e))?; + let rewritten = drop_versioning(&text); + if rewritten == text { + continue; + } + let dropped = text.lines().count() - rewritten.lines().count(); + if write { + yaml::write_text(&path, &rewritten)?; + } + out.push((relative(root, &path), dropped)); + } + } + Ok(out) +} + +/// Removes `version:` keys and `history:` blocks from a YAML document. +fn drop_versioning(text: &str) -> String { + let mut out: Vec<&str> = Vec::new(); + let mut skipping: Option = None; + + for line in text.lines() { + if let Some(depth) = skipping { + match indent_of(line) { + Some(n) if n > depth => continue, + None => continue, + _ => skipping = None, + } + } + let indent = match indent_of(line) { + Some(n) => n, + None => { + out.push(line); + continue; + } + }; + // `- version: 2` is a list item, which `history:` entries are; the key + // form is what an item or a placement writes. + let trimmed = line.trim_start().trim_start_matches("- "); + if trimmed == "history:" || trimmed.starts_with("history:") { + skipping = Some(indent); + continue; + } + if is_key(trimmed, "version") { + continue; + } + out.push(line); + } + + let mut joined = out.join("\n"); + if text.ends_with('\n') { + joined.push('\n'); + } + joined +} + +/// Whether a trimmed line declares exactly this key. +fn is_key(trimmed: &str, key: &str) -> bool { + match trimmed.strip_prefix(key) { + Some(rest) => rest.starts_with(':'), + None => false, + } +} + +/// Fills in the stored `variant` column from the assessment records. +/// +/// Nothing in the tool needs this: the reader derives a variant from the +/// placement when the row has none. It is for the other readers. The store is +/// Parquet precisely so pandas, DuckDB, and R can use it without this tool, and +/// a store that only carries `item_ref` cannot tell two option sets of one stem +/// apart — it will average them into one item that never existed. +/// +/// # Arguments +/// +/// * `root` - the course directory. +/// * `write` - whether to write, or only to count. +/// +/// # Returns +/// +/// One entry per file that gains variants, with how many rows it gains. +/// +/// # Errors +/// +/// Propagates catalog, record, and store failures. +pub fn store_variants(root: &Path, write: bool) -> Result { + let layout = Layout::new(root); + let catalog = crate::catalog::Catalog::load(root)?; + + // (assessment, item number) -> variant. Keyed on the number rather than the + // item id: a record may place one item twice, and the row knows which. + let mut by_question: BTreeMap<(String, u32), String> = BTreeMap::new(); + for record in crate::assessment::AssessmentFile::load_all(&layout.assessments())? { + for placement in &record.items { + let Some(entry) = catalog.get(&placement.item) else { + continue; + }; + by_question.insert( + (record.assessment.id.clone(), placement.number), + placement.variant_of(&entry.item), + ); + } + } + + let store = Store::open(layout.data())?; + let mut out = Vec::new(); + for path in store.files()? { + let mut rows = store::read_flat(&path)?; + let mut touched = 0; + for row in &mut rows { + if !row.variant.is_empty() { + continue; + } + let key = (row.assessment_id.clone(), row.item_number); + if let Some(variant) = by_question.get(&key) { + row.variant = variant.clone(); + touched += 1; + } + } + if touched > 0 { + if write { + store::write_flat(&path, &rows)?; + } + out.push((relative(root, &path), touched)); + } + } + Ok(out) +} + +/// Drops the trailing counter from every item id. +/// +/// `q-fastq-quality-line-001` becomes `q-fastq-quality-line`. The counter was a +/// per-bank sequence number, which says when an item was written — something git +/// knows — and collides with another bank's numbering the moment an item moves. +/// +/// This is a rename, not a normalization, and that distinction decides how it +/// has to be done. Stripping `bank::` from an id recovered a name that was +/// already inside it, so an old reference could simply be read as the new one. +/// A counter carries no such fallback: `q-x-001` and `q-x` are two unrelated +/// strings, and every file that names one has to be rewritten in the same pass +/// or the join to four terms of response data quietly breaks. +/// +/// So: banks, assessment records, seals, and the response store, or nothing. +/// Two ids that would collide after stripping abort the whole migration rather +/// than merging two questions into one. +/// +/// # Arguments +/// +/// * `root` - the course directory. +/// * `write` - whether to write, or only to count. +/// +/// # Returns +/// +/// The renames, and one entry per file that changes with how many references it +/// holds. +/// +/// # Errors +/// +/// Returns [`Error::Invalid`] when two ids would collide, listing both, and +/// propagates load and write failures. +pub fn counters(root: &Path, write: bool) -> Result<(Renames, Report)> { + let layout = Layout::new(root); + let catalog = crate::catalog::Catalog::load(root)?; + + let mut rename: Renames = BTreeMap::new(); + let mut taken: Renames = BTreeMap::new(); + let mut collisions = Vec::new(); + for entry in &catalog.entries { + let Some(bare) = strip_counter(&entry.uid) else { + taken.insert(entry.uid.clone(), entry.uid.clone()); + continue; + }; + if let Some(first) = taken.get(&bare) { + collisions.push(format!( + "`{}` and `{first}` would both become `{bare}`. Give one of them a name that \ + says what it asks rather than when it was written.", + entry.uid + )); + continue; + } + taken.insert(bare.clone(), entry.uid.clone()); + rename.insert(entry.uid.clone(), bare); + } + if !collisions.is_empty() { + return Err(Error::Invalid(collisions)); + } + if rename.is_empty() { + return Ok((rename, Vec::new())); + } + + let mut touched = Vec::new(); + + // Banks: the definition, and any `supersedes` pointing at one. + for path in yaml::list_yaml(&layout.banks())? { + let text = std::fs::read_to_string(&path).map_err(|e| Error::io(&path, e))?; + let (out, n) = rename_ids(&text, &rename, &["- id", "supersedes"]); + if n > 0 { + if write { + yaml::write_text(&path, &out)?; + } + touched.push((relative(root, &path), n)); + } + } + + // Records: the `item:` on every placement. + for path in yaml::list_yaml(&layout.assessments())? { + let text = std::fs::read_to_string(&path).map_err(|e| Error::io(&path, e))?; + let (out, n) = rename_ids(&text, &rename, &["item"]); + if n > 0 { + if write { + yaml::write_text(&path, &out)?; + } + touched.push((relative(root, &path), n)); + } + } + + // Seals go through the model, not the text, because the ids are inside the + // digest and a seal that has been rewritten has to say so. + for path in yaml::list_yaml(&layout.seals())? { + let mut seal = crate::seal::SealFile::load(&path)?; + let n = seal.rename_items(&rename); + if n > 0 { + if write { + seal.save(&path)?; + } + touched.push((relative(root, &path), n)); + } + } + + // The store, where the id is the join key. + let store = Store::open(layout.data())?; + for path in store.files()? { + let mut rows = store::read_flat(&path)?; + let mut n = 0; + for row in &mut rows { + if let Some(new) = rename.get(&row.item_ref) { + row.item_ref = new.clone(); + n += 1; + } + } + if n > 0 { + if write { + store::write_flat(&path, &rows)?; + } + touched.push((relative(root, &path), n)); + } + } + + Ok((rename, touched)) +} + +/// An id with a trailing `-123` removed, or `None` when it has none. +fn strip_counter(id: &str) -> Option { + let (head, tail) = id.rsplit_once('-')?; + if head.is_empty() || tail.is_empty() || !tail.chars().all(|c| c.is_ascii_digit()) { + return None; + } + Some(head.to_string()) +} + +/// Rewrites the value of the named keys wherever it is an id being renamed. +/// +/// Keyed on the field name so that an id appearing in prose — a rationale that +/// mentions the item it replaced, a note — is left alone. A rename that edited +/// every matching string in the file would also edit the sentences about it. +fn rename_ids(text: &str, rename: &Renames, keys: &[&str]) -> (String, usize) { + let mut out: Vec = Vec::new(); + let mut count = 0; + + for line in text.lines() { + let mut replaced = None; + for key in keys { + let Some(at) = field_value(line, key) else { + continue; + }; + let value = &line[at.0..at.1]; + // Canonicalized before the lookup, so a record that still carries a + // pre-2.0 `bank::` qualifier is renamed rather than skipped. Left + // alone it would keep pointing at an id the bank no longer has. + let Some(new) = rename.get(crate::item::canonical_id(value)) else { + continue; + }; + replaced = Some(format!("{}{new}{}", &line[..at.0], &line[at.1..])); + count += 1; + break; + } + out.push(replaced.unwrap_or_else(|| line.to_string())); + } + + let mut joined = out.join("\n"); + if text.ends_with('\n') { + joined.push('\n'); + } + (joined, count) +} + +/// The byte range of a `key: value` scalar on one line, quotes excluded. +fn field_value(line: &str, key: &str) -> Option<(usize, usize)> { + let needle = format!("{key}:"); + let at = line.find(&needle)?; + let before = &line[..at]; + if before.chars().next_back().is_some_and(|c| { + c.is_ascii_alphanumeric() || c == '_' || (c == '-' && !key.starts_with('-')) + }) { + return None; + } + let rest = &line[at + needle.len()..]; + let lead = rest.len() - rest.trim_start().len(); + let value = rest.trim_start(); + let quoted = value.starts_with(['"', '\'']); + let start = at + needle.len() + lead + usize::from(quoted); + let inner = &line[start..]; + let end = inner + .find(|c: char| { + c == '"' || c == '\'' || c == ',' || c == '}' || c == ']' || c.is_whitespace() + }) + .map(|n| start + n) + .unwrap_or(line.len()); + (end > start).then_some((start, end)) +} + +/// Removes the hand-kept `order:` integers. +/// +/// Nothing is written in their place. An objective's teaching order is its +/// position in a lecture's `teaches` list, which `migrate split` already +/// produced. A target has no order at all — an objective's targets are a set of +/// question templates, sampled from rather than worked through — so the integer +/// was asserting a sequence that was never taught. Where a list has to be +/// printed, [`crate::course::CourseFile::targets`] orders it by ceiling. +/// +/// Two hundred and four integers in a real course, each of which had to be +/// bumped by hand when a target was inserted in the middle. +/// +/// # Arguments +/// +/// * `root` - the course directory. +/// * `write` - whether to write, or only to count. +/// +/// # Returns +/// +/// One entry per file that changes, with how many `order:` lines it loses. +/// +/// # Errors +/// +/// Propagates catalog and write failures, and refuses when the lectures do not +/// declare what they teach, since objective order would then have no source. +pub fn order(root: &Path, write: bool) -> Result { + let layout = Layout::new(root); + let course = CourseFile::load_dir(root)?; + + if course.lectures.values().all(|l| l.teaches.is_empty()) + && !course.learning_objectives.is_empty() + { + return Err(Error::usage( + "no lecture declares what it teaches, so objective order would have nothing to come from. Run `coursebank migrate split` first." + .to_string(), + )); + } + + let mut out = Vec::new(); + let mut files = vec![layout.course_file()]; + files.extend(yaml::list_yaml(&layout.objectives())?); + for path in files { + if !path.is_file() { + continue; + } + let text = std::fs::read_to_string(&path).map_err(|e| Error::io(&path, e))?; + let (rewritten, dropped) = rewrite_order(&text); + if rewritten == text { + continue; + } + if write { + yaml::write_text(&path, &rewritten)?; + } + out.push((relative(root, &path), dropped)); + } + Ok(out) +} + +/// Removes every `order:` line. +/// +/// Nothing replaces them. An objective's position comes from where its lecture +/// lists it in `teaches`, and a target has no position to derive: an +/// objective's targets are a set of question templates rather than steps in a +/// sequence, so the integer was asserting an order that was never taught. +fn rewrite_order(text: &str) -> (String, usize) { + let mut out: Vec<&str> = Vec::new(); + let mut dropped = 0; + + for line in text.lines() { + if key_at(line, 4).as_deref() == Some("order") { + dropped += 1; + continue; + } + out.push(line); + } + + let mut joined = out.join("\n"); + if text.ends_with('\n') { + joined.push('\n'); + } + (joined, dropped) +} + +/// Turns the citations written into `note:` fields into real fields. +/// +/// A hand-built bibliography collects entries whose journal, volume, pages, and +/// DOI are all sitting in the one field that means "anything else worth saying". +/// Nothing can use them there: a reading list cannot link a DOI it cannot see, +/// and an export to Hayagriva or BibTeX has no journal to put in `parent` or +/// `journal`. +/// +/// Textual, like the rest of this module, and for a reason specific to this +/// file: a bibliography is usually ordered the way its author thinks about it — +/// books, then the papers that matter — and the model holds it in a map, so a +/// serde round trip would alphabetize all of it to change twenty entries. +/// +/// A field the entry already declares is never overwritten. Where the note +/// disagrees with it, the note's version is left in place for a person to look +/// at rather than silently replacing something that was typed deliberately. +/// +/// # Arguments +/// +/// * `root` - the course directory. +/// * `write` - whether to write, or only to count. +/// +/// # Returns +/// +/// One entry per file that changes, with how many notes were taken apart, and +/// one message per disagreement found. +/// +/// # Errors +/// +/// Propagates read and write failures. +pub fn references(root: &Path, write: bool) -> Result<(Report, Vec)> { + let layout = Layout::new(root); + let mut touched = Vec::new(); + let mut notes = Vec::new(); + + // Either file may hold the bibliography: its own, or the course file that + // has not been split yet. + for path in [layout.references_file(), layout.course_file()] { + if !path.is_file() { + continue; + } + let text = std::fs::read_to_string(&path).map_err(|e| Error::io(&path, e))?; + let (rewritten, count, said) = expand_notes(&text); + notes.extend(said); + if count > 0 { + if write { + yaml::write_text(&path, &rewritten)?; + } + touched.push((relative(root, &path), count)); + } + } + Ok((touched, notes)) +} + +/// Replaces each reference's `note:` line with the fields it was carrying. +fn expand_notes(text: &str) -> (String, usize, Vec) { + let lines: Vec = text.lines().map(str::to_string).collect(); + let Some(section) = blocks(&lines, 0) + .into_iter() + .find(|b| b.key == "references") + else { + return (text.to_string(), 0, Vec::new()); + }; + + // Which lines belong to which entry, so a note is parsed with its own year + // and checked against its own fields. + let mut rewritten: BTreeMap> = BTreeMap::new(); + let mut said = Vec::new(); + let mut count = 0; + + let start = lines.len() - section.lines.len(); + for entry in blocks(§ion.body(), 2) { + let offset = lines + .iter() + .enumerate() + .skip(start) + .find(|(_, l)| **l == entry.lines[0]) + .map(|(i, _)| i); + let Some(offset) = offset else { continue }; + + let Some((note_at, note)) = entry + .lines + .iter() + .enumerate() + .find_map(|(i, l)| scalar(l, 4, "note").map(|v| (i, v))) + else { + continue; + }; + + let year = entry + .lines + .iter() + .find_map(|l| scalar(l, 4, "year")) + .and_then(|v| v.parse::().ok()); + let parsed = crate::citation::parse(¬e, year); + if parsed.container.is_none() && parsed.doi.is_none() { + continue; + } + + let declared = |field: &str| entry.lines.iter().any(|l| scalar(l, 4, field).is_some()); + let mut out: Vec = Vec::new(); + for (field, value, quote) in [ + ("container", parsed.container.clone(), false), + ("volume", parsed.volume.clone(), true), + ("pages", parsed.pages.clone(), true), + ("doi", parsed.doi.clone(), true), + ] { + let Some(value) = value else { continue }; + if declared(field) { + said.push(format!( + "`{}` already declares {field}; the note's `{value}` was left in place", + entry.key + )); + continue; + } + out.push(format!(" {field}: {}", render(&value, quote))); + } + if let Some(note) = &parsed.note { + out.push(format!(" note: {}", render(note, true))); + } + + // A note whose every part was already declared is not a change. + if out.is_empty() { + continue; + } + if said + .iter() + .any(|m| m.starts_with(&format!("`{}`", entry.key))) + && parsed.note.is_some() + && out.len() == 1 + { + continue; + } + rewritten.insert(offset + note_at, out); + count += 1; + } + + if count == 0 { + return (text.to_string(), 0, said); + } + + let mut out: Vec = Vec::new(); + for (i, line) in lines.iter().enumerate() { + match rewritten.get(&i) { + Some(replacement) => out.extend(replacement.clone()), + None => out.push(line.clone()), + } + } + let mut joined = out.join("\n"); + if text.ends_with('\n') { + joined.push('\n'); + } + (joined, count, said) +} + +/// The value of a `key: value` line at an exact indentation, unquoted. +fn scalar(line: &str, indent: usize, key: &str) -> Option { + if key_at(line, indent).as_deref() != Some(key) { + return None; + } + let value = line[indent + key.len() + 1..].trim(); + if value.is_empty() { + return None; + } + for quote in ['\'', '"'] { + if let Some(inner) = value + .strip_prefix(quote) + .and_then(|v| v.strip_suffix(quote)) + { + return Some(inner.replace("''", "'")); + } + } + Some(value.to_string()) +} + +/// A YAML scalar, quoted when it has to be. +/// +/// Numbers are quoted whether they need it or not: `volume: 48` reads back as an +/// integer and the field is a string, so the file would stop loading. +fn render(value: &str, quote: bool) -> String { + let risky = quote + || value.contains(": ") + || value.ends_with(':') + || value.contains(" #") + || value.starts_with([ + '[', '{', '&', '*', '!', '|', '>', '%', '@', '`', '\'', '"', '-', + ]); + if risky { + format!("'{}'", value.replace('\'', "''")) + } else { + value.to_string() + } +} + +/// One block of YAML: a key, the lines under it, and the comments above it. +#[derive(Debug, Clone, Default)] +struct Block { + /// The key, unquoted. + key: String, + /// The key line and everything indented under it, verbatim. + lines: Vec, + /// Comment lines that sat directly above the key. + lead: Vec, +} + +impl Block { + /// The block as text, comments first. + fn text(&self) -> String { + join(&self.lead, &self.lines) + } + + /// The lines under the key, without the key line. + fn body(&self) -> Vec { + self.lines.iter().skip(1).cloned().collect() + } +} + +/// Splits lines into the blocks keyed at one indentation. +fn blocks(lines: &[String], indent: usize) -> Vec { + let mut out: Vec = Vec::new(); + let mut current: Option = None; + let mut lead: Vec = Vec::new(); + + for line in lines { + if let Some(key) = key_at(line, indent) { + if let Some(mut block) = current.take() { + lead = detach_trailing_comments(&mut block); + out.push(block); + } + current = Some(Block { + key, + lines: vec![line.clone()], + lead: std::mem::take(&mut lead), + }); + continue; + } + match current.as_mut() { + Some(block) => block.lines.push(line.clone()), + // Comments above the first key belong to it. + None if line.trim().starts_with('#') => lead.push(line.clone()), + None => {} + } + } + if let Some(block) = current { + out.push(block); + } + out +} + +/// Moves a block's trailing comment run out, for the block that follows it. +/// +/// Trailing blank lines are dropped rather than carried: the caller writes its +/// own separators. +fn detach_trailing_comments(block: &mut Block) -> Vec { + let mut tail: Vec = Vec::new(); + while let Some(last) = block.lines.last() { + let trimmed = last.trim(); + if trimmed.is_empty() || trimmed.starts_with('#') { + tail.push(block.lines.pop().unwrap_or_default()); + } else { + break; + } + } + tail.reverse(); + match tail.iter().position(|l| l.trim().starts_with('#')) { + Some(first) => tail[first..].to_vec(), + None => Vec::new(), + } +} + +/// The key a line declares at an exact indentation, if it declares one. +fn key_at(line: &str, indent: usize) -> Option { + let prefix = line.get(..indent)?; + if !prefix.chars().all(|c| c == ' ') { + return None; + } + let rest = &line[indent..]; + if rest.is_empty() || rest.starts_with(' ') || rest.starts_with('#') || rest.starts_with('-') { + return None; + } + let colon = rest.find(':')?; + let after = &rest[colon + 1..]; + if !(after.is_empty() || after.starts_with(' ')) { + return None; + } + let key = rest[..colon].trim(); + if key.is_empty() { + return None; + } + Some(key.trim_matches(['\'', '"']).to_string()) +} + +/// The leading spaces of a line, or `None` for a blank one. +fn indent_of(line: &str) -> Option { + if line.trim().is_empty() { + return None; + } + Some(line.len() - line.trim_start().len()) +} + +/// Drops a key and everything indented under it. +fn without_key(lines: &[String], indent: usize, key: &str) -> Vec { + let mut out = Vec::new(); + let mut skipping = false; + for line in lines { + if skipping { + match indent_of(line) { + Some(n) if n > indent => continue, + _ => skipping = false, + } + } + if key_at(line, indent).as_deref() == Some(key) { + skipping = true; + continue; + } + out.push(line.clone()); + } + out +} + +/// Inserts a `teaches:` list into one lecture block. +/// +/// It goes above `readings:` when there is one, since the reading list is the +/// long part and a reader should not have to scroll past it to see what the +/// lecture covers. +fn with_teaches(lines: &[String], objectives: &[&str]) -> Vec { + let flow = format!(" teaches: [{}]", objectives.join(", ")); + let rendered: Vec = if flow.len() <= FLOW_WIDTH { + vec![flow] + } else { + let mut out = vec![" teaches:".to_string()]; + out.extend(objectives.iter().map(|o| format!(" - {o}"))); + out + }; + + let at = lines + .iter() + .position(|l| key_at(l, 4).as_deref() == Some("readings")) + .unwrap_or(lines.len()); + + let mut out: Vec = lines[..at].to_vec(); + while out.last().map(|l| l.trim().is_empty()).unwrap_or(false) { + out.pop(); + } + out.extend(rendered); + out.extend_from_slice(&lines[at..]); + out +} + +/// Drops a `# === label` group header, which the file it is moving into is now +/// entirely about. +fn strip_group_header(lead: &[String], labels: &[&str]) -> Vec { + lead.iter() + .filter(|line| { + let bare = line + .trim() + .trim_start_matches('#') + .trim_matches(|c: char| c == '=' || c.is_whitespace()); + !labels.contains(&bare) + }) + .cloned() + .collect() +} + +/// Joins a lead-comment run and a body into text. +fn join(lead: &[String], body: &[String]) -> String { + let mut out = String::new(); + for line in lead.iter().chain(body.iter()) { + out.push_str(line); + out.push('\n'); + } + out +} + +/// Collapses runs of blank lines and ensures exactly one trailing newline. +fn tidy(text: &str) -> String { + let mut out: Vec<&str> = Vec::new(); + for line in text.lines() { + if line.trim().is_empty() && out.last().map(|l| l.trim().is_empty()).unwrap_or(true) { + continue; + } + out.push(line); + } + while out.last().map(|l| l.trim().is_empty()).unwrap_or(false) { + out.pop(); + } + format!("{}\n", out.join("\n")) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::course::fragment; + + const MONOLITH: &str = r#"schema_version: '1.0' + +course: + code: BIOSC 1540 + title: Computational Biology + term: 2026f + +policy: + points_per_item: 1.0 + +units: + - id: u1 + title: Search and Similarity + +references: + ismail2023: + kind: book + title: Bioinformatics + authors: ['Ismail, H. D.'] + +lectures: + L1.2: + title: 'The Digital Genome' + unit: u1 + readings: + + - ref: ismail2023 + locator: 'ch. 1, §1.4' + targets: [t-fastq-structure] + summary: >- + The four-line FASTQ record, and how a quality score is packed into one + character. + +learning_objectives: + + # === L1.2 + lo-read-file-formats: + text: 'Read the text formats that carry sequences.' + unit: u1 + lectures: [L1.2] + order: 2 + level_ceiling: 2 + assessed: true + +learning_targets: + + # === lo-read-file-formats + t-fastq-structure: + text: 'Identify the four lines of a FASTQ record.' + objective: lo-read-file-formats + lectures: [L1.2] + order: 2 + assessed: true +"#; + + fn tmp(tag: &str) -> PathBuf { + let p = std::env::temp_dir().join(format!("coursebank-split-{tag}-{}", std::process::id())); + let _ = std::fs::remove_dir_all(&p); + std::fs::create_dir_all(&p).unwrap(); + p + } + + fn course_dir(tag: &str) -> PathBuf { + let root = tmp(tag); + std::fs::write(root.join(COURSE_FILE), MONOLITH).unwrap(); + root + } + + #[test] + fn a_split_reassembles_into_the_same_course() { + let root = course_dir("roundtrip"); + let before = CourseFile::load(&root.join(COURSE_FILE)).unwrap(); + + let plan = split_plan(&root).unwrap(); + apply(&root, &plan).unwrap(); + + let after = fragment::assemble(&root).unwrap(); + let diffs = differences(&before, &after).unwrap(); + assert!(diffs.is_empty(), "{diffs:?}"); + assert!(after.validate().is_empty(), "{:?}", after.validate()); + } + + #[test] + fn the_split_writes_one_file_per_lecture_and_objective() { + let root = course_dir("files"); + let plan = split_plan(&root).unwrap(); + apply(&root, &plan).unwrap(); + + assert!(root.join("references.yaml").is_file()); + assert!(root.join("lectures/l-1-2.yaml").is_file()); + assert!(root.join("objectives/lo-read-file-formats.yaml").is_file()); + assert!(root.join("course.yaml.bak").is_file()); + + let root_text = std::fs::read_to_string(root.join(COURSE_FILE)).unwrap(); + assert!(root_text.contains("course:")); + assert!(root_text.contains("units:")); + assert!(!root_text.contains("learning_objectives:")); + assert!(!root_text.contains("references:")); + } + + #[test] + fn prose_and_comments_survive_verbatim() { + let root = course_dir("prose"); + let plan = split_plan(&root).unwrap(); + apply(&root, &plan).unwrap(); + + let lecture = std::fs::read_to_string(root.join("lectures/l-1-2.yaml")).unwrap(); + // The folded scalar is still folded, and still wrapped where it was. + assert!(lecture.contains("summary: >-")); + assert!( + lecture + .contains("The four-line FASTQ record, and how a quality score is packed into one") + ); + assert!(lecture.contains("teaches: [lo-read-file-formats]")); + + // A `# === L1.2` header is about the file it now lives in, so it goes. + let objective = + std::fs::read_to_string(root.join("objectives/lo-read-file-formats.yaml")).unwrap(); + assert!(!objective.contains("# ==="), "{objective}"); + } + + #[test] + fn derived_lecture_lists_are_dropped_from_the_fragments() { + let root = course_dir("derived"); + let plan = split_plan(&root).unwrap(); + apply(&root, &plan).unwrap(); + + let objective = + std::fs::read_to_string(root.join("objectives/lo-read-file-formats.yaml")).unwrap(); + assert!(!objective.contains("lectures:"), "{objective}"); + assert!(objective.contains("order: 2")); + assert!(objective.contains("objective: lo-read-file-formats")); + } + + #[test] + fn splitting_twice_is_refused() { + let root = course_dir("twice"); + let plan = split_plan(&root).unwrap(); + apply(&root, &plan).unwrap(); + let message = split_plan(&root).unwrap_err().to_string(); + assert!(message.contains("already"), "{message}"); + } + + /// The same course, with the objective taught twice and the target reached + /// in only one of the two sessions. + fn two_lecture_monolith() -> String { + MONOLITH + .replace( + " lectures: [L1.2]\n order: 2\n level_ceiling: 2", + " lectures: [L1.2, L1.3]\n order: 2\n level_ceiling: 2", + ) + .replace( + "learning_objectives:", + " L1.3:\n title: 'Sequence similarity'\n\nlearning_objectives:", + ) + } + + #[test] + fn a_narrower_target_keeps_its_own_lectures() { + let root = tmp("narrower"); + std::fs::write(root.join(COURSE_FILE), two_lecture_monolith()).unwrap(); + + let plan = split_plan(&root).unwrap(); + let objective = plan + .files + .iter() + .find(|f| f.path.ends_with("lo-read-file-formats.yaml")) + .unwrap(); + // One `lectures:` left in the file: the objective's was derived away, + // the target's says something the objective does not. + assert_eq!( + objective.text.matches("lectures:").count(), + 1, + "{}", + objective.text + ); + assert!( + objective.text.contains("lectures: [L1.2]"), + "{}", + objective.text + ); + } + + #[test] + fn an_objective_taught_twice_is_registered_by_both_lectures() { + let root = tmp("twice-taught"); + std::fs::write(root.join(COURSE_FILE), two_lecture_monolith()).unwrap(); + let before = CourseFile::load(&root.join(COURSE_FILE)).unwrap(); + + let plan = split_plan(&root).unwrap(); + apply(&root, &plan).unwrap(); + + for file in ["lectures/l-1-2.yaml", "lectures/l-1-3.yaml"] { + let text = std::fs::read_to_string(root.join(file)).unwrap(); + assert!( + text.contains("teaches: [lo-read-file-formats]"), + "{file}: {text}" + ); + } + + let after = fragment::assemble(&root).unwrap(); + let diffs = differences(&before, &after).unwrap(); + assert!(diffs.is_empty(), "{diffs:?}"); + } + + #[test] + fn a_long_teaches_list_becomes_a_block_list() { + let many: Vec = (0..8) + .map(|i| format!("lo-a-fairly-long-objective-{i}")) + .collect(); + let refs: Vec<&str> = many.iter().map(String::as_str).collect(); + let out = with_teaches( + &[" L1.2:".to_string(), " title: One".to_string()], + &refs, + ); + assert_eq!(out[2], " teaches:"); + assert_eq!(out[3], " - lo-a-fairly-long-objective-0"); + } + + #[test] + fn the_scanner_finds_keys_only_at_its_own_depth() { + assert_eq!(key_at("lectures:", 0).as_deref(), Some("lectures")); + assert_eq!(key_at(" L1.2:", 2).as_deref(), Some("L1.2")); + assert_eq!(key_at(" 'A+': 1", 2).as_deref(), Some("A+")); + assert_eq!(key_at(" L1.2:", 0), None); + assert_eq!(key_at(" title: x", 2), None); + assert_eq!(key_at(" # comment", 2), None); + assert_eq!(key_at(" - id: u1", 2), None); + // A URL in a value is not a key. + assert_eq!( + key_at(" slides_url: https://x.test/a", 2).as_deref(), + Some("slides_url") + ); + } + + #[test] + fn a_qualified_item_id_is_stripped_in_both_yaml_styles() { + assert_eq!( + unqualify_line(" item: b-1-2::q-fastq-quality-line-001"), + " item: q-fastq-quality-line-001" + ); + assert_eq!( + unqualify_line(" - { number: 1, item: \"b1::q-a-001\", points: 1.5, key: [B] }"), + " - { number: 1, item: \"q-a-001\", points: 1.5, key: [B] }" + ); + assert_eq!(unqualify_line(" item: 'b::q-1'"), " item: 'q-1'"); + } + + #[test] + fn nothing_else_on_the_line_is_touched() { + // Already canonical. + assert_eq!(unqualify_line(" item: q-x"), " item: q-x"); + // A key that merely ends in `item`. + assert_eq!( + unqualify_line(" parent_item: b::q-1"), + " parent_item: b::q-1" + ); + // A `::` somewhere that is not an item id. + assert_eq!( + unqualify_line(" notes: 'see b::q-1 for the earlier wording'"), + " notes: 'see b::q-1 for the earlier wording'" + ); + // A digest that contains no qualifier. + assert_eq!( + unqualify_line(" fingerprint: 3f9a1cb2"), + " fingerprint: 3f9a1cb2" + ); + } + + #[test] + fn rewriting_ids_leaves_the_rest_of_the_record_alone() { + let root = tmp("ids"); + std::fs::create_dir_all(root.join("assessments")).unwrap(); + let record = r#"# assessments/a-1-2.yaml +schema_version: '1.0' +assessment: + id: A1.2 + title: 'A1.2 - The Digital Genome' +items: + - number: 1 + item: b-1-2::q-fastq-quality-line-001 + key: [D] +"#; + std::fs::write(root.join("assessments/a-1-2.yaml"), record).unwrap(); + + let changes = qualified_ids(&root).unwrap(); + assert_eq!(changes.len(), 1); + assert_eq!(changes[0].1, 1, "one line changes"); + apply_ids(&root, &changes).unwrap(); + + let after = std::fs::read_to_string(root.join("assessments/a-1-2.yaml")).unwrap(); + assert!(after.contains("item: q-fastq-quality-line-001"), "{after}"); + assert!(after.starts_with("# assessments/a-1-2.yaml"), "{after}"); + assert!( + after.contains("title: 'A1.2 - The Digital Genome'"), + "{after}" + ); + // Idempotent: a second pass finds nothing. + assert!(qualified_ids(&root).unwrap().is_empty()); + } + + #[test] + fn option_ids_come_from_the_leading_words() { + let ids = option_ids(&[ + "The first line".to_string(), + "The second line".to_string(), + "The third line".to_string(), + "The fourth line".to_string(), + ]); + assert_eq!( + ids, + vec![ + "o-first-line", + "o-second-line", + "o-third-line", + "o-fourth-line" + ] + ); + } + + #[test] + fn colliding_options_are_extended_by_what_tells_them_apart() { + let ids = option_ids(&[ + "Its error probability falls to about one half of the previous value".to_string(), + "Its error probability rises to about ten times the previous value".to_string(), + "Its error probability falls to about one tenth of the previous value".to_string(), + "Its error probability falls to about one hundredth of the previous value".to_string(), + ]); + // Every member of the colliding group is extended, including the last: + // a bare prefix beside two extended siblings reads like an oversight. + assert!(ids[0].ends_with("-half"), "{ids:?}"); + assert!(ids[2].ends_with("-tenth"), "{ids:?}"); + assert!(ids[3].ends_with("-hundredth"), "{ids:?}"); + assert_eq!(ids.iter().collect::>().len(), 4); + } + + #[test] + fn math_survives_long_enough_to_tell_two_options_apart() { + let ids = option_ids(&[ + "Alignment 2; score $-1$".to_string(), + "Alignment 1; score $3$".to_string(), + "Alignment 1; score $2$".to_string(), + "Alignment 2; score $5$".to_string(), + ]); + assert_eq!(ids.iter().collect::>().len(), 4, "{ids:?}"); + assert!(ids[0].contains("minus"), "{ids:?}"); + } + + #[test] + fn options_nothing_tells_apart_get_no_id_at_all() { + let ids = option_ids(&["Yes".to_string(), "Yes".to_string()]); + assert!(ids.is_empty()); + } + + #[test] + fn an_id_does_not_end_on_a_function_word() { + let ids = option_ids(&["A descriptive identifier for the sequence".to_string()]); + assert_eq!(ids, vec!["o-descriptive-identifier"]); + } + + #[test] + fn renaming_a_bank_touches_only_option_ids() { + let bank = r#"items: + - id: q-x + stem: Which line? + options: + - id: A + text: 'The first line' + - id: D + text: 'The fourth line' +"#; + let mut map: OptionMap = BTreeMap::new(); + map.insert( + "q-x".to_string(), + [ + ("A".to_string(), "o-first-line".to_string()), + ("D".to_string(), "o-fourth-line".to_string()), + ] + .into_iter() + .collect(), + ); + + let (out, n) = rename_in_bank(bank, &map); + assert_eq!(n, 2); + assert!(out.contains(" - id: o-first-line"), "{out}"); + assert!(out.contains(" - id: o-fourth-line"), "{out}"); + // The item id, the stem, and the option text are untouched. + assert!(out.contains(" - id: q-x"), "{out}"); + assert!(out.contains("text: 'The fourth line'"), "{out}"); + } + + #[test] + fn renaming_a_record_rewrites_the_key_and_the_overrides() { + let mut map: OptionMap = BTreeMap::new(); + map.insert( + "q-x".to_string(), + [ + ("B".to_string(), "o-second-line".to_string()), + ("D".to_string(), "o-fourth-line".to_string()), + ] + .into_iter() + .collect(), + ); + + let block = r#"items: + - number: 1 + item: q-x + key: [D] + credit_overrides: { B: 0.5 } + level: 1 +"#; + let (out, n) = rename_in_record(block, &map); + assert_eq!(n, 2, "{out}"); + assert!(out.contains("key: [o-fourth-line]"), "{out}"); + assert!( + out.contains("credit_overrides: { o-second-line: 0.5 }"), + "{out}" + ); + assert!(out.contains("level: 1"), "{out}"); + + // The flow form the older records use, with a pre-2.0 qualified id. + let flow = " - { number: 1, item: \"b1::q-x\", key: [B], level: 1 }\n"; + let (out, n) = rename_in_record(flow, &map); + assert_eq!(n, 1, "{out}"); + assert!(out.contains("key: [o-second-line]"), "{out}"); + assert!(out.contains("item: \"b1::q-x\""), "{out}"); + } + + #[test] + fn a_stored_response_column_is_renamed_member_by_member() { + let item: BTreeMap = [ + ("A".to_string(), "o-first".to_string()), + ("C".to_string(), "o-third".to_string()), + ] + .into_iter() + .collect(); + assert_eq!(rename_list("A", &item), "o-first"); + assert_eq!(rename_list("A;C", &item), "o-first;o-third"); + // An option with no mapping is left as it is rather than dropped. + assert_eq!(rename_list("A;Z", &item), "o-first;Z"); + assert_eq!(rename_list("", &item), ""); + } + + #[test] + fn versioning_is_dropped_key_by_key() { + let bank = r#"items: + - id: q-x + version: 1 + status: approved + stem: >- + Which version of the file format is this? + options: + - id: o-a + text: 'The version line' + history: + - version: 2 + date: 2026-01-01 + change: reworded + - version: 1 + date: 2025-12-01 + change: written + topics: [formats] +"#; + let out = drop_versioning(bank); + assert!(!out.contains("version: 1"), "{out}"); + assert!(!out.contains("history:"), "{out}"); + assert!(!out.contains("change: reworded"), "{out}"); + // The stem and the option text mention a version and must survive. + assert!( + out.contains("Which version of the file format is this?"), + "{out}" + ); + assert!(out.contains("text: 'The version line'"), "{out}"); + assert!(out.contains("topics: [formats]"), "{out}"); + assert!(out.contains("status: approved"), "{out}"); + } + + #[test] + fn a_placement_version_goes_too() { + let record = "items:\n - number: 1\n item: q-x\n version: 1\n key: [o-a]\n"; + let out = drop_versioning(record); + assert_eq!( + out, + "items:\n - number: 1\n item: q-x\n key: [o-a]\n" + ); + } + + #[test] + fn the_order_integers_are_dropped_and_nothing_replaces_them() { + let text = r#"learning_objectives: + lo-x: + text: 'Read the formats.' + unit: u1 + order: 2 + level_ceiling: 2 + +learning_targets: + t-a: + text: 'First.' + objective: lo-x + order: 1 + level_ceiling: 1 + t-b: + text: 'Second.' + objective: lo-x + order: 2 + level_ceiling: 2 +"#; + let (out, dropped) = rewrite_order(text); + assert_eq!(dropped, 3, "one on the objective, one on each target"); + assert!(!out.contains("order:"), "{out}"); + // Nothing is written in their place: the objective's position comes + // from its lecture's `teaches`, and a target has no position. Matched + // on the inserted form rather than on `targets:`, which is a substring + // of the `learning_targets:` section header. + assert!(!out.contains("targets: ["), "{out}"); + // Everything else is untouched, prose and ceilings included. + assert!(out.contains("text: 'Read the formats.'"), "{out}"); + assert!(out.contains("level_ceiling: 2"), "{out}"); + assert!(out.contains(" t-a:"), "{out}"); + + // Idempotent. + let (again, dropped) = rewrite_order(&out); + assert_eq!(dropped, 0); + assert_eq!(again, out); + } + + #[test] + fn a_note_becomes_the_fields_it_was_carrying() { + let text = r#"references: + altschul1997gapped: + label: GBLAST97 + kind: article + role: supplemental + title: 'Gapped BLAST and PSI-BLAST' + year: 1997 + note: 'Nucleic Acids Res 25:3389-3402. doi:10.1093/nar/25.17.3389' + brown2013next: + label: BSM + kind: book + title: Next-generation DNA sequencing informatics + year: 2013 +"#; + let (out, count, said) = expand_notes(text); + assert_eq!(count, 1); + assert!(said.is_empty(), "{said:?}"); + assert!(out.contains(" container: Nucleic Acids Res"), "{out}"); + assert!(out.contains(" volume: '25'"), "{out}"); + assert!(out.contains(" pages: '3389-3402'"), "{out}"); + assert!(out.contains(" doi: '10.1093/nar/25.17.3389'"), "{out}"); + // Nothing left of the note, and nothing invented to replace it. + assert!(!out.contains("note:"), "{out}"); + assert!(!out.contains("issue:"), "{out}"); + // The book had no note and is untouched, label and all. + assert!(out.contains(" label: BSM"), "{out}"); + assert!(out.contains(" brown2013next:"), "{out}"); + } + + #[test] + fn a_field_already_declared_is_never_overwritten() { + let text = r#"references: + steinegger2017mmseqs2: + kind: article + title: MMseqs2 + year: 2017 + note: 'Nat Biotechnol 35:1026-1028. doi:10.1038/nbt.3988' + doi: '10.1038/nbt.3988' +"#; + let (out, _, said) = expand_notes(text); + assert!( + said.iter().any(|m| m.contains("already declares doi")), + "{said:?}" + ); + // One doi line, the one that was typed deliberately. + assert_eq!(out.matches("doi:").count(), 1, "{out}"); + assert!(out.contains(" container: Nat Biotechnol"), "{out}"); + } + + #[test] + fn a_trailing_counter_is_recognized_and_nothing_else_is() { + assert_eq!( + strip_counter("q-fastq-quality-line-001").as_deref(), + Some("q-fastq-quality-line") + ); + assert_eq!( + strip_counter("q-blast-seed-14").as_deref(), + Some("q-blast-seed") + ); + // Already bare. + assert_eq!(strip_counter("q-fastq-quality-line"), None); + // A number that is part of the name, not a counter after it. + assert_eq!(strip_counter("q-fastqc-3prime-decay"), None); + assert_eq!(strip_counter("q-001"), Some("q".to_string())); + assert_eq!(strip_counter("q-x-"), None); + } + + #[test] + fn a_rename_touches_the_named_fields_and_leaves_prose_alone() { + let mut rename = BTreeMap::new(); + rename.insert("q-x-001".to_string(), "q-x".to_string()); + + let bank = r#"items: + - id: q-x-001 + supersedes: q-x-001 + stem: Which line? + design: + rationale: >- + Replaces q-x-001, which had two defensible answers. +"#; + let (out, n) = rename_ids(bank, &rename, &["- id", "supersedes"]); + assert_eq!(n, 2, "the id and the supersedes, not the sentence"); + assert!(out.contains(" - id: q-x\n"), "{out}"); + assert!(out.contains(" supersedes: q-x\n"), "{out}"); + // The rationale is prose about the item; editing it would rewrite the + // sentence as well as the reference. + assert!(out.contains("Replaces q-x-001, which had"), "{out}"); + } + + #[test] + fn a_record_that_never_dropped_its_bank_prefix_is_still_renamed() { + let mut rename = BTreeMap::new(); + rename.insert("q-x-001".to_string(), "q-x".to_string()); + + let flow = " - { number: 1, item: \"b-1-2::q-x-001\", key: [o-a] }\n"; + let (out, n) = rename_ids(flow, &rename, &["item"]); + assert_eq!(n, 1); + // Left alone it would point at an id the bank no longer has. + assert!(out.contains("item: \"q-x\""), "{out}"); + + let block = " item: q-x-001\n"; + let (out, n) = rename_ids(block, &rename, &["item"]); + assert_eq!(n, 1); + assert_eq!(out, " item: q-x\n"); + } + + #[test] + fn lecture_slugs_match_the_house_naming() { + assert_eq!(lecture_slug("L1.2"), "l-1-2"); + assert_eq!(lecture_slug("L11"), "l-11"); + assert_eq!(lecture_slug("Week 3"), "week-3"); + } +} diff --git a/src/model.rs b/src/model.rs index acded0c..f35f594 100644 --- a/src/model.rs +++ b/src/model.rs @@ -12,7 +12,9 @@ //! ```text //! taxonomy levels, cognitive processes, error types, status, flags //! │ -//! course course.yaml: identity, policy, objectives, lectures, stimuli +//! course identity, policy, objectives, lectures, stimuli +//! │ └─ course::fragment merges course.yaml, references.yaml, +//! │ lectures/*.yaml, objectives/*.yaml //! │ //! item one question: stem, options, design intent, calibration //! │ @@ -26,9 +28,11 @@ pub mod assessment; pub mod bank; +pub mod calibration; pub mod catalog; pub mod course; pub mod history; pub mod item; pub mod layout; +pub mod seal; pub mod taxonomy; diff --git a/src/model/assessment.rs b/src/model/assessment.rs index e406959..2c5336e 100644 --- a/src/model/assessment.rs +++ b/src/model/assessment.rs @@ -248,11 +248,28 @@ pub struct Placement { /// Printed question number. This is the join key to grading exports, which /// is the entire reason this record exists. pub number: u32, - /// The item's global id, `bank::item`. + /// The item's id, which names it course-wide. A pre-2.0 `bank::item` + /// value still resolves; `coursebank migrate ids` rewrites it. pub item: String, /// The item version used. - #[serde(default, skip_serializing_if = "Option::is_none")] + /// Retained only so a pre-2.0 record still loads. Ignored. + /// + /// What it was for — knowing whether the item has changed since this + /// administration — is [`Placement::stem_digest`] and + /// [`Placement::fingerprint`], which say *what* changed rather than that + /// something did. + #[serde(default, skip_serializing)] pub version: Option, + + /// The stem's digest as administered. See [`crate::item::Item::stem_digest`]. + /// + /// Distinct from `fingerprint`, which covers the options too. A changed + /// fingerprint means the pooled statistics describe an older wording; a + /// changed stem digest means this is no longer the same question, which + /// [`crate::catalog::Catalog::validate_record`] treats as an error rather + /// than a note. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub stem_digest: Option, /// The content fingerprint as used, so later edits are detectable. #[serde(default, skip_serializing_if = "Option::is_none")] pub fingerprint: Option, @@ -265,12 +282,36 @@ pub struct Placement { /// The keyed letters as administered. #[serde(default, skip_serializing_if = "Vec::is_empty")] pub key: Vec, + + /// The option ids offered alongside the key. + /// + /// Resolved when the assessment is assembled and written out explicitly, + /// never sampled at export time. A blueprint may ask for a draw; the record + /// holds what was drawn. Otherwise a bank edit between assembling and + /// printing silently changes the paper, and the key printed on Tuesday + /// disagrees with the one printed on Wednesday. + /// + /// Empty means the whole pool, which is what every pre-2.0 record meant. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub distractors: Vec, + + /// The digest of the item as this administration showed it. + /// + /// See [`crate::item::Item::variant_digest`]. The key statistics pool on: + /// two administrations of one stem with different distractors are two + /// items, and averaging them is averaging different questions. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub variant: Option, /// The level as administered, denormalized so a record reads standalone. #[serde(default, skip_serializing_if = "Option::is_none")] pub level: Option, - /// Objectives as administered, denormalized for the same reason. - #[serde(default, skip_serializing_if = "Vec::is_empty")] - pub learning_objectives: Vec, + /// Targets as administered, denormalized for the same reason. + #[serde( + default, + alias = "learning_objectives", + skip_serializing_if = "Vec::is_empty" + )] + pub learning_targets: Vec, /// Credit awarded to non-keyed options after the fact, keyed by letter. /// /// When item analysis or a student challenge leads you to credit a @@ -281,6 +322,104 @@ pub struct Placement { /// Set when an item was dropped from scoring after administration. #[serde(default, skip_serializing_if = "is_false")] pub dropped: bool, + /// How the drop was applied on the grading platform. + /// + /// Two ways to throw a question out, and they produce different + /// percentages. [`DropStyle::Removed`] takes the item out of the numerator + /// and the denominator: a student with 27 of 35 scores 77.1%. + /// [`DropStyle::FullCredit`] is what you do when the grade of record lives + /// somewhere else and the question cannot be removed from it: every option + /// is keyed, everyone earns the point, and the same student scores 28 of 36, + /// or 77.8%. + /// + /// Defaults to [`DropStyle::Removed`], which is what `dropped: true` meant + /// before this field existed. Set it to `full_credit` when you have credited + /// every option in the platform, so that the report agrees with the grade the + /// student can see. + /// + /// Either way the item is out of the item statistics, the objective + /// evidence, and the IRT fit: an item everyone got right has no variance to + /// contribute. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub dropped_as: Option, + /// Set only when the item was pulled *before* the paper was printed. + /// + /// `dropped` on its own means what its own documentation says: the item was + /// printed, students answered it, and it was then taken out of scoring. + /// Such an item keeps its printed position, because it occupied one on the + /// page the students held, and the responses that come back are numbered + /// around it. + /// + /// An item pulled before printing never occupied a position, so every later + /// question moves up one. That case has to be distinguished, and it cannot + /// be inferred: both look identical in the record. Getting it wrong is not + /// a cosmetic error. Sealing a post-administration drop as if it had never + /// been printed renumbers every question after it, so each response is + /// attributed to the wrong item, the statistics for those items are + /// computed from answers to different questions, and nothing in the output + /// looks obviously wrong. + /// + /// Practically: leave this alone when you discover a bad question after the + /// exam, which is the common case. Set it when you cut a question from the + /// draft and reprinted. + #[serde(default, skip_serializing_if = "is_false")] + pub dropped_before_printing: bool, +} + +/// How a dropped item was handled on the grading platform. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum DropStyle { + /// Taken out of the numerator and the denominator. + Removed, + /// Every option credited, so the item stays in both. + FullCredit, +} + +impl Placement { + /// The variant this placement administered. + /// + /// Recorded when the assessment was assembled; derived from the option set + /// otherwise, which is what makes every record written before 2.0 poolable + /// without being rewritten. A pre-2.0 placement names its key and no + /// distractors, which means the whole pool — a well-defined option set, and + /// so a well-defined variant. + /// + /// # Arguments + /// + /// * `item` - the item this placement names. + /// + /// # Returns + /// + /// The digest. + pub fn variant_of(&self, item: &crate::item::Item) -> String { + self.variant + .clone() + .unwrap_or_else(|| item.variant_digest(&self.key, &self.distractors)) + } + + /// Whether this placement was dropped by crediting every option. + /// + /// # Returns + /// + /// `true` only when the item is dropped *and* the drop was applied as full + /// credit, so the item still belongs in the points of record. + pub fn dropped_with_credit(&self) -> bool { + self.dropped && self.dropped_as == Some(DropStyle::FullCredit) + } + + /// Whether this placement occupied a printed position on the paper. + /// + /// Everything that lays out a page or reads a page back goes through this, + /// so the printed form, the seal, and the decoder cannot disagree about + /// which question sat where. + /// + /// # Returns + /// + /// `true` unless the item was pulled before printing. + pub fn was_printed(&self) -> bool { + !(self.dropped && self.dropped_before_printing) + } } impl AssessmentFile { @@ -451,6 +590,20 @@ impl AssessmentFile { )); } } + + if p.dropped_as.is_some() && !p.dropped { + issues.push(format!( + "question {}: `dropped_as` is set but `dropped` is not, so nothing is dropped", + p.number + )); + } + if p.dropped && !p.credit_overrides.is_empty() { + issues.push(format!( + "question {}: dropped and carrying credit overrides. Pick one: an override \ + rescores an option, a drop removes the question", + p.number + )); + } } let mut form_ids: Vec<&str> = Vec::new(); diff --git a/src/model/bank.rs b/src/model/bank.rs index b7569a5..19b5bd8 100644 --- a/src/model/bank.rs +++ b/src/model/bank.rs @@ -79,7 +79,7 @@ pub struct BankMeta { /// What a bank is scoped to. /// -/// A bank may be scoped by lecture, by objective, by topic, or by none of them. +/// A bank may be scoped by lecture, by learning target, by topic, or by none of them. /// Declaring the scope is what lets the catalog report *gaps*: it can only tell /// you that lecture 12 has no Apply-level items if it knows lecture 12 is /// supposed to be covered here. @@ -89,9 +89,13 @@ pub struct Scope { /// Lectures this bank draws from. #[serde(default, skip_serializing_if = "Vec::is_empty")] pub lectures: Vec, - /// Objectives this bank is responsible for covering. - #[serde(default, skip_serializing_if = "Vec::is_empty")] - pub learning_objectives: Vec, + /// Learning targets this bank is responsible for covering. + #[serde( + default, + alias = "learning_objectives", + skip_serializing_if = "Vec::is_empty" + )] + pub learning_targets: Vec, /// Units this bank belongs to. #[serde(default, skip_serializing_if = "Vec::is_empty")] pub units: Vec, @@ -254,9 +258,9 @@ impl BankFile { issues.push(format!("bank.scope: unknown lecture `{lec}`")); } } - for lo in &self.bank.scope.learning_objectives { - if !c.learning_objectives.contains_key(lo) { - issues.push(format!("bank.scope: unknown learning objective `{lo}`")); + for target in &self.bank.scope.learning_targets { + if !c.is_target(target) && !c.is_objective(target) { + issues.push(format!("bank.scope: unknown learning target `{target}`")); } } } @@ -348,14 +352,26 @@ fn validate_item( if it.stem.trim().is_empty() { issues.push("empty stem".into()); } - if it.version == 0 { - issues.push("version must be at least 1".into()); + if let Some(replaced) = &it.supersedes { + if *replaced == it.id { + issues.push("supersedes names this item".into()); + } } - // --- options ----------------------------------------------------------- - if it.options.len() < 2 { + // --- options ---- + // An open-response item takes no options; its answer lives in `solution`. + // Every other format needs at least two things to choose between. + if it.format.has_options() { + if it.options.len() < 2 { + issues.push(format!( + "needs at least 2 options, has {}", + it.options.len() + )); + } + } else if !it.options.is_empty() { issues.push(format!( - "needs at least 2 options, has {}", + "{} items take no options, but {} were given; put the answer in `solution`", + it.format.as_str(), it.options.len() )); } @@ -365,21 +381,37 @@ fn validate_item( if o.text.trim().is_empty() { issues.push(format!("option {pos}: empty text")); } - let letter_ok = o.id.len() == 1 - && o.id - .chars() - .next() - .map(|c| c.is_ascii_uppercase() && c <= 'H') - .unwrap_or(false); - if !letter_ok { + // Two forms are accepted: the 2.0 name, and the letter that preceded + // it. A bank migrates when `coursebank migrate options` is run on it, + // not when the tool is upgraded, so a course mid-migration still loads. + let slug_ok = o.id.strip_prefix("o-").is_some_and(|rest| { + !rest.is_empty() + && !rest.starts_with('-') + && !rest.ends_with('-') + && !rest.contains("--") + && rest + .chars() + .all(|c| c.is_ascii_lowercase() || c.is_ascii_digit() || c == '-') + }); + if !slug_ok && !Item::is_legacy_option_id(&o.id) { issues.push(format!( - "option {pos}: id `{}` must be a single letter A through H", + "option {pos}: id `{}` is neither a name such as `o-fourth-line` nor a pre-2.0 \ + letter A through H", o.id )); } if seen.contains(&o.id.as_str()) { issues.push(format!("option {pos}: duplicate option id `{}`", o.id)); } + if let Some(retirement) = &o.retired { + if retirement.reason.trim().is_empty() { + issues.push(format!( + "option {pos}: retired without a reason. The reason is the finding — what \ + the option did or failed to do — and it is the only part of a retirement \ + that is worth anything later." + )); + } + } seen.push(&o.id); let credit = o.credit(); @@ -418,15 +450,28 @@ fn validate_item( } } - // --- key --------------------------------------------------------------- + // --- key --- + // Counted over the pool that can still be drawn: a retired option is a + // record, not an offer. + let (live_keys, live_distractors) = it.pool(); let keys = it.key_indices(); match it.format { Format::SingleBestAnswer => { - if keys.len() != 1 { - issues.push(format!( - "single_best_answer needs exactly one keyed option, has {}", - keys.len() - )); + // Several defensible keys is a pool, not a bug — it is what lets you + // test whether "fourth" or "last of the four" is doing the work. + // Exactly one of them reaches a student, and that is the + // placement's business: see + // [`crate::catalog::Catalog::validate_record`]. + if live_keys.is_empty() { + issues.push("single_best_answer needs at least one keyed option".into()); + } + if live_distractors.is_empty() { + let retired = it.options.iter().any(|o| o.retired.is_some()); + issues.push(if retired { + "every distractor is retired, so nothing can be drawn against the key".into() + } else { + "has no option that is not keyed correct, so it asks nothing".to_string() + }); } } Format::MultipleResponse => { @@ -448,9 +493,14 @@ fn validate_item( issues.push("true_false needs exactly one keyed option".into()); } } + Format::OpenResponse => { + if !keys.is_empty() { + issues.push("open_response items have no keyed option".into()); + } + } } - // --- level and process must agree ------------------------------------- + // --- level and process must agree ------- if let Some(p) = it.cognitive_process { if !it.level.allows(p) { issues.push(format!( @@ -461,7 +511,7 @@ fn validate_item( } } - // --- design plausibility ---------------------------------------------- + // --- design plausibility ------ if let Some(d) = &it.design { if let Some(x) = d.expected_difficulty { if !(0.0..=1.0).contains(&x) { @@ -479,61 +529,16 @@ fn validate_item( } } - // --- calibration plausibility ----------------------------------------- - if let Some(c) = &it.calibration { - if let Some(p) = c.p_value { - if !(0.0..=1.0).contains(&p) { - issues.push(format!( - "calibration.p_value must be between 0 and 1, got {p}" - )); - } - } - if let Some(r) = c.point_biserial { - if !(-1.0..=1.0).contains(&r) { - issues.push(format!( - "calibration.point_biserial must be between -1 and 1, got {r}" - )); - } - } - for letter in c.option_stats.keys() { - if it.option(letter).is_none() { - issues.push(format!( - "calibration.option_stats has `{letter}`, which is not an option of this item" - )); - } - } - if let Some(irt) = &c.irt { - if irt.a <= 0.0 { - issues.push(format!("calibration.irt.a must be positive, got {}", irt.a)); - } - if let Some(cp) = irt.c { - if !(0.0..1.0).contains(&cp) { - issues.push(format!("calibration.irt.c must be in [0, 1), got {cp}")); - } - } - } + // --- calibration plausibility ------ + if it.calibration.is_some() { + issues.push( + "has a `calibration:` block, but statistics live in analysis/calibration.yaml since \ + 2.0. A bank's diff should be a change of intent, not the output of a grading run." + .to_string(), + ); } - // --- history must be coherent ----------------------------------------- - let mut last_version = 0u32; - for (i, h) in it.history.iter().enumerate() { - if h.version <= last_version { - issues.push(format!( - "history entry {} has version {} which does not increase", - i + 1, - h.version - )); - } - last_version = h.version; - } - if !it.history.is_empty() && last_version > it.version { - issues.push(format!( - "history records version {last_version} but the item says version {}", - it.version - )); - } - - // --- retirement ------------------------------------------------------- + // --- retirement ----- if it.retired.is_some() && it.status != Status::Retired { issues.push(format!( "has a `retired` block but status is `{}`", @@ -541,15 +546,15 @@ fn validate_item( )); } - // --- approval gate ---------------------------------------------------- + // --- approval gate ------- // Approval is what permits an item onto a graded assessment, so it is the // right place to require that the item is fully sourced and designed. if it.status == Status::Approved { if it.cognitive_process.is_none() { issues.push("approved items must declare a cognitive_process".into()); } - if it.learning_objectives.is_empty() { - issues.push("approved items must reference at least one learning objective".into()); + if it.learning_targets.is_empty() { + issues.push("approved items must reference at least one learning target".into()); } if it.sources.is_empty() { issues.push("approved items must cite at least one source".into()); @@ -557,30 +562,82 @@ fn validate_item( if it.design.is_none() { issues.push("approved items must carry a design block".into()); } + // An open-response item is graded from its solution, so approving one with + // neither a model answer nor a rubric would leave nothing to mark it by. + if !it.format.has_options() { + let gradeable = it + .solution + .as_ref() + .is_some_and(|s| s.model_answer.is_some() || !s.rubric.is_empty()); + if !gradeable { + issues.push( + "approved open_response items need a solution with a model_answer or a rubric" + .into(), + ); + } + } } - // --- cross-file references -------------------------------------------- + // --- cross-file references ---- if let Some(c) = course { - for lo in &it.learning_objectives { - match c.learning_objectives.get(lo) { - None => issues.push(format!("unknown learning objective `{lo}`")), - Some(obj) => { - if let Some(ceiling) = obj.level_ceiling { - if it.level > ceiling { - issues.push(format!( - "level {} exceeds the ceiling {} declared for objective `{lo}`", - it.level.code(), - ceiling.code() - )); - } - } - if !obj.assessed { - issues.push(format!( - "objective `{lo}` is marked `assessed: false` but this item measures it" - )); - } + for tag in &it.learning_targets { + let is_target = c.is_target(tag); + if !is_target && !c.is_objective(tag) { + issues.push(format!("unknown learning target `{tag}`")); + continue; + } + // Items are tagged at the target tier. Tagging an objective that has + // targets would put the item in that objective's denominator without + // recording which performance the question actually asked for, and + // that record is what a report drills into and what coverage + // analysis counts. An objective with no targets stands as its own. + let its_targets = c.targets(tag); + if !its_targets.is_empty() { + issues.push(format!( + "`{tag}` is an objective with {} learning target(s); tag the specific \ + target this item measures instead", + its_targets.len() + )); + } + // The ceiling may be inherited from the objective, so a target that + // declares none of its own is still bounded. + if let Some(ceiling) = c.effective_level_ceiling(tag) { + if it.level > ceiling { + let declares_its_own = if is_target { + c.learning_targets + .get(tag) + .is_some_and(|t| t.level_ceiling.is_some()) + } else { + true + }; + let source = if declares_its_own { + format!("declared for `{tag}`") + } else { + format!( + "inherited by `{tag}` from its objective `{}`", + c.objective_for(tag) + ) + }; + issues.push(format!( + "level {} exceeds the ceiling {} {source}", + it.level.code(), + ceiling.code() + )); } } + if !c.is_assessed(tag) { + let parked_above = + is_target && c.learning_targets.get(tag).is_some_and(|t| t.assessed); + let which = if parked_above { + format!( + "its objective `{}` is marked `assessed: false`", + c.objective_for(tag) + ) + } else { + format!("`{tag}` is marked `assessed: false`") + }; + issues.push(format!("{which} but this item measures it")); + } } for s in &it.sources { if !c.lectures.contains_key(&s.lecture) { @@ -592,6 +649,15 @@ fn validate_item( issues.push(format!("unknown stimulus `{st}`")); } } + // A citation that names a reference key must name a real one, so a review + // pointer in the solutions document never resolves to nothing. + for citation in it.solution.iter().flat_map(|s| &s.review) { + if let Some(key) = &citation.reference { + if !c.references.contains_key(key) { + issues.push(format!("solution.review cites unknown reference `{key}`")); + } + } + } if let Some(floor) = c.policy.partial_credit_floor_level { for o in &it.options { if o.is_partial() && it.level < floor { @@ -644,6 +710,53 @@ mod tests { assert!(b.validate(None).is_empty(), "{:?}", b.validate(None)); } + #[test] + fn open_response_validates_without_options_and_rejects_them() { + // No options is fine, and no key is required. + let ok = bank( + r#" + - id: q-a-op-001 + status: draft + level: 2 + format: open_response + stem: Explain the first law. + solution: + model_answer: Energy is conserved. +"#, + ); + assert!(ok.validate(None).is_empty(), "{:?}", ok.validate(None)); + + // Giving an open-response item options is the mistake, and so is approving + // one with nothing to grade it by. + let bad = bank( + r#" + - id: q-a-op-002 + status: approved + level: 2 + format: open_response + cognitive_process: explain + learning_targets: [lo-x] + sources: [{ lecture: L1.1 }] + design: { rationale: r } + stem: Explain the first law. + options: + - { id: A, text: a, correct: true } + - { id: B, text: b } +"#, + ); + let issues = bad.validate(None); + assert!( + issues.iter().any(|i| i.contains("take no options")), + "{issues:?}" + ); + assert!( + issues + .iter() + .any(|i| i.contains("model_answer or a rubric")), + "{issues:?}" + ); + } + #[test] fn catches_missing_and_multiple_keys() { let b = bank( @@ -663,20 +776,43 @@ mod tests { options: - { id: A, text: a, correct: true } - { id: B, text: b, correct: true } + - id: q-a-003 + status: draft + level: 1 + format: single_best_answer + stem: s + options: + - { id: o-key-one, text: a, correct: true } + - { id: o-key-two, text: b, correct: true } + - { id: o-wrong-one, text: c } + - { id: o-wrong-two, text: d } "#, ); let issues = b.validate(None); + // No key at all is still a bank problem: nothing can be drawn from it. assert!( issues .iter() - .any(|i| i.contains("exactly one keyed option")) + .any(|i| i.starts_with("q-a-001") && i.contains("at least one keyed option")), + "{issues:?}" ); - assert_eq!( + // Keying every option is still a bank problem, for the older reason: a + // question with nothing to choose against asks nothing. + assert!( issues .iter() - .filter(|i| i.contains("exactly one keyed option")) - .count(), - 2 + .any(|i| i.starts_with("q-a-002") && i.contains("asks nothing")), + "{issues:?}" + ); + // Two defensible keys alongside real distractors is not a problem. It + // is the pool doing its job — it is what lets you test whether the + // wording of the key is what students are answering. Exactly one of + // them reaches a student, and that is checked against the placement + // that administers it, in `Catalog::validate_placement`. + assert!( + !issues.iter().any(|i| i.starts_with("q-a-003") + && (i.contains("keyed option") || i.contains("asks nothing"))), + "{issues:?}" ); } @@ -742,7 +878,7 @@ mod tests { let issues = b.validate(None); for want in [ "cognitive_process", - "learning objective", + "learning target", "source", "design block", ] { @@ -829,7 +965,7 @@ learning_objectives: status: draft level: 4 stem: s - learning_objectives: [lo-known, lo-unknown] + learning_targets: [lo-known, lo-unknown] sources: [{ lecture: L99 }] options: - { id: A, text: a, correct: true } @@ -840,7 +976,7 @@ learning_objectives: assert!( issues .iter() - .any(|i| i.contains("unknown learning objective `lo-unknown`")) + .any(|i| i.contains("unknown learning target `lo-unknown`")) ); assert!(issues.iter().any(|i| i.contains("unknown lecture `L99`"))); assert!( @@ -850,24 +986,54 @@ learning_objectives: } #[test] - fn history_versions_must_increase() { + fn an_item_must_tag_a_target_not_an_objective_that_has_them() { + let course: CourseFile = serde_yaml_ng::from_str( + r#" +course: { code: X, title: Y, term: Z } +learning_objectives: + lo-binding: { text: Quantify binding., level_ceiling: 3 } + lo-solo: { text: An objective with no targets. } +learning_targets: + t-kd: { text: Write the expression., objective: lo-binding } +"#, + ) + .unwrap(); let b = bank( r#" - id: q-a-001 - version: 2 status: draft level: 1 stem: s + learning_targets: [lo-binding] + options: + - { id: A, text: a, correct: true } + - { id: B, text: b } + - id: q-a-002 + status: draft + level: 1 + stem: s + learning_targets: [t-kd, lo-solo] options: - { id: A, text: a, correct: true } - { id: B, text: b } - history: - - { version: 2, date: 2026-01-01, change: second } - - { version: 1, date: 2026-01-02, change: first } "#, ); - let issues = b.validate(None); - assert!(issues.iter().any(|i| i.contains("does not increase"))); + let issues = b.validate(Some(&course)); + assert!( + issues + .iter() + .any(|i| i.contains("`lo-binding` is an objective with 1 learning target(s)")), + "got {issues:?}" + ); + // A target is fine, and so is an objective that has no targets: it + // stands as its own, which is what lets a course migrate a unit at a + // time. + assert!( + !issues + .iter() + .any(|i| i.contains("`t-kd`") || i.contains("`lo-solo`")), + "got {issues:?}" + ); } #[test] @@ -879,7 +1045,7 @@ learning_objectives: level: 1 cognitive_process: recall stem: s - learning_objectives: [lo] + learning_targets: [lo] sources: [{ lecture: L1 }] design: { expected_difficulty: 0.8 } options: @@ -898,7 +1064,7 @@ learning_objectives: cognitive_process: generate bonus: true stem: s - learning_objectives: [lo] + learning_targets: [lo] sources: [{ lecture: L1 }] design: { expected_difficulty: 0.3 } options: diff --git a/src/model/calibration.rs b/src/model/calibration.rs new file mode 100644 index 0000000..114cf3d --- /dev/null +++ b/src/model/calibration.rs @@ -0,0 +1,838 @@ +// SPDX-License-Identifier: Prosperity-3.0.0 +// Copyright Scientific Computing Studio +// Source: https://git.scient.ing/education/coursebank + +//! Where the statistics live, which is not in the bank. +//! +//! A bank file is a reviewed artifact: someone wrote the question, someone +//! argued about the distractors, and the diff on it should be a change of +//! intent. Statistics are neither reviewed nor intended — they are what +//! happened — and writing them back into the bank means every grading run +//! produces a diff on a file whose history is supposed to be about wording. +//! +//! So the evidence lives in `analysis/`, in two kinds of file: +//! +//! | File | Format | Rewritten? | Holds | +//! |:--|:--|:--|:--| +//! | `analysis/administrations/-items.csv` | CSV | never | one row per question | +//! | `analysis/administrations/-options.csv` | CSV | never | one row per question and option | +//! | `analysis/calibration.yaml` | YAML | by `calibrate` | the pooled per-item view | +//! +//! The format follows the shape. An administration record is a table — fixed +//! columns, one row per question, machine-written, never hand-edited — so it is +//! CSV: one line per item rather than fifteen, which diffs better, and it loads +//! straight into pandas or DuckDB, which is much of the point of committing it. +//! The pooled view is not a table. It is three levels deep, item to variant to +//! option history, with fitted parameters and variable-length lists, and as CSV +//! that would be three files joined by keys — a relational schema for the one +//! file a person actually reads in a pull request. That stays YAML. +//! +//! Neither is Parquet, and the reason is the review workflow: a binary file +//! shows nothing in a diff and cannot be merged. Parquet is right for `data/` +//! precisely because that is bulk, ignored, and never reviewed. +//! +//! An administration file is written once and not touched again, for the same +//! reason a seal is not: it is a record of a thing that happened on a day. The +//! calibration file is the accepted rollup — what `lint` compares your +//! predictions against, and what a report reads — and `calibrate` proposes +//! changes to it as a diff you review before committing. +//! +//! Both are meant to be committed. Neither can carry student data, and that is +//! a property of the types rather than a promise: there is no field for a +//! student key, an identifier, a section, or an ability estimate, and the +//! structures reject unknown keys, so a file carrying one fails to load rather +//! than being quietly accepted. Everything per-person stays in `data/`, which +//! is what your `.gitignore` is for. +//! +//! # Linking back to the bank +//! +//! By item id, which since 2.0 names the item course-wide and has no file name +//! in it, and by variant digest, which says which option set the numbers +//! describe. A record also carries the stem digest it was measured against, so +//! [`CalibrationFile::validate`] can say that an item has been reworded since — +//! the statistics then describe a question that no longer exists under that id. + +use std::collections::BTreeMap; +use std::path::{Path, PathBuf}; + +use serde::{Deserialize, Serialize}; + +use crate::course::SCHEMA_VERSION; +use crate::date::Date; +use crate::error::{Error, Result}; +use crate::item::{Calibration, IrtModel, IrtParams, OptionStat}; +use crate::taxonomy::Flag; +use crate::yaml; + +/// The file name of the pooled calibration store, under `analysis/`. +pub const CALIBRATION_FILE: &str = "calibration.yaml"; + +/// The pooled per-item statistics: `analysis/calibration.yaml`. +/// +/// Keyed by item id. This is the file `calibrate` rewrites and the one +/// everything else reads; [`crate::catalog::Catalog::load`] fills each item's +/// in-memory calibration from it, so nothing downstream has to know where the +/// numbers came from. +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct CalibrationFile { + /// Schema version. + #[serde( + default = "default_version", + deserialize_with = "yaml::flexible_string" + )] + pub schema_version: String, + + /// One entry per calibrated item, by item id. + #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] + pub items: BTreeMap, +} + +impl CalibrationFile { + /// Loads the store, or an empty one when the file does not exist yet. + /// + /// Absence is not an error: a course that has not graded anything has no + /// statistics, and every command that reads them has to work anyway. + /// + /// # Arguments + /// + /// * `path` - the file, usually `analysis/calibration.yaml`. + /// + /// # Returns + /// + /// The store. + /// + /// # Errors + /// + /// Returns [`Error::Yaml`] when the file exists and does not parse. + pub fn load(path: &Path) -> Result { + if !path.is_file() { + return Ok(CalibrationFile::default()); + } + yaml::read(path) + } + + /// Writes the store. + /// + /// # Arguments + /// + /// * `path` - the destination. + /// + /// # Errors + /// + /// Returns [`Error::Io`] on a write failure. + pub fn save(&self, path: &Path) -> Result<()> { + yaml::write(path, self) + } + + /// The calibration recorded for one item. + /// + /// # Arguments + /// + /// * `item` - the item id. + /// + /// # Returns + /// + /// The record, or `None` when the item has never been calibrated. + pub fn get(&self, item: &str) -> Option<&Calibration> { + self.items.get(item) + } + + /// Checks the store against the bank it describes. + /// + /// The checks that matter for a file kept apart from what it refers to: a + /// record for an item that no longer exists, an option id the item does not + /// have, and — the one worth having — statistics measured against a stem + /// that has since been reworded, which since 2.0 means they describe a + /// different question wearing the same id. + /// + /// # Arguments + /// + /// * `catalog` - the loaded course. + /// + /// # Returns + /// + /// One message per problem, empty when the store agrees with the bank. + pub fn validate(&self, catalog: &crate::catalog::Catalog) -> Vec { + let mut issues = Vec::new(); + + for (id, calibration) in &self.items { + let Some(entry) = catalog.get(id) else { + issues.push(format!( + "calibration for `{id}`: no such item. Statistics outlive an item only if \ + it is retired, not deleted — a retired item keeps its id so its numbers \ + still mean something." + )); + continue; + }; + let item = &entry.item; + + for variant in &calibration.variants { + for option in variant.option_stats.keys() { + if item.option(option).is_none() { + issues.push(format!( + "calibration for `{id}`: variant `{}` has statistics for `{option}`, \ + which is not an option of this item", + short(&variant.variant) + )); + } + } + if !variant.key.is_empty() + && variant.variant != item.variant_digest(&variant.key, &variant.distractors) + { + issues.push(format!( + "calibration for `{id}`: variant `{}` was measured against an option set \ + that has since been reworded, so its numbers describe wording no \ + student now sees", + short(&variant.variant) + )); + } + } + for option in calibration.options.keys() { + if item.option(option).is_none() { + issues.push(format!( + "calibration for `{id}`: an option history names `{option}`, which is \ + not an option of this item" + )); + } + } + } + issues + } +} + +/// What one administration measured: `analysis/administrations/.yaml`. +/// +/// Written once, when the exam is analyzed, and never rewritten. It is the +/// audit trail under [`CalibrationFile`]: the pooled numbers say an item sits +/// at 0.63, and these say which exams that came from and what each one saw. +/// Keeping them also means the history survives losing `data/`, which is +/// ignored by git and rotates. +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct MeasurementFile { + /// Schema version. + #[serde( + default = "default_version", + deserialize_with = "yaml::flexible_string" + )] + pub schema_version: String, + + /// What was administered, and how it was analyzed. + pub administration: MeasurementMeta, + + /// One entry per question, in the order it was printed. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub items: Vec, +} + +/// What an administration was, for a reader two years later. +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct MeasurementMeta { + /// The administration id these numbers came from. + pub id: String, + /// The assessment that was administered. + pub assessment: String, + /// The term. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub term: Option, + /// The date it was given. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub date: Option, + /// The forms in play. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub forms: Vec, + /// How many examinees the numbers pool over. + /// + /// The one number to read before any of the others. A point-biserial on + /// twenty-seven students is a different kind of claim than one on three + /// hundred, and nothing below records how thin it is. + pub n_examinees: usize, + /// The item response model fitted, when one was. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub model: Option, + /// When the analysis was run. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub generated: Option, + /// The version of the tool that ran it, since the numbers depend on it. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub coursebank: Option, +} + +/// One question's statistics from one administration. +/// +/// Cohort aggregates only. There is deliberately no per-section or per-form +/// breakdown: those get small, and a small cell crossed with anything else is +/// how an aggregate stops being one. +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct Measurement { + /// The item id. + pub item: String, + /// The question number it was printed as. + pub number: u32, + /// The option set administered. See [`crate::item::Item::variant_digest`]. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub variant: Option, + /// The stem as administered. See [`crate::item::Item::stem_digest`]. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub stem_digest: Option, + /// Examinees who saw it. + pub n: usize, + /// Proportion correct. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub p_value: Option, + /// Corrected item-total point-biserial correlation. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub point_biserial: Option, + /// Upper-minus-lower-group discrimination index. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub discrimination_index: Option, + /// The option ids keyed correct, so a row in the options file says whether + /// it describes the answer or a distractor without a join. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub key: Vec, + /// Per-option behaviour, by option id. + #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] + pub option_stats: BTreeMap, + /// Fitted parameters, when the sample supported a fit. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub irt: Option, + /// Machine-detected problems with this question on this administration. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub flags: Vec, +} + +/// The columns of an items file, and the only ones accepted on read. +/// +/// An allowlist rather than a type-level guarantee. In YAML the structures +/// reject unknown keys, so a file carrying a student column could not be +/// loaded; CSV readers are tolerant of extra columns, so the same assurance has +/// to be an explicit check. This is it, and [`MeasurementFile::read_csv`] +/// refuses any header not named here. +pub const ITEM_COLUMNS: [&str; 16] = [ + "administration_id", + "assessment", + "term", + "date", + "forms", + "n_examinees", + "coursebank", + "generated", + "item", + "number", + "variant", + "stem_digest", + "n", + "p_value", + "point_biserial", + "discrimination_index", +]; + +/// The columns of an options file, and the only ones accepted on read. +pub const OPTION_COLUMNS: [&str; 9] = [ + "administration_id", + "item", + "number", + "option", + "keyed", + "selection_rate", + "point_biserial", + "upper_group_rate", + "lower_group_rate", +]; + +/// One row of an items file. +#[derive(Debug, Clone, Serialize, Deserialize)] +struct ItemRow { + administration_id: String, + assessment: String, + term: String, + date: String, + forms: String, + n_examinees: usize, + coursebank: String, + generated: String, + item: String, + number: u32, + variant: String, + stem_digest: String, + n: usize, + p_value: String, + point_biserial: String, + discrimination_index: String, +} + +/// One row of an options file. +#[derive(Debug, Clone, Serialize, Deserialize)] +struct OptionRow { + administration_id: String, + item: String, + number: u32, + option: String, + keyed: bool, + selection_rate: String, + point_biserial: String, + upper_group_rate: String, + lower_group_rate: String, +} + +impl MeasurementFile { + /// The two file names this administration writes, items first. + /// + /// # Arguments + /// + /// * `dir` - usually `analysis/administrations`. + /// + /// # Returns + /// + /// The items path and the options path. + pub fn paths(&self, dir: &Path) -> (PathBuf, PathBuf) { + let stem = crate::course::slugify(&self.administration.id); + ( + dir.join(format!("{stem}-items.csv")), + dir.join(format!("{stem}-options.csv")), + ) + } + + /// Writes the two files, refusing to overwrite either. + /// + /// An administration is a thing that happened once, so replacing its record + /// is a deliberate act: delete the files first if you mean to re-analyze. + /// + /// Two files rather than one because the data is two shapes — one row per + /// question, one row per question and option — and a single sparse table + /// serves neither. The administration's metadata repeats on every row, + /// which is what makes each file independently loadable and is the same + /// convention the response store already uses. + /// + /// # Arguments + /// + /// * `dir` - the directory to write into. + /// + /// # Returns + /// + /// The paths written. + /// + /// # Errors + /// + /// Returns [`Error::Usage`] when either file exists, and [`Error::Io`] or + /// [`Error::Csv`] on a write failure. + pub fn write_csv(&self, dir: &Path) -> Result> { + let (items_path, options_path) = self.paths(dir); + for path in [&items_path, &options_path] { + if path.exists() { + return Err(Error::usage(format!( + "{} already records this administration. It happened once, so replacing it \ + is a deliberate act: delete it first if you mean to re-analyze.", + path.display() + ))); + } + } + std::fs::create_dir_all(dir).map_err(|e| Error::io(dir, e))?; + + let meta = &self.administration; + let mut items = csv::Writer::from_path(&items_path).map_err(|e| Error::Csv { + path: items_path.clone(), + source: e, + })?; + let mut options = csv::Writer::from_path(&options_path).map_err(|e| Error::Csv { + path: options_path.clone(), + source: e, + })?; + + for measurement in &self.items { + items + .serialize(ItemRow { + administration_id: meta.id.clone(), + assessment: meta.assessment.clone(), + term: meta.term.clone().unwrap_or_default(), + date: meta.date.map(|d| d.to_string()).unwrap_or_default(), + forms: meta.forms.join(";"), + n_examinees: meta.n_examinees, + coursebank: meta.coursebank.clone().unwrap_or_default(), + generated: meta.generated.map(|d| d.to_string()).unwrap_or_default(), + item: measurement.item.clone(), + number: measurement.number, + variant: measurement.variant.clone().unwrap_or_default(), + stem_digest: measurement.stem_digest.clone().unwrap_or_default(), + n: measurement.n, + p_value: number(measurement.p_value), + point_biserial: number(measurement.point_biserial), + discrimination_index: number(measurement.discrimination_index), + }) + .map_err(|e| Error::Csv { + path: items_path.clone(), + source: e, + })?; + + for (option, stat) in &measurement.option_stats { + options + .serialize(OptionRow { + administration_id: meta.id.clone(), + item: measurement.item.clone(), + number: measurement.number, + option: option.clone(), + keyed: measurement.key.iter().any(|k| k == option), + selection_rate: number(stat.selection_rate), + point_biserial: number(stat.point_biserial), + upper_group_rate: number(stat.upper_group_rate), + lower_group_rate: number(stat.lower_group_rate), + }) + .map_err(|e| Error::Csv { + path: options_path.clone(), + source: e, + })?; + } + } + + items.flush().map_err(|e| Error::io(&items_path, e))?; + options.flush().map_err(|e| Error::io(&options_path, e))?; + Ok(vec![items_path, options_path]) + } + + /// Reads one administration back from its two files. + /// + /// # Arguments + /// + /// * `items_path` - the items file. The options file is found beside it. + /// + /// # Returns + /// + /// The administration, with its per-option statistics reattached. + /// + /// # Errors + /// + /// Returns [`Error::Csv`] on a parse failure and [`Error::Invalid`] when a + /// file carries a column that is not in [`ITEM_COLUMNS`] or + /// [`OPTION_COLUMNS`] — which is how a student column is caught. + pub fn read_csv(items_path: &Path) -> Result { + let options_path = PathBuf::from( + items_path + .to_string_lossy() + .replace("-items.csv", "-options.csv"), + ); + + let mut reader = open_csv(items_path, &ITEM_COLUMNS)?; + let mut meta = MeasurementMeta::default(); + let mut items: Vec = Vec::new(); + for row in reader.deserialize::() { + let row = row.map_err(|e| Error::Csv { + path: items_path.to_path_buf(), + source: e, + })?; + meta = MeasurementMeta { + id: row.administration_id.clone(), + assessment: row.assessment.clone(), + term: some(&row.term), + date: parse_date(&row.date), + forms: row + .forms + .split(';') + .filter(|f| !f.is_empty()) + .map(str::to_string) + .collect(), + n_examinees: row.n_examinees, + model: meta.model, + generated: parse_date(&row.generated), + coursebank: some(&row.coursebank), + }; + items.push(Measurement { + item: row.item, + number: row.number, + variant: some(&row.variant), + stem_digest: some(&row.stem_digest), + n: row.n, + p_value: parse(&row.p_value), + point_biserial: parse(&row.point_biserial), + discrimination_index: parse(&row.discrimination_index), + ..Measurement::default() + }); + } + + if options_path.is_file() { + let mut reader = open_csv(&options_path, &OPTION_COLUMNS)?; + for row in reader.deserialize::() { + let row = row.map_err(|e| Error::Csv { + path: options_path.clone(), + source: e, + })?; + let Some(target) = items.iter_mut().find(|i| i.number == row.number) else { + continue; + }; + if row.keyed && !target.key.contains(&row.option) { + target.key.push(row.option.clone()); + } + target.option_stats.insert( + row.option, + OptionStat { + selection_rate: parse(&row.selection_rate), + point_biserial: parse(&row.point_biserial), + upper_group_rate: parse(&row.upper_group_rate), + lower_group_rate: parse(&row.lower_group_rate), + }, + ); + } + } + + Ok(MeasurementFile { + schema_version: default_version(), + administration: meta, + items, + }) + } + + /// Reads every administration under a directory, oldest first. + /// + /// # Arguments + /// + /// * `dir` - usually `analysis/administrations`. + /// + /// # Returns + /// + /// The administrations, empty when the directory does not exist. + /// + /// # Errors + /// + /// Propagates read failures. + pub fn load_all(dir: &Path) -> Result> { + if !dir.is_dir() { + return Ok(Vec::new()); + } + let mut paths: Vec = std::fs::read_dir(dir) + .map_err(|e| Error::io(dir, e))? + .filter_map(|e| e.ok().map(|e| e.path())) + .filter(|p| p.to_string_lossy().ends_with("-items.csv")) + .collect(); + paths.sort(); + + let mut out = Vec::new(); + for path in &paths { + out.push(MeasurementFile::read_csv(path)?); + } + out.sort_by(|a, b| { + a.administration + .date + .cmp(&b.administration.date) + .then(a.administration.id.cmp(&b.administration.id)) + }); + Ok(out) + } +} + +/// Opens a CSV and refuses any column that is not on the allowlist. +fn open_csv(path: &Path, allowed: &[&str]) -> Result> { + let mut reader = csv::Reader::from_path(path).map_err(|e| Error::Csv { + path: path.to_path_buf(), + source: e, + })?; + let headers = reader + .headers() + .map_err(|e| Error::Csv { + path: path.to_path_buf(), + source: e, + })? + .clone(); + + let unexpected: Vec<&str> = headers.iter().filter(|h| !allowed.contains(h)).collect(); + if !unexpected.is_empty() { + return Err(Error::Invalid(vec![format!( + "{}: unexpected column(s) {}. These files are committed, so they hold cohort \ + aggregates and nothing else — anything per-student belongs in data/, which is \ + ignored.", + path.display(), + unexpected.join(", ") + )])); + } + Ok(reader) +} + +/// A float as a CSV cell, empty when absent. +fn number(value: Option) -> String { + value.map(|v| format!("{v}")).unwrap_or_default() +} + +/// A CSV cell as a float, absent when empty or unparseable. +fn parse(cell: &str) -> Option { + cell.trim().parse().ok() +} + +/// A CSV cell as a string, absent when empty. +fn some(cell: &str) -> Option { + (!cell.trim().is_empty()).then(|| cell.trim().to_string()) +} + +/// A CSV cell as a date, absent when empty or unparseable. +fn parse_date(cell: &str) -> Option { + some(cell).and_then(|c| c.parse().ok()) +} + +/// The schema version new files are written with. +fn default_version() -> String { + SCHEMA_VERSION.to_string() +} + +/// A digest shortened for a message. +fn short(digest: &str) -> String { + digest.chars().take(8).collect() +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::item::VariantCalibration; + + fn tmp(tag: &str) -> std::path::PathBuf { + let p = std::env::temp_dir().join(format!("coursebank-cal-{tag}-{}", std::process::id())); + let _ = std::fs::remove_dir_all(&p); + std::fs::create_dir_all(&p).unwrap(); + p + } + + #[test] + fn an_absent_store_is_empty_rather_than_an_error() { + let dir = tmp("absent"); + let store = CalibrationFile::load(&dir.join(CALIBRATION_FILE)).unwrap(); + assert!(store.items.is_empty()); + assert!(store.get("q-x").is_none()); + } + + #[test] + fn the_store_round_trips() { + let dir = tmp("round"); + let path = dir.join(CALIBRATION_FILE); + + let mut store = CalibrationFile::default(); + store.items.insert( + "q-x".to_string(), + Calibration { + n_examinees: Some(27), + p_value: Some(0.63), + variants: vec![VariantCalibration { + variant: "4c81fa".into(), + n_examinees: Some(27), + ..VariantCalibration::default() + }], + ..Calibration::default() + }, + ); + store.save(&path).unwrap(); + + let back = CalibrationFile::load(&path).unwrap(); + assert_eq!(back.get("q-x").unwrap().p_value, Some(0.63)); + assert_eq!(back.get("q-x").unwrap().variants.len(), 1); + let _ = std::fs::remove_dir_all(&dir); + } + + #[test] + fn an_administration_round_trips_through_two_csvs() { + let dir = tmp("csv"); + let file = MeasurementFile { + schema_version: default_version(), + administration: MeasurementMeta { + id: "e1-2026f".into(), + assessment: "e1".into(), + term: Some("2026f".into()), + date: Some(Date::new(2026, 9, 15).unwrap()), + forms: vec!["A".into(), "B".into()], + n_examinees: 27, + model: None, + generated: Some(Date::new(2026, 9, 27).unwrap()), + coursebank: Some("0.0.0".into()), + }, + items: vec![Measurement { + item: "q-fastq-quality-length-match".into(), + number: 1, + variant: Some("237d62f9af222f78".into()), + stem_digest: Some("8b22e0".into()), + n: 27, + p_value: Some(0.5926), + point_biserial: Some(0.31), + discrimination_index: None, + key: vec!["o-fourth-line".into()], + option_stats: [ + ( + "o-fourth-line".to_string(), + OptionStat { + selection_rate: Some(0.5926), + point_biserial: Some(0.31), + upper_group_rate: None, + lower_group_rate: None, + }, + ), + ( + "o-third-line".to_string(), + OptionStat { + selection_rate: Some(0.1852), + point_biserial: Some(-0.18), + upper_group_rate: None, + lower_group_rate: None, + }, + ), + ] + .into_iter() + .collect(), + irt: None, + flags: Vec::new(), + }], + }; + + let written = file.write_csv(&dir).unwrap(); + assert_eq!(written.len(), 2, "one table per shape"); + assert!(written[0].ends_with("e1-2026f-items.csv")); + assert!(written[1].ends_with("e1-2026f-options.csv")); + + // One line per question, loadable on its own: the administration's + // metadata repeats on the row, as the response store already does. + let items = std::fs::read_to_string(&written[0]).unwrap(); + assert!( + items.starts_with("administration_id,assessment,term,date,forms"), + "{items}" + ); + assert!( + items.contains("e1-2026f,e1,2026f,2026-09-15,A;B,27"), + "{items}" + ); + + let options = std::fs::read_to_string(&written[1]).unwrap(); + assert!(options.contains("o-fourth-line,true"), "{options}"); + assert!(options.contains("o-third-line,false"), "{options}"); + + let back = MeasurementFile::read_csv(&written[0]).unwrap(); + assert_eq!(back.administration.id, "e1-2026f"); + assert_eq!(back.administration.n_examinees, 27); + assert_eq!(back.administration.forms, vec!["A", "B"]); + assert_eq!(back.items.len(), 1); + assert_eq!(back.items[0].p_value, Some(0.5926)); + assert_eq!(back.items[0].key, vec!["o-fourth-line"]); + assert_eq!(back.items[0].option_stats.len(), 2); + assert_eq!(back.items[0].discrimination_index, None); + + assert_eq!(MeasurementFile::load_all(&dir).unwrap().len(), 1); + + // It happened once, so the record is not replaced by accident. + let err = file.write_csv(&dir).unwrap_err().to_string(); + assert!(err.contains("deliberate"), "{err}"); + let _ = std::fs::remove_dir_all(&dir); + } + + #[test] + fn a_student_column_is_refused_on_read() { + let dir = tmp("identifiers"); + let path = dir.join("e1-2026f-items.csv"); + std::fs::write( + &path, + "administration_id,assessment,item,number,n,student_key\n e1-2026f,e1,q-x,1,27,abc123\n", + ) + .unwrap(); + + // In YAML this fell out of the type, which rejects unknown keys. A CSV + // reader tolerates extra columns, so the same assurance has to be an + // explicit allowlist — and this is the test that it is one. + let err = MeasurementFile::read_csv(&path).unwrap_err().to_string(); + assert!(err.contains("student_key"), "{err}"); + assert!(err.contains("cohort aggregates"), "{err}"); + let _ = std::fs::remove_dir_all(&dir); + } +} diff --git a/src/model/catalog.rs b/src/model/catalog.rs index 49004e1..3e1a15e 100644 --- a/src/model/catalog.rs +++ b/src/model/catalog.rs @@ -5,7 +5,7 @@ //! Loading a whole course at once, and reporting on what it contains. //! //! A [`Catalog`] is every bank in a course, indexed so that an item can be found -//! by its global id (`bank::item`), and so that questions like "how many Apply +//! by its id, and so that questions like "how many Apply //! level items do I have on lecture 12" have a cheap answer. //! //! The global id is the join key for everything downstream: assessment records @@ -20,19 +20,24 @@ use std::collections::{BTreeMap, BTreeSet}; use std::path::{Path, PathBuf}; -use crate::assessment::AssessmentFile; +use crate::assessment::{AssessmentFile, Placement}; use crate::bank::BankFile; use crate::course::CourseFile; use crate::error::{Error, Result}; use crate::item::Item; use crate::layout::Layout; -use crate::taxonomy::{Level, Status}; +use crate::taxonomy::{Format, Level, Status, Tier}; use crate::yaml; /// One item plus everything needed to locate it again. #[derive(Debug, Clone)] pub struct Entry { - /// The globally unique id, `bank::item`. + /// The item's id, which names it course-wide. + /// + /// The bank is in [`Entry::bank`] and the file in [`Entry::path`], neither + /// of which is part of the identity: this is the join key that response + /// data carries, and a join key with a file name in it renames itself every + /// time the files are reorganized. pub uid: String, /// The bank id. pub bank: String, @@ -70,6 +75,12 @@ pub struct Catalog { pub entries: Vec, /// Bank metadata by bank id. pub banks: BTreeMap, + /// The statistics, loaded from `analysis/` rather than from the banks. + /// + /// Each entry's [`crate::item::Item::calibration`] is filled from this at + /// load, so everything downstream reads one item and does not have to know + /// that the numbers and the wording come from different files. + pub calibration: crate::calibration::CalibrationFile, /// Map from global id to index into `entries`. index: BTreeMap, } @@ -92,14 +103,17 @@ impl Catalog { /// id, since either makes the join key ambiguous. pub fn load(root: &Path) -> Result { let layout = Layout::new(root); - let course = CourseFile::load(&layout.course_file())?; + let course = CourseFile::load_dir(root)?; let mut catalog = Catalog { course, layout, entries: Vec::new(), banks: BTreeMap::new(), index: BTreeMap::new(), + calibration: crate::calibration::CalibrationFile::default(), }; + catalog.calibration = + crate::calibration::CalibrationFile::load(&catalog.layout.calibration_file())?; let mut problems = Vec::new(); let mut files = yaml::list_yaml(&catalog.layout.banks())?; @@ -117,9 +131,13 @@ impl Catalog { } catalog.banks.insert(bank_id.clone(), bank.bank.clone()); for (i, item) in bank.items.into_iter().enumerate() { - let uid = format!("{bank_id}::{}", item.id); - if catalog.index.contains_key(&uid) { - problems.push(format!("duplicate global item id `{uid}`")); + let uid = item.id.clone(); + if let Some(first) = catalog.index.get(&uid) { + problems.push(format!( + "item id `{uid}` is used twice: in bank `{}` and in bank `{bank_id}`. An \ + id names one question course-wide, because response data joins on it.", + catalog.entries[*first].bank + )); continue; } catalog.index.insert(uid.clone(), catalog.entries.len()); @@ -136,20 +154,36 @@ impl Catalog { if !problems.is_empty() { return Err(Error::Invalid(problems)); } + // Statistics are attached here rather than parsed from the bank, which + // is why an item can be read as one thing while its wording and its + // evidence live in files with different review cycles. + for entry in &mut catalog.entries { + if let Some(calibration) = catalog.calibration.items.get(&entry.uid) { + entry.item.calibration = Some(calibration.clone()); + } + } + Ok(catalog) } /// Looks up an item by global id. /// + /// A pre-2.0 `bank::item` id resolves to the item it used to name, so an + /// assessment record or a parquet file written before the change still + /// joins. See [`crate::item::canonical_id`]. + /// /// # Arguments /// - /// * `uid` - the global id, `bank::item`. + /// * `uid` - the item id, in either form. /// /// # Returns /// /// The entry, or `None`. pub fn get(&self, uid: &str) -> Option<&Entry> { - self.index.get(uid).map(|i| &self.entries[*i]) + self.index + .get(uid) + .or_else(|| self.index.get(crate::item::canonical_id(uid))) + .map(|i| &self.entries[*i]) } /// Looks up an item by global id, erroring when absent. @@ -173,45 +207,37 @@ impl Catalog { }) } - /// Resolves a possibly-unqualified id to a global id. + /// Resolves an id written in either form to the canonical one. /// - /// Typing `q-mm-kinetics-001` on the command line should work when that id is - /// unambiguous across the course, because remembering which bank a question - /// lives in is exactly the sort of bookkeeping this tool exists to remove. + /// Since 2.0 an item id is already course-wide, so this is the identity for + /// anything current. What it is still for is the old `bank::item` form, + /// which appears in assessment records, seals, and response files written + /// before the change, and which a person may well still type. /// /// # Arguments /// - /// * `id` - a global id, or a bare item id. + /// * `id` - an item id, in either form. /// /// # Returns /// - /// The global id. + /// The canonical id. /// /// # Errors /// - /// Returns [`Error::Unresolved`] when nothing matches, or [`Error::Usage`] - /// when a bare id matches items in more than one bank. + /// Returns [`Error::Unresolved`] when nothing matches. pub fn resolve(&self, id: &str) -> Result { if self.index.contains_key(id) { return Ok(id.to_string()); } - let matches: Vec<&Entry> = self.entries.iter().filter(|e| e.item.id == id).collect(); - match matches.len() { - 0 => Err(Error::Unresolved { - kind: "item", - id: id.to_string(), - context: None, - }), - 1 => Ok(matches[0].uid.clone()), - _ => Err(Error::usage(format!( - "`{id}` is ambiguous; it exists in {}. Use the full `bank::item` form.", - matches - .iter() - .map(|e| e.bank.as_str()) - .collect::>() - .join(", ") - ))), + let canonical = crate::item::canonical_id(id); + if self.index.contains_key(canonical) { + return Ok(canonical.to_string()); } + Err(Error::Unresolved { + kind: "item", + id: id.to_string(), + context: None, + }) } /// Every item that may be placed on a graded assessment. @@ -232,12 +258,10 @@ impl Catalog { /// /// Problems, prefixed with the file they came from. pub fn validate(&self) -> Result> { - let mut issues: Vec = self - .course - .validate() - .into_iter() - .map(|m| format!("course.yaml: {m}")) - .collect(); + // Not prefixed here: `CourseFile::validate` attributes each message to + // the fragment that defined the id, which for an unsplit course is + // `course.yaml` and for a split one is the file worth opening. + let mut issues: Vec = self.course.validate(); for path in yaml::list_yaml(&self.layout.banks())? { let bank = BankFile::load_resolved(&path)?; @@ -255,6 +279,154 @@ impl Catalog { Ok(issues) } + /// Checks one placement's option set against the item's pool. + /// + /// The checks that moved here from the bank when options became a pool. A + /// bank holding two defensible keys and six distractors is sound; what has + /// to hold for a *form* is that exactly one key reached the student, that + /// none of the distractors was true, and that the count matches policy. + /// None of that can be decided by looking at the item alone. + /// + /// # Arguments + /// + /// * `p` - the placement. + /// * `item` - the item it names. + /// + /// # Returns + /// + /// One message per problem. + fn validate_placement(&self, p: &Placement, item: &Item) -> Vec { + let mut issues = Vec::new(); + if !item.format.has_options() { + return issues; + } + let at = |number: u32| format!("question {number} ({})", p.item); + + for id in p.key.iter().chain(p.distractors.iter()) { + if item.option(id).is_none() { + issues.push(format!( + "{}: `{id}` is not an option of this item", + at(p.number) + )); + } + } + for id in &p.distractors { + if p.key.iter().any(|k| k == id) { + issues.push(format!( + "{}: `{id}` is listed as both the key and a distractor", + at(p.number) + )); + } + if item.option(id).is_some_and(|o| o.correct) { + issues.push(format!( + "{}: `{id}` is offered as a distractor but the bank keys it correct", + at(p.number) + )); + } + } + for id in &p.key { + if item.option(id).is_some_and(|o| !o.correct) { + issues.push(format!( + "{}: `{id}` is keyed correct here but the bank does not key it. An option is \ + true or it is not; which true option a form uses is this record's choice, \ + but not whether it is true.", + at(p.number) + )); + } + } + + let single = item.format == Format::SingleBestAnswer; + if single && p.key.len() > 1 { + issues.push(format!( + "{}: single_best_answer administers exactly one key, this names {}", + at(p.number), + p.key.len() + )); + } + + // No distractor list means the whole pool, which is what a pre-2.0 + // record means and what an item whose pool is its form still means. + // The record still has to say which key, when the pool offers a choice. + if p.distractors.is_empty() { + let (keys, _) = item.pool(); + if p.key.is_empty() && keys.len() > 1 && single { + issues.push(format!( + "{}: the item offers {} defensible keys, so the record has to say which one \ + this assessment used", + at(p.number), + keys.len() + )); + } + } else { + let shown = item.administered(&p.key, &p.distractors); + let expected = self.course.policy.options_per_item; + if shown.len() != expected { + issues.push(format!( + "{}: administers {} option(s), but course policy is {expected} per item", + at(p.number), + shown.len() + )); + } + if single && p.key.is_empty() { + issues.push(format!( + "{}: names its distractors but not its key, so what was marked correct is \ + left to whatever the bank says today", + at(p.number) + )); + } + } + + if let Some(recorded) = &p.variant { + if *recorded != item.variant_digest(&p.key, &p.distractors) { + issues.push(format!( + "{}: an administered option has been reworded since this assessment. \ + Statistics pooled under this variant describe the older wording.", + at(p.number) + )); + } + } + issues + } + + /// Checks every sealed administration's stems against the bank. + /// + /// The seal is the authority on what was administered, so this is the + /// comparison that matters: a record can be edited, but a seal is written + /// before the exam is printed and digested against tampering. A stem that + /// no longer matches the one a cohort answered means the id now names a + /// different question, and every statistic pooled under it is describing + /// two things at once. + /// + /// # Arguments + /// + /// * `seals` - the sealed administrations to check. + /// + /// # Returns + /// + /// One message per stem that has moved out from under its seal. + pub fn validate_seals(&self, seals: &[crate::seal::SealFile]) -> Vec { + let mut issues = Vec::new(); + for seal in seals { + for item in &seal.items { + let Some(digest) = &item.stem_digest else { + continue; + }; + let Some(entry) = self.get(&item.item) else { + continue; + }; + if *digest != entry.item.stem_digest() { + issues.push(format!( + "{}: question {} (`{}`) was administered with a different stem than the \ + bank now holds. Statistics from that administration describe the older \ + wording; give the new wording its own id.", + seal.seal.assessment, item.number, item.item + )); + } + } + } + issues + } + /// Validates a record's internal invariants, then its references against this /// catalog: unknown items, keys that drifted, and fingerprints showing the /// item was reworded since it was administered. @@ -264,6 +436,22 @@ impl Catalog { match self.get(&p.item) { None => issues.push(format!("question {}: unknown item `{}`", p.number, p.item)), Some(entry) => { + // The rule the stem digest exists to enforce. A changed + // fingerprint is a note: the statistics describe an older + // wording. A changed stem is an error: whatever was + // administered is not the question the bank now holds, so + // the id is being reused for two different questions. + if let Some(digest) = &p.stem_digest { + if *digest != entry.item.stem_digest() { + issues.push(format!( + "question {} ({}): the stem has been reworded since this \ + assessment. A reworded stem is a new question: give the new \ + wording a new id with `supersedes: {}`, and leave this one as \ + it was administered.", + p.number, p.item, p.item + )); + } + } if let Some(fp) = &p.fingerprint { if *fp != entry.item.fingerprint() { issues.push(format!( @@ -273,11 +461,7 @@ impl Catalog { )); } } - if !p.key.is_empty() && p.key != entry.item.key_letters() { - issues.push(format!( - "question {} ({}): the recorded key {:?} differs from the item's current key {:?}", - p.number, p.item, p.key, entry.item.key_letters())); - } + issues.extend(self.validate_placement(p, &entry.item)); } } } @@ -321,19 +505,19 @@ impl Catalog { out } - /// Items that measure a given objective. + /// Items tagged with a given learning target. /// /// # Arguments /// - /// * `objective` - the objective id. + /// * `target` - the target id. /// /// # Returns /// /// Matching entries. - pub fn by_objective(&self, objective: &str) -> Vec<&Entry> { + pub fn by_target(&self, target: &str) -> Vec<&Entry> { self.entries .iter() - .filter(|e| e.item.learning_objectives.iter().any(|o| o == objective)) + .filter(|e| e.item.learning_targets.iter().any(|t| t == target)) .collect() } @@ -384,55 +568,124 @@ impl Catalog { out } - /// Builds the coverage report. + /// Items that measure a given objective, through any of its targets. + /// + /// # Arguments + /// + /// * `objective` - the objective id. /// /// # Returns /// - /// One row per assessed objective plus a list of course-wide gaps. + /// Entries tagged with the objective itself or with any of its targets, + /// each appearing once even when it is tagged with two of them. + pub fn by_objective(&self, objective: &str) -> Vec<&Entry> { + self.entries + .iter() + .filter(|e| { + e.item + .learning_targets + .iter() + .any(|t| t == objective || self.course.objective_for(t) == objective) + }) + .collect() + } + + /// Builds the coverage report. + /// + /// Rows cover both tiers, because the two answer different questions. An + /// objective row answers "can I build an exam that reports on this + /// objective", and aggregates every item under it. A target row answers + /// "which specific things have I written items for", which is the + /// authoring queue, and a target with no items is the most common and least + /// visible hole in a bank: the objective looks well covered while a third of + /// what it claims has never been asked. + /// + /// # Returns + /// + /// One row per assessed entry plus a list of course-wide gaps. pub fn coverage(&self) -> Coverage { let mut rows = Vec::new(); - for id in self.course.objectives_in_order() { - let obj = &self.course.learning_objectives[&id]; - if !obj.assessed { + for id in self.course.registry_in_order() { + if !self.course.is_objective(&id) && !self.course.is_target(&id) { continue; } - let items = self.by_objective(&id); + if !self.course.is_assessed(&id) { + continue; + } + let targets = self.course.targets(&id); + let tier = if self.course.is_objective(&id) { + Tier::Objective + } else { + Tier::Target + }; + // An objective's pool is everything under it; a target's is what is + // tagged to it directly. An objective with no targets is its own + // target, so the two agree there. + let items = if targets.is_empty() { + self.by_target(&id) + } else { + self.by_objective(&id) + }; let usable: Vec<&&Entry> = items.iter().filter(|e| e.item.is_assemblable()).collect(); let mut levels: BTreeSet = BTreeSet::new(); for e in &usable { levels.insert(e.item.level); } + let targets_covered = targets + .iter() + .filter(|target| { + self.by_target(target) + .iter() + .any(|e| e.item.is_assemblable()) + }) + .count(); rows.push(CoverageRow { - objective: id.clone(), - text: obj.text.clone(), - unit: obj.unit.clone(), + id: id.clone(), + text: self.course.text_for(&id), + tier, + objective: self + .course + .is_target(&id) + .then(|| self.course.objective_for(&id).to_string()), + unit: self.course.objective_unit(&id).map(str::to_string), + targets: targets.len(), + targets_covered, total: items.len(), assemblable: usable.len(), max_level: levels.iter().next_back().copied(), levels: levels.into_iter().collect(), - ceiling: obj.level_ceiling, + ceiling: self.course.effective_level_ceiling(&id), }); } let mut gaps = Vec::new(); for row in &rows { + // Gaps are raised against the tier that can be acted on. "No items + // at all" is worth saying about a target, because writing one is the + // fix. "Resting on a single item" is worth saying about an + // objective, because a target resting on one item is the normal and + // intended case, and flagging two hundred of them would bury the + // rows that matter. if row.total == 0 { - gaps.push(Gap::Uncovered(row.objective.clone())); + gaps.push(Gap::Uncovered(row.id.clone())); } else if row.assemblable == 0 { - gaps.push(Gap::NoApprovedItems(row.objective.clone())); - } else if row.assemblable == 1 { - gaps.push(Gap::SingleItem(row.objective.clone())); + gaps.push(Gap::NoApprovedItems(row.id.clone())); + } else if row.assemblable == 1 && row.tier == Tier::Objective { + gaps.push(Gap::SingleItem(row.id.clone())); } // An objective assessed only at the recall level is the most common // and most consequential blind spot: it looks covered in a count and // is not covered in fact. - if row.assemblable > 0 && row.max_level == Some(Level::Remember) { + if row.tier == Tier::Objective + && row.assemblable > 0 + && row.max_level == Some(Level::Remember) + { if let Some(ceiling) = row.ceiling { if ceiling > Level::Remember { - gaps.push(Gap::RecallOnly(row.objective.clone())); + gaps.push(Gap::RecallOnly(row.id.clone())); } } else { - gaps.push(Gap::RecallOnly(row.objective.clone())); + gaps.push(Gap::RecallOnly(row.id.clone())); } } } @@ -451,7 +704,7 @@ impl Catalog { // Items with no objective at all cannot appear in any student report. for e in &self.entries { - if e.item.learning_objectives.is_empty() && e.item.is_assemblable() { + if e.item.learning_targets.is_empty() && e.item.is_assemblable() { gaps.push(Gap::ItemWithoutObjective(e.uid.clone())); } } @@ -463,21 +716,35 @@ impl Catalog { /// The coverage report. #[derive(Debug, Clone)] pub struct Coverage { - /// One row per assessed objective. + /// One row per assessed registry entry, objectives first with their targets + /// following each. pub rows: Vec, /// Course-wide gaps worth acting on. pub gaps: Vec, } -/// Coverage of one objective. +/// Coverage of one registry entry, at either tier. #[derive(Debug, Clone)] pub struct CoverageRow { - /// The objective id. - pub objective: String, - /// The objective text. + /// The registry id. + pub id: String, + /// Its text. pub text: String, - /// The unit it belongs to. + /// Whether this is an objective, whose counts aggregate every item under it, + /// or a target, whose counts are its own items. + pub tier: Tier, + /// The objective this row sits under, for a target. + pub objective: Option, + /// The unit it belongs to, inherited from the objective when not declared. pub unit: Option, + /// Targets in the registry, for an objective row. + pub targets: usize, + /// Targets with at least one usable item. + /// + /// The number to look at when an objective looks well covered: twenty items + /// spread over four of its nine targets is a different bank from twenty + /// items spread over all nine. + pub targets_covered: usize, /// Items referencing it, at any status. pub total: usize, /// Items that could actually be used. @@ -486,16 +753,16 @@ pub struct CoverageRow { pub max_level: Option, /// Every level assessed. pub levels: Vec, - /// The declared ceiling, when set. + /// The ceiling in force, inherited from the objective when not declared. pub ceiling: Option, } /// A specific, actionable hole in the item pool. #[derive(Debug, Clone, PartialEq, Eq)] pub enum Gap { - /// An assessed objective with no items at all. + /// An assessed objective or target with no items at all. Uncovered(String), - /// An objective whose items are all drafts or retired. + /// An entry whose items are all drafts or retired. NoApprovedItems(String), /// An objective resting on a single item, so one bad item hides it entirely. SingleItem(String), @@ -606,7 +873,7 @@ items: level: 1 cognitive_process: recall stem: What is x? - learning_objectives: [lo-covered] + learning_targets: [lo-covered] sources: [{ lecture: L01 }] design: { expected_difficulty: 0.8 } options: @@ -622,13 +889,32 @@ items: let cat = Catalog::load(&dir).expect("catalog loads"); assert_eq!(cat.entries.len(), 1); - assert_eq!(cat.entries[0].uid, "b1::q-x-001"); + // The id names the item course-wide; the bank is where it is kept. + assert_eq!(cat.entries[0].uid, "q-x-001"); + assert_eq!(cat.entries[0].bank, "b1"); + assert!(cat.get("q-x-001").is_some()); + assert_eq!(cat.resolve("q-x-001").unwrap(), "q-x-001"); + // A record or a parquet file written before 2.0 still joins. assert!(cat.get("b1::q-x-001").is_some()); - assert_eq!(cat.resolve("q-x-001").unwrap(), "b1::q-x-001"); + assert_eq!(cat.resolve("b1::q-x-001").unwrap(), "q-x-001"); + assert!(cat.get("b1::q-nonexistent").is_none()); assert!(cat.validate().unwrap().is_empty()); let _ = std::fs::remove_dir_all(&dir); } + #[test] + fn one_item_id_in_two_banks_is_fatal() { + let dir = tmp("dupitem"); + write_course(&dir, ""); + write_bank(&dir, "b1.yaml", APPROVED); + write_bank(&dir, "b2.yaml", &APPROVED.replace("id: b1", "id: b2")); + + let message = Catalog::load(&dir).unwrap_err().to_string(); + assert!(message.contains("`q-x-001` is used twice"), "{message}"); + assert!(message.contains("course-wide"), "{message}"); + let _ = std::fs::remove_dir_all(&dir); + } + #[test] fn duplicate_bank_ids_are_fatal() { let dir = tmp("dupbank"); @@ -640,21 +926,6 @@ items: let _ = std::fs::remove_dir_all(&dir); } - #[test] - fn ambiguous_bare_ids_are_rejected() { - let dir = tmp("ambig"); - write_course(&dir, ""); - write_bank(&dir, "a.yaml", APPROVED); - write_bank(&dir, "b.yaml", &APPROVED.replace("id: b1", "id: b2")); - let cat = Catalog::load(&dir).expect("distinct banks load"); - assert_eq!(cat.entries.len(), 2); - let err = cat.resolve("q-x-001").expect_err("bare id is ambiguous"); - assert!(format!("{err}").contains("ambiguous")); - // The fully qualified form still works. - assert_eq!(cat.resolve("b2::q-x-001").unwrap(), "b2::q-x-001"); - let _ = std::fs::remove_dir_all(&dir); - } - #[test] fn coverage_finds_real_gaps() { let dir = tmp("coverage"); @@ -672,6 +943,58 @@ items: let _ = std::fs::remove_dir_all(&dir); } + #[test] + fn coverage_aggregates_targets_and_names_the_untested_ones() { + let dir = tmp("tiered-coverage"); + std::fs::create_dir_all(dir.join("banks")).unwrap(); + std::fs::write( + dir.join("course.yaml"), + r#" +course: { code: TEST 101, title: Testing, term: Fall 2026 } +lectures: + L01: { title: One } +learning_objectives: + lo-binding: { text: Quantify binding., lectures: [L01], order: 1, level_ceiling: 3 } +learning_targets: + t-kd: { text: Write the expression., objective: lo-binding, order: 1 } + t-plot: { text: Read a plot., objective: lo-binding, order: 2 } +"#, + ) + .unwrap(); + // Two items, both on the same target. + let bank = APPROVED.replace("lo-covered", "t-kd").replace( + " - { id: B, text: wrong }\n", + " - { id: B, text: wrong }\n - id: q-x-002\n status: approved\n level: 3\n \ + cognitive_process: implement\n stem: And again?\n learning_targets: [t-kd]\n \ + sources: [{ lecture: L01 }]\n design: { expected_difficulty: 0.5 }\n options:\n \ + - { id: A, text: right, correct: true }\n - { id: B, text: wrong }\n", + ); + write_bank(&dir, "b1.yaml", &bank); + let cat = Catalog::load(&dir).unwrap(); + let cov = cat.coverage(); + + let objective = cov + .rows + .iter() + .find(|r| r.id == "lo-binding") + .expect("the objective row"); + assert_eq!(objective.tier, Tier::Objective); + assert_eq!( + objective.assemblable, 2, + "the objective aggregates its targets" + ); + assert_eq!( + (objective.targets_covered, objective.targets), + (1, 2), + "two items, but only one of the two targets has been asked about" + ); + // Resting on one item is the normal case for a target, so it is not a + // gap; never having been asked about at all is. + assert!(!cov.gaps.contains(&Gap::SingleItem("t-kd".into()))); + assert!(cov.gaps.contains(&Gap::Uncovered("t-plot".into()))); + let _ = std::fs::remove_dir_all(&dir); + } + #[test] fn recall_only_respects_a_recall_ceiling() { let dir = tmp("ceiling"); diff --git a/src/model/course.rs b/src/model/course.rs index a648461..d9746a7 100644 --- a/src/model/course.rs +++ b/src/model/course.rs @@ -4,29 +4,52 @@ //! The course file: identity plus the registries every bank references. //! -//! Learning objectives and lectures are declared once, in `course.yaml`, and -//! referenced by id from items. That is the single most load-bearing decision in +//! Learning objectives, their learning targets, and lectures are declared once +//! and referenced by id from items. That is the single most load-bearing decision in //! the schema. It means an objective's wording lives in exactly one place, so //! rewording it updates every report; it means a report can name what a student //! missed by objective rather than by question number; and it means a dangling //! reference is a hard error instead of a silently misspelled string that splits //! your coverage table into two near-identical rows. //! +//! "Once" is a claim about ids, not about files. [`CourseFile`] is the resolved +//! model, and it may be assembled from a directory of fragments — one file per +//! lecture, one per objective — as well as from a single `course.yaml`. Either +//! way an id has exactly one definition site, and [`CourseFile::origins`] +//! records which file that was. See [`fragment`] for the merge and the rules +//! that keep it honest. +//! //! The course file also declares the term. Items live across terms, so the term //! belongs to the course and the administration, never to the item. -use std::collections::BTreeMap; -use std::path::Path; +pub mod fragment; -use serde::{Deserialize, Serialize}; +use std::collections::BTreeMap; +use std::fmt; +use std::path::{Path, PathBuf}; + +use serde::de::{self, MapAccess, Visitor}; +use serde::ser::SerializeMap; +use serde::{Deserialize, Deserializer, Serialize, Serializer}; use crate::date::Date; use crate::error::{Error, Result}; use crate::taxonomy::Level; use crate::yaml; +use fragment::Section; + /// The schema version this build of the tool writes. -pub const SCHEMA_VERSION: &str = "1.0"; +pub const SCHEMA_VERSION: &str = "2.0"; + +/// The schema major versions this build can read. +/// +/// A 1.0 repository loads unchanged. What 2.0 changes is the shape of two +/// things, and both are tolerated on the way in: a course file may be split +/// into fragments, and an item is named course-wide rather than as +/// `bank::item`. `coursebank migrate` rewrites files into the 2.0 form when you +/// are ready; nothing forces it. +pub const SUPPORTED_MAJORS: [&str; 2] = ["1", "2"]; /// The canonical file name inside a course directory. pub const COURSE_FILE: &str = "course.yaml"; @@ -58,12 +81,46 @@ pub struct CourseFile { pub lectures: BTreeMap, /// Learning objectives, keyed by id such as `lo-mm-kinetics`. + /// + /// The tier a syllabus lists and a report classifies. Objectives only: the + /// performances they are met by live in [`CourseFile::learning_targets`]. #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] pub learning_objectives: BTreeMap, + /// Learning targets, keyed by id such as `t-mm-kcat-from-plot`. + /// + /// The tier items are tagged to. Each names the objective it belongs to, so + /// the two tiers are separate sections rather than one section with a field + /// distinguishing them: reading `course.yaml` you can see the twelve claims + /// the course makes without scrolling past the two hundred performances they + /// are built from, and a target cannot accidentally be written as an + /// objective by leaving a field out. + /// + /// Target ids are deliberately spelled differently from objective ids — no + /// `lo` prefix — so that any id appearing in an item, a reading, or a report + /// says which tier it belongs to without a lookup. + #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] + pub learning_targets: BTreeMap, + + /// Works the course cites, keyed by citation key such as + /// `kuriyan2013molecules`. Readings point in here rather than restating a + /// citation, so a reference is written once and a changed edition is one edit. + #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] + pub references: BTreeMap, + /// Shared stimuli for case-based testlets, keyed by id. #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] pub stimuli: BTreeMap, + + /// Which file defined each id, relative to the course root. + /// + /// Populated by [`fragment::assemble`] and empty for a course parsed + /// straight out of one file by [`CourseFile::load`]. It is what makes a + /// validation message able to name the file to open, which matters rather a + /// lot once one course is forty files. Not serialized: it describes where + /// the model came from, not what it says. + #[serde(skip)] + pub origins: BTreeMap<(Section, String), PathBuf>, } /// Course identity. @@ -128,6 +185,58 @@ pub struct Policy { /// The fewest items on an objective before a report will call it mastered. #[serde(default = "two_usize")] pub min_items_for_mastery: usize, + /// The letter-grade bands, highest first or in any order. + /// + /// Empty by default, because a grading scale belongs to a course rather than + /// to a tool. When it is set, a class report bins the score distribution by + /// letter instead of by ten-point interval, which is the only binning a + /// student or an instructor actually acts on. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub grade_scale: Vec, +} + +/// One letter-grade band. +/// +/// Only the lower bound is recorded. An upper bound would be a second copy of +/// the next band's lower bound, and the two would eventually disagree: a scale +/// written as `93.0 - 96.9` leaves 96.95 in no band at all. Bands are read as +/// "this letter or better from here up", so the top band needs no ceiling. +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct GradeBand { + /// The letter as it appears on a transcript. + pub letter: String, + /// The lowest percentage that earns it, inclusive. + pub min: f64, + /// The grade points it carries, when the course records them. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub gpa: Option, + /// The attainment word attached to the band, such as `Meritorious`. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub attainment: Option, + /// A colour group, so a report can tint A bands alike without parsing + /// letters. Defaults to the letter's first character. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub group: Option, +} + +impl GradeBand { + /// The group a band belongs to: its own `group`, else its first character. + /// + /// # Returns + /// + /// An uppercase group key such as `A`. + pub fn group_key(&self) -> String { + match &self.group { + Some(group) => group.to_ascii_uppercase(), + None => self + .letter + .chars() + .next() + .map(|c| c.to_ascii_uppercase().to_string()) + .unwrap_or_default(), + } + } } impl Default for Policy { @@ -140,10 +249,44 @@ impl Default for Policy { partial_credit_floor_level: None, mastery_threshold: mastery_default(), min_items_for_mastery: 2, + grade_scale: Vec::new(), } } } +impl Policy { + /// The grade bands, highest lower bound first. + /// + /// # Returns + /// + /// The bands in descending order, empty when the course sets no scale. + pub fn bands(&self) -> Vec<&GradeBand> { + let mut out: Vec<&GradeBand> = self.grade_scale.iter().collect(); + out.sort_by(|a, b| { + b.min + .partial_cmp(&a.min) + .unwrap_or(std::cmp::Ordering::Equal) + }); + out + } + + /// The band a percentage falls in. + /// + /// # Arguments + /// + /// * `percent` - a score out of 100. + /// + /// # Returns + /// + /// The band, or `None` when the course sets no scale or the score sits below + /// every band in it. + pub fn band_for(&self, percent: f64) -> Option<&GradeBand> { + self.bands() + .into_iter() + .find(|band| percent + 1e-9 >= band.min) + } +} + /// A unit or module of the course. #[derive(Debug, Clone, Serialize, Deserialize)] #[serde(deny_unknown_fields)] @@ -172,12 +315,479 @@ pub struct Lecture { /// Where the slides live, for study guidance in student reports. #[serde(default, skip_serializing_if = "Option::is_none")] pub slides_url: Option, - /// Assigned readings for the session. + /// The objectives this session develops. + /// + /// The registration direction: you write what a lecture covers while + /// planning the lecture, and each named objective gains this lecture in its + /// [`Objective::lectures`] list during [`fragment::assemble`]. Declaring the + /// pair from the objective's side instead is equivalent, and declaring it + /// from both is redundant rather than contradictory — the two are unioned. #[serde(default, skip_serializing_if = "Vec::is_empty")] - pub readings: Vec, + pub teaches: Vec, + /// Assigned readings for the session, in the order you assign them. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub readings: Vec, } -/// A learning objective. +/// A work the course cites: a textbook, an article, a dataset, a recording. +/// +/// Keyed by citation key, so this registry is a bibliography rather than a second +/// naming scheme. The field names follow BibTeX where BibTeX has one, which makes +/// import and export mechanical. +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct Reference { + /// The short form a reading list shows, such as `KKW`. Unique across the + /// registry, because reports print it in place of the key. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub label: Option, + /// What kind of work this is, which decides how a citation renders. + #[serde(default)] + pub kind: ReferenceKind, + /// Whether the course requires it or lists it as background. + #[serde(default)] + pub role: ReferenceRole, + /// Full title. + pub title: String, + /// Authors as `Family, Given`, in the order printed on the work. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub authors: Vec, + /// Year of publication. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub year: Option, + /// Edition as printed: `7th`, `Revised`. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub edition: Option, + /// Publisher, for a book. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub publisher: Option, + /// The journal, edited volume, or series this sits inside. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub container: Option, + /// Volume within the container. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub volume: Option, + /// Issue within the volume. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub issue: Option, + /// Page range of the work as a whole, not of any one reading. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub pages: Option, + /// DOI, bare: `10.1038/nature12373`. + /// + /// For a manuscript this is usually the only link worth storing: it is the + /// identifier of the work rather than of one copy of it, and [`Reference::href`] + /// turns it into a URL. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub doi: Option, + /// arXiv id, bare: `2301.00001` or `q-bio/0501001`. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub arxiv: Option, + /// PubMed Central id, which hosts the full text: `PMC3084216`. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub pmcid: Option, + /// PubMed id, which hosts a record about the work: `21471563`. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub pmid: Option, + /// ISBN, for a book. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub isbn: Option, + /// Canonical URL for the work as a whole. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub url: Option, + /// Prefix a reading's `path` is appended to. Having this means the citation + /// key appears once rather than once per reading. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub base_url: Option, + /// Anything students need to know about getting hold of it: reserve shelf, + /// license, paywall. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub note: Option, +} + +impl Reference { + /// Where to send a reader, most specific first. + /// + /// The one link resolution in the crate. Every exporter used to carry its + /// own copy of the `base_url` join, which meant a reading list, a printed + /// key, a practice sheet, and a student report could disagree about where a + /// citation points — and that none of them linked a journal article, since + /// an article has no `base_url` to join a path to. + /// + /// The order is from the exact location outward: a link to §1.4 beats a link + /// to the work, and a link to the work beats nothing. + /// + /// # Arguments + /// + /// * `url` - a full URL for the exact location, from a reading or citation. + /// * `path` - a location under this work's `base_url`. + /// + /// # Returns + /// + /// The most specific link available, or `None` for a work with no online + /// location at all. + pub fn href(&self, url: Option<&str>, path: Option<&str>) -> Option { + if let Some(url) = url { + return Some(url.to_string()); + } + if let (Some(base), Some(path)) = (self.base_url.as_deref(), path) { + return Some(join_url(base, path)); + } + if let Some(url) = &self.url { + return Some(url.clone()); + } + self.identifier_url() + } + + /// The link this work's identifiers resolve to, ignoring any location inside + /// it. + /// + /// DOI first, because it names the work rather than one copy of it. Then + /// arXiv and PubMed Central, which host the article itself, before PubMed, + /// which hosts a record about it. + /// + /// # Returns + /// + /// A URL, or `None` when the work carries no identifier. + pub fn identifier_url(&self) -> Option { + if let Some(doi) = self.doi.as_deref().map(bare_doi) { + return Some(format!("https://doi.org/{doi}")); + } + if let Some(id) = self.arxiv.as_deref().map(bare_arxiv) { + return Some(format!("https://arxiv.org/abs/{id}")); + } + if let Some(id) = self.pmcid.as_deref().map(str::trim) { + let id = if id.starts_with("PMC") { + id.to_string() + } else { + format!("PMC{id}") + }; + return Some(format!("https://www.ncbi.nlm.nih.gov/pmc/articles/{id}/")); + } + if let Some(id) = self.pmid.as_deref().map(str::trim) { + return Some(format!("https://pubmed.ncbi.nlm.nih.gov/{id}/")); + } + None + } + + /// The short form a reading list shows: the label, or the citation key. + /// + /// # Arguments + /// + /// * `key` - the citation key, used when the work declares no label. + /// + /// # Returns + /// + /// The label to print. + pub fn label_or<'a>(&'a self, key: &'a str) -> &'a str { + self.label.as_deref().unwrap_or(key) + } +} + +/// A DOI with any resolver prefix stripped, so `href` cannot produce +/// `https://doi.org/https://doi.org/10...`. +fn bare_doi(doi: &str) -> &str { + let doi = doi.trim(); + for prefix in [ + "https://doi.org/", + "http://doi.org/", + "https://dx.doi.org/", + "http://dx.doi.org/", + "doi:", + ] { + if let Some(rest) = doi.strip_prefix(prefix) { + return rest; + } + } + doi +} + +/// An arXiv id with the `arXiv:` prefix stripped. +fn bare_arxiv(id: &str) -> &str { + let id = id.trim(); + for prefix in ["arXiv:", "arxiv:", "https://arxiv.org/abs/"] { + if let Some(rest) = id.strip_prefix(prefix) { + return rest; + } + } + id +} + +/// Joins a base URL and a path without doubling or dropping the separator. +fn join_url(base: &str, path: &str) -> String { + match (base.ends_with('/'), path.starts_with('/')) { + (true, true) => format!("{base}{}", &path[1..]), + (false, false) => format!("{base}/{path}"), + _ => format!("{base}{path}"), + } +} + +/// The kind of work, chosen to map onto BibTeX entry types. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)] +#[serde(rename_all = "kebab-case")] +pub enum ReferenceKind { + /// A whole book. + #[default] + Book, + /// A chapter in an edited volume. + Chapter, + /// A journal article. + Article, + /// A preprint, which is an article without a container. + Preprint, + /// A thesis or dissertation. + Thesis, + /// A page or resource that exists only online. + Website, + /// A program or library. + Software, + /// A published dataset. + Dataset, + /// A recording. + Video, + /// Anything else. + Other, +} + +/// Whether the course requires a work or offers it as background. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)] +#[serde(rename_all = "kebab-case")] +pub enum ReferenceRole { + /// A course text. Assigned readings come from it. + Required, + /// Listed so students know it exists. Never assigned. + #[default] + Supplemental, +} + +/// One assigned location inside a [`Reference`], and what it is assigned for. +/// +/// The prose splits three ways because each part answers a different question and +/// each has a different consumer. `summary` says what the section contains, `focus` +/// says what to take from it, and `skip` says what to ignore. A student report +/// quotes `focus` at somebody who missed the objective; a lecture page prints all +/// three. +/// +/// A reading written as a bare string, which is what this field held before the +/// schema existed, still parses: the whole string lands in `text`, and serializing +/// writes it back out as a string rather than a mapping. +#[derive(Debug, Clone, Default)] +pub struct Reading { + /// Citation key into [`CourseFile::references`]. + pub reference: Option, + /// Where inside the work: `§6.1`, `pp. 212-219`, `ch. 3`, `fig. 4`. + pub locator: Option, + /// Appended to the reference's `base_url` to reach this location. + pub path: Option, + /// A full URL, for a location that is not under the reference's `base_url`. + pub url: Option, + /// Whether it is assigned or offered alongside. + pub role: ReadingRole, + /// The learning targets this reading serves. + pub targets: Vec, + /// What the section contains. + pub summary: Option, + /// What to take from it, which is the sentence a study suggestion quotes. + pub focus: Option, + /// What to gloss, and why it is out of scope. + pub skip: Option, + /// A reading written as a bare string before this schema existed, held + /// unparsed. + pub text: Option, +} + +/// Whether a reading is assigned or offered alongside. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)] +#[serde(rename_all = "kebab-case")] +pub enum ReadingRole { + /// Assigned, and therefore fair to assess. + #[default] + Assigned, + /// Offered as background. Not separately assessed. + Supplemental, +} + +impl Reading { + /// The URL for this location. + /// + /// # Arguments + /// + /// * `reference` - the work this reading is inside. + /// + /// # Returns + /// + /// The most specific link available, which for a manuscript with a DOI and + /// no `path` is the DOI. See [`Reference::href`]. + pub fn resolve_url(&self, reference: &Reference) -> Option { + reference.href(self.url.as_deref(), self.path.as_deref()) + } + + /// A short citation for a report: `KKW §6.1`. + /// + /// # Arguments + /// + /// * `key` - the citation key, used when the reference declares no label. + /// * `reference` - the work, for its label. + /// + /// # Returns + /// + /// The label and locator, or the unparsed `text` for a legacy reading. + pub fn cite(&self, key: &str, reference: &Reference) -> String { + if let Some(text) = &self.text { + return text.clone(); + } + let label = reference.label_or(key); + match &self.locator { + Some(locator) => format!("{label} {locator}"), + None => label.to_string(), + } + } +} + +/// Writes a reading as a mapping, or as a bare string when that is all it holds. +/// +/// The string case keeps a course file that predates this schema byte-identical +/// through a load-and-save cycle, so migrating is something you choose rather than +/// something the tool does to your file the first time it writes it. +impl Serialize for Reading { + fn serialize(&self, s: S) -> std::result::Result { + if let Some(text) = &self.text { + if self.reference.is_none() && self.locator.is_none() && self.targets.is_empty() { + return s.serialize_str(text); + } + } + let mut map = s.serialize_map(None)?; + if let Some(v) = &self.reference { + map.serialize_entry("ref", v)?; + } + if let Some(v) = &self.locator { + map.serialize_entry("locator", v)?; + } + if let Some(v) = &self.path { + map.serialize_entry("path", v)?; + } + if let Some(v) = &self.url { + map.serialize_entry("url", v)?; + } + if self.role != ReadingRole::Assigned { + map.serialize_entry("role", &self.role)?; + } + if !self.targets.is_empty() { + map.serialize_entry("targets", &self.targets)?; + } + if let Some(v) = &self.summary { + map.serialize_entry("summary", v)?; + } + if let Some(v) = &self.focus { + map.serialize_entry("focus", v)?; + } + if let Some(v) = &self.skip { + map.serialize_entry("skip", v)?; + } + if let Some(v) = &self.text { + map.serialize_entry("text", v)?; + } + map.end() + } +} + +/// Accepts a reading written either as a mapping or as a bare string. +/// +/// The string form is what `readings` held before this schema, so course files +/// written against the old shape keep loading. It is the same courtesy +/// [`yaml::flexible_string`] extends to an unquoted `schema_version: 1.0`. +impl<'de> Deserialize<'de> for Reading { + fn deserialize>(d: D) -> std::result::Result { + /// The mapping form, with the field set kept in one place. + #[derive(Deserialize)] + #[serde(deny_unknown_fields)] + struct Mapping { + #[serde(rename = "ref", default)] + reference: Option, + #[serde(default)] + locator: Option, + #[serde(default)] + path: Option, + #[serde(default)] + url: Option, + #[serde(default)] + role: ReadingRole, + #[serde(default, alias = "objectives")] + targets: Vec, + #[serde(default)] + summary: Option, + #[serde(default)] + focus: Option, + #[serde(default)] + skip: Option, + #[serde(default)] + text: Option, + } + + struct V; + impl<'a> Visitor<'a> for V { + type Value = Reading; + + fn expecting(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str("a reading mapping with a `ref`, or a plain citation string") + } + + fn visit_str(self, v: &str) -> std::result::Result { + Ok(Reading { + text: Some(v.to_string()), + ..Reading::default() + }) + } + + fn visit_map>(self, map: M) -> std::result::Result { + let m = Mapping::deserialize(de::value::MapAccessDeserializer::new(map))?; + Ok(Reading { + reference: m.reference, + locator: m.locator, + path: m.path, + url: m.url, + role: m.role, + targets: m.targets, + summary: m.summary, + focus: m.focus, + skip: m.skip, + text: m.text, + }) + } + } + d.deserialize_any(V) + } +} + +/// A learning objective: the tier a claim is made about. +/// +/// # On objectives and targets +/// +/// The vocabulary follows the assessment literature, where the two words name +/// two different jobs rather than two sizes of the same thing. +/// +/// A **learning objective** is what a syllabus lists, what a blueprint requires +/// items against, and what a report classifies as met or not met. It is the unit +/// a claim is made about. +/// +/// A **learning target** is the specific performance an item is written against +/// and tagged to: what a student aims at in one class and what one question can +/// actually measure. Targets are the evidence a claim about an objective rests +/// on. +/// +/// The distinction earns its keep because the two tiers cannot be the same +/// thing. An objective specific enough to write a good item against is too +/// specific to report on, since three hundred of them means one question each +/// and one question supports no claim. An objective broad enough to report on +/// says nothing about what the item should ask. +/// +/// The two live in separate registries, and [`Target`] is a separate type that +/// names its objective in a required field. That makes the two-tier depth a +/// property of the schema rather than a rule the validator has to enforce: +/// there is nowhere for a third level to be written. A deeper tree would make +/// "met this objective" ambiguous, because the answer would depend on which +/// level you rolled up to, and it would put one item in three or four different +/// denominators. #[derive(Debug, Clone, Serialize, Deserialize)] #[serde(deny_unknown_fields)] pub struct Objective { @@ -190,9 +800,34 @@ pub struct Objective { /// The lectures that develop it. #[serde(default, skip_serializing_if = "Vec::is_empty")] pub lectures: Vec, + /// Position in teaching order, low first. Derived; authoring it is + /// deprecated. + /// + /// The registry is a map, so declaration order is lost on load and sorting + /// by id would put `lo-enthalpy` before `lo-first-law` when the second is a + /// prerequisite of the first. Something has to supply the order. + /// + /// Since 2.0 that something is [`Lecture::teaches`], which is a sequence: + /// the position of an objective in the list of what a lecture covers, and + /// the position of the lecture in the course, together say when it is + /// taught. [`fragment::assemble`] fills this in from those two, so a course + /// that used to carry thirty-nine hand-kept integers now carries none, and + /// inserting an objective is a one-line edit rather than a renumber. + /// + /// An authored value still wins, so a 1.0 course loads unchanged. + /// `coursebank migrate order` removes them. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub order: Option, + /// The highest level you intend to assess this objective at. Assembling an /// item above the ceiling is a warning: either the item overreaches or the /// ceiling needs raising. + /// + /// A target without one of its own inherits this; see + /// [`CourseFile::effective_level_ceiling`]. So an objective's ceiling is the + /// ceiling of everything under it, which is the useful reading: you decide + /// once that recognition is assessed no higher than Analyze, and every + /// target you add afterwards is checked against that. #[serde(default, skip_serializing_if = "Option::is_none")] pub level_ceiling: Option, /// Objectives that must be secure before this one is reachable. Student @@ -203,10 +838,77 @@ pub struct Objective { #[serde(default, skip_serializing_if = "Vec::is_empty")] pub tags: Vec, /// Whether this objective is assessed at all, or is aspirational. + /// + /// Marking an objective unassessed exempts its targets too; see + /// [`CourseFile::is_assessed`]. It is the one-line way to park a whole topic + /// you taught but decided not to test, without editing twenty targets. #[serde(default = "yes")] pub assessed: bool, } +/// A learning target: one performance an item can be written against. +/// +/// Targets are what items are tagged to, what readings are cited against, and +/// what a report names when explaining why an objective came out the way it did. +/// Results roll up to [`Target::objective`], so a target's own rate is evidence +/// rather than a claim: it usually rests on one or two questions. +/// +/// It is a distinct type from [`Objective`] rather than the same type with a +/// nullable link, so that the objective it belongs to is required by the schema +/// and a third tier has nowhere to be written. +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct Target { + /// The target as you would state it to students. Reports quote this + /// verbatim, so write it in the second person and start with a verb. + pub text: String, + /// The objective this target belongs to. + /// + /// Required: a target with no objective would be measured and never + /// reported, since every statistic a report makes is computed per objective + /// from the items tagged to its targets. + pub objective: String, + /// The lectures that develop it. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub lectures: Vec, + /// Position among the other targets of the same objective, low first. + /// + /// Ordered within its objective rather than across the course, so two + /// targets under different objectives never compete for a position and + /// Retained only so a pre-2.0 course still loads. Ignored. + /// + /// Targets do not have an order. An objective's targets are a set of + /// question templates, not steps in a sequence: they are not taught in + /// order, an exam samples from them rather than working through them, and + /// the study workflow reads the list as a checklist and counts what it can + /// do cold. A position would assert a sequence that does not exist. + /// + /// Where a list has to be printed, [`CourseFile::targets`] orders it by + /// ceiling and then by id. Where one target genuinely depends on another, + /// that is [`Target::prerequisites`], which says so directly. + /// + /// `coursebank migrate order` removes it. + #[serde(default, skip_serializing)] + pub order: Option, + /// The highest level you intend to assess this target at. Omit it to inherit + /// the objective's. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub level_ceiling: Option, + /// Targets or objectives that must be secure before this one is reachable. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub prerequisites: Vec, + /// Free-form tags. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub tags: Vec, + /// Whether this target is assessed at all, or is aspirational. An + /// unassessed objective exempts its targets regardless of this. + #[serde(default = "yes")] + pub assessed: bool, +} + +/// The prefix an objective id is expected to carry. +pub const OBJECTIVE_PREFIX: &str = "lo"; + /// A shared stem or vignette used by several items. #[derive(Debug, Clone, Serialize, Deserialize)] #[serde(deny_unknown_fields)] @@ -241,7 +943,12 @@ impl CourseFile { yaml::read(path) } - /// Finds and loads the course file for a course directory. + /// Loads the course for a course directory, merging every fragment it holds. + /// + /// This is the entry point every command uses. A directory holding only + /// `course.yaml` gives the same result it always did; one that also holds + /// `lectures/`, `objectives/`, or `references.yaml` gets them merged in. See + /// [`fragment::assemble`]. /// /// # Arguments /// @@ -249,34 +956,167 @@ impl CourseFile { /// /// # Returns /// - /// The parsed course file. + /// The merged course file. /// /// # Errors /// - /// Propagates load errors, including absence of `course.yaml`. + /// Propagates load errors, including absence of `course.yaml`, and returns + /// [`Error::Invalid`] when two files define the same id. pub fn load_dir(dir: &Path) -> Result { - CourseFile::load(&dir.join(COURSE_FILE)) + fragment::assemble(dir) + } + + /// Which file defined an id, and which registry it was in. + /// + /// # Arguments + /// + /// * `id` - a unit, lecture, objective, target, reference, or stimulus id. + /// + /// # Returns + /// + /// The section and the path relative to the course root, or `None` for an + /// unknown id or a course that was not assembled from fragments. + pub fn origin(&self, id: &str) -> Option<(Section, &Path)> { + Section::ALL.iter().find_map(|section| { + self.origins + .get(&(*section, id.to_string())) + .map(|path| (*section, path.as_path())) + }) + } + + /// Every file this course was assembled from, in sorted order. + /// + /// # Returns + /// + /// The paths relative to the course root, empty for a course parsed from a + /// single file by [`CourseFile::load`]. + pub fn fragment_paths(&self) -> Vec<&Path> { + let mut paths: Vec<&Path> = self.origins.values().map(PathBuf::as_path).collect(); + paths.sort_unstable(); + paths.dedup(); + paths } /// Writes the course file back out as YAML. /// + /// Refuses to write a course that was assembled from more than one file, + /// because the merged model has no home on disk: writing it to + /// `course.yaml` would leave every fragment defining ids the root file also + /// defines, which is the one thing [`fragment::assemble`] treats as an + /// error. Use [`CourseFile::write_resolved`] for an inspection copy. + /// /// # Arguments /// /// * `path` - destination path. /// /// # Errors /// - /// Returns [`Error::Io`] on a write failure. + /// Returns [`Error::Usage`] for a fragmented course and [`Error::Io`] on a + /// write failure. pub fn save(&self, path: &Path) -> Result<()> { + let sources = self.fragment_paths(); + if sources.len() > 1 { + return Err(Error::usage(format!( + "this course is assembled from {} files, so it cannot be written back to one. \ + Edit the fragment that owns what you are changing, or use `coursebank course \ + build` for a merged copy.", + sources.len() + ))); + } yaml::write(path, self) } + /// Writes the merged course as YAML, for reading rather than for loading. + /// + /// The output carries a banner saying so. It is what `coursebank course + /// build` writes, and nothing in the tool reads it back: a generated file + /// that commands depend on is a file that goes stale. + /// + /// # Arguments + /// + /// * `path` - destination path. + /// + /// # Errors + /// + /// Returns [`Error::Io`] on a write failure, or [`Error::Other`] if the + /// model cannot be represented as YAML. + pub fn write_resolved(&self, path: &Path) -> Result<()> { + let body = yaml::to_string(self)?; + let banner = format!( + "# Generated by `coursebank course build` from {} file(s). Do not edit: nothing\n\ + # reads this, and the next build overwrites it. Edit the fragments instead.\n", + self.fragment_paths().len().max(1) + ); + yaml::write_text(path, &format!("{banner}{body}")) + } + /// Checks internal consistency of the registries. /// + /// Each message is prefixed with the file that defined the id it is about, + /// when that is known. For an unsplit course that is always `course.yaml`, + /// which is what the messages used to say. + /// /// # Returns /// /// Every problem found, empty when the file is sound. pub fn validate(&self) -> Vec { + self.problems() + .into_iter() + .map(|issue| self.attribute(issue)) + .collect() + } + + /// Prefixes one validation message with the fragment it concerns. + /// + /// The id is taken from the first backticked token in the message, since + /// every message that is about a registry entry names it first. Messages + /// about `course`, `policy`, or `units` are attributed by section instead, + /// because the first thing they quote is a field or a grade letter. + /// + /// # Arguments + /// + /// * `issue` - the message. + /// + /// # Returns + /// + /// The message, prefixed with a path when one is known. + fn attribute(&self, issue: String) -> String { + let section = if issue.starts_with("course.") { + Some(Section::Course) + } else if issue.starts_with("policy.") { + Some(Section::Policy) + } else if issue.starts_with("units") { + Some(Section::Units) + } else { + None + }; + + let path = match section { + Some(section) => self.section_origin(section), + None => issue + .split('`') + .nth(1) + .and_then(|id| self.origin(id)) + .map(|(_, path)| path), + }; + + match path { + Some(path) => format!("{}: {issue}", path.display()), + None => issue, + } + } + + /// The file that declared a whole section, for the sections that are not + /// keyed by id. + fn section_origin(&self, section: Section) -> Option<&Path> { + self.origins + .iter() + .find(|((s, _), _)| *s == section) + .map(|(_, path)| path.as_path()) + } + + /// The validation messages, before they are attributed to files. + fn problems(&self) -> Vec { let mut issues = Vec::new(); if self.course.code.trim().is_empty() { @@ -295,6 +1135,49 @@ impl CourseFile { )); } + // A scale with a hole in it silently drops students into no band at all, + // and the report would show a distribution that does not sum to the + // class. Cheaper to say so here. + let mut seen_letters: BTreeMap<&str, usize> = BTreeMap::new(); + let mut seen_mins: Vec = Vec::new(); + for band in &self.policy.grade_scale { + *seen_letters.entry(band.letter.as_str()).or_insert(0) += 1; + if !(0.0..=100.0).contains(&band.min) { + issues.push(format!( + "policy.grade_scale: band `{}` has min {}, which is not a percentage", + band.letter, band.min + )); + } + if seen_mins.iter().any(|m| (m - band.min).abs() < 1e-9) { + issues.push(format!( + "policy.grade_scale: two bands start at {}%, so the lower one is unreachable", + band.min + )); + } + seen_mins.push(band.min); + } + for (letter, n) in &seen_letters { + if *n > 1 { + issues.push(format!( + "policy.grade_scale: duplicate letter `{letter}` declared {n} times" + )); + } + } + if !self.policy.grade_scale.is_empty() { + let lowest = self + .policy + .bands() + .last() + .map(|b| b.min) + .unwrap_or(f64::INFINITY); + if lowest > 0.0 { + issues.push(format!( + "policy.grade_scale: the lowest band starts at {lowest}%, so a score below \ + that falls in no band. Give the failing grade a min of 0." + )); + } + } + let unit_ids: Vec<&String> = self.units.iter().map(|u| &u.id).collect(); let mut unit_counts: BTreeMap<&str, usize> = BTreeMap::new(); for u in &self.units { @@ -306,6 +1189,71 @@ impl CourseFile { } } + let mut labels: BTreeMap<&str, Vec<&str>> = BTreeMap::new(); + for (key, reference) in &self.references { + if reference.title.trim().is_empty() { + issues.push(format!("reference `{key}`: empty title")); + } + // Checked rather than silently coerced: `href` strips a resolver + // prefix, but something that is not a DOI at all would become a + // link that 404s on a student's reading list. + if let Some(doi) = &reference.doi { + if !bare_doi(doi).starts_with("10.") { + issues.push(format!( + "reference `{key}`: `{doi}` is not a DOI. Write it bare, as \ + 10.1038/nature12373." + )); + } + } + // A manuscript with no journal is a citation nobody can print. The + // fields exist; a note that carries them instead is data the reading + // list cannot link and the bibliography exporters cannot use. + if matches!( + reference.kind, + ReferenceKind::Article | ReferenceKind::Preprint + ) && reference.container.is_none() + { + issues.push(format!( + "reference `{key}`: an {} needs a `container` — the journal, preprint \ + server, or proceedings it appeared in. Run `coursebank migrate references` \ + if it is sitting in the `note`.", + match reference.kind { + ReferenceKind::Preprint => "preprint", + _ => "article", + } + )); + } + if let Some(note) = &reference.note { + if crate::citation::looks_like_a_citation(note) { + issues.push(format!( + "reference `{key}`: the note still carries a citation. Volume, pages, \ + and DOI have their own fields, and a DOI in a note is a link nobody \ + can follow. `coursebank migrate references` takes it apart." + )); + } + } + if let Some(pmid) = &reference.pmid { + if !pmid.trim().chars().all(|c| c.is_ascii_digit()) { + issues.push(format!( + "reference `{key}`: pmid `{pmid}` is not a number. A `PMC...` id goes in \ + `pmcid`." + )); + } + } + if let Some(label) = &reference.label { + labels.entry(label.as_str()).or_default().push(key); + } + } + for (label, keys) in &labels { + if keys.len() > 1 { + issues.push(format!( + "references: `{label}` is the label of {}; a label has to name one work \ + because reports print it instead of the key", + keys.join(" and ") + )); + } + } + for (id, lec) in &self.lectures { if lec.title.trim().is_empty() { issues.push(format!("lecture `{id}`: empty title")); @@ -315,23 +1263,41 @@ impl CourseFile { issues.push(format!("lecture `{id}`: unknown unit `{u}`")); } } + for objective in &lec.teaches { + if !self.learning_objectives.contains_key(objective) { + if self.learning_targets.contains_key(objective) { + issues.push(format!( + "lecture `{id}`: `teaches` names the target `{objective}`, but it \ + registers objectives. A target is reached through its objective." + )); + } else { + issues.push(format!( + "lecture `{id}`: `teaches` names an unknown objective `{objective}`" + )); + } + } + } + let mut seen: Vec<(&str, &str)> = Vec::new(); + for (index, reading) in lec.readings.iter().enumerate() { + issues.extend(self.reading_issues(id, index, reading, &mut seen)); + } } - for (id, lo) in &self.learning_objectives { - if lo.text.trim().is_empty() { + for (id, objective) in &self.learning_objectives { + if objective.text.trim().is_empty() { issues.push(format!("objective `{id}`: empty text")); } - if let Some(u) = &lo.unit { + if let Some(u) = &objective.unit { if !unit_ids.contains(&u) { issues.push(format!("objective `{id}`: unknown unit `{u}`")); } } - for lec in &lo.lectures { + for lec in &objective.lectures { if !self.lectures.contains_key(lec) { issues.push(format!("objective `{id}`: unknown lecture `{lec}`")); } } - for pre in &lo.prerequisites { + for pre in &objective.prerequisites { if !self.learning_objectives.contains_key(pre) { issues.push(format!( "objective `{id}`: unknown prerequisite objective `{pre}`" @@ -343,10 +1309,182 @@ impl CourseFile { } } + for (id, target) in &self.learning_targets { + issues.extend(self.target_issues(id, target)); + } + + // An id in both registries would make every lookup order-dependent, and + // `objective_for` would answer differently depending on which map it + // consulted first. + for id in self.learning_targets.keys() { + if self.learning_objectives.contains_key(id) { + issues.push(format!( + "`{id}` is declared as both a learning objective and a learning target" + )); + } + } + issues.extend(self.prerequisite_cycles()); issues } + /// Checks one learning target. + /// + /// The schema already rules out a third tier and a target with no objective, + /// so what is left are the cross-references and the two places where a + /// target and its objective could contradict each other: a ceiling above the + /// objective's would leave the pair disagreeing about how hard the topic is + /// assessed, and an assessed target under an unassessed objective would be + /// measured and never reported. + /// + /// # Arguments + /// + /// * `id` - the target's id. + /// * `target` - the target. + /// + /// # Returns + /// + /// One message per problem. + fn target_issues(&self, id: &str, target: &Target) -> Vec { + let mut issues = Vec::new(); + + if target.text.trim().is_empty() { + issues.push(format!("target `{id}`: empty text")); + } + // Target ids are spelled differently from objective ids so that an id in + // an item, a reading, or a report says which tier it belongs to without + // a lookup. Enforced, because the moment one target is named `lo-...` + // the convention stops being usable for reading a file. + if id + .trim_start_matches(|c: char| !c.is_ascii_alphanumeric()) + .starts_with(OBJECTIVE_PREFIX) + { + issues.push(format!( + "target `{id}`: a target id must not start with `{OBJECTIVE_PREFIX}`, which is \ + how an objective id is spelled. Try `t-{}`.", + id.trim_start_matches(|c: char| !c.is_ascii_alphanumeric()) + .trim_start_matches(OBJECTIVE_PREFIX) + .trim_start_matches('-') + )); + } + + for lec in &target.lectures { + if !self.lectures.contains_key(lec) { + issues.push(format!("target `{id}`: unknown lecture `{lec}`")); + } + } + for pre in &target.prerequisites { + if !self.learning_targets.contains_key(pre) + && !self.learning_objectives.contains_key(pre) + { + issues.push(format!("target `{id}`: unknown prerequisite `{pre}`")); + } + if pre == id { + issues.push(format!("target `{id}`: lists itself as a prerequisite")); + } + } + + let Some(objective) = self.learning_objectives.get(&target.objective) else { + issues.push(format!( + "target `{id}`: unknown objective `{}`", + target.objective + )); + return issues; + }; + if let (Some(target_ceiling), Some(objective_ceiling)) = + (target.level_ceiling, objective.level_ceiling) + { + if target_ceiling > objective_ceiling { + issues.push(format!( + "target `{id}`: level_ceiling {target_ceiling} is above objective `{}`'s \ + ceiling of {objective_ceiling}. Raise the objective's ceiling if you mean \ + to assess the topic that high.", + target.objective + )); + } + } + if target.assessed && !objective.assessed { + issues.push(format!( + "target `{id}`: marked assessed, but objective `{}` is marked \ + `assessed: false`, so nothing under it is reported", + target.objective + )); + } + issues + } + + /// Checks one reading, collecting every problem with it. + /// + /// # Arguments + /// + /// * `lecture` - the lecture id, for the message. + /// * `index` - position in the lecture's list, since a reading has no id. + /// * `reading` - the reading. + /// * `seen` - reference and locator pairs already found in this lecture, + /// extended as it goes. + /// + /// # Returns + /// + /// One message per problem. + fn reading_issues<'a>( + &self, + lecture: &str, + index: usize, + reading: &'a Reading, + seen: &mut Vec<(&'a str, &'a str)>, + ) -> Vec { + let mut issues = Vec::new(); + let at = format!("lecture `{lecture}` reading {}", index + 1); + + let Some(key) = reading.reference.as_deref() else { + if reading.text.is_none() { + issues.push(format!( + "{at}: needs a `ref` naming a reference, or a plain citation string" + )); + } + return issues; + }; + + match self.references.get(key) { + None => issues.push(format!("{at}: unknown reference `{key}`")), + Some(reference) => { + if reading.path.is_some() && reference.base_url.is_none() && reading.url.is_none() { + issues.push(format!( + "{at}: has a `path` but reference `{key}` has no `base_url` to join it to" + )); + } + } + } + + if let Some(locator) = reading.locator.as_deref() { + if seen.contains(&(key, locator)) { + issues.push(format!( + "{at}: `{key} {locator}` is assigned twice in one lecture" + )); + } + seen.push((key, locator)); + } + + for target in &reading.targets { + if self.learning_targets.contains_key(target) { + continue; + } + match self.learning_objectives.get(target) { + None => issues.push(format!("{at}: unknown learning target `{target}`")), + // An objective with no targets stands as its own, so citing it + // is fine; an objective with targets is never the place to cite + // a reading, since a reading backs one performance. + Some(_) if !self.targets(target).is_empty() => issues.push(format!( + "{at}: `{target}` is an objective with {} target(s); cite the specific \ + target this section serves", + self.targets(target).len() + )), + Some(_) => {} + } + } + issues + } + /// Detects cycles in the objective prerequisite graph. /// /// A cycle would make a study-order suggestion loop forever, so it is worth @@ -429,6 +1567,30 @@ impl CourseFile { }) } + /// Looks up a target, erroring on a dangling reference. + /// + /// # Arguments + /// + /// * `id` - the target id. + /// * `context` - what referenced it, for the error message. + /// + /// # Returns + /// + /// The target. + /// + /// # Errors + /// + /// Returns [`Error::Unresolved`] when the id is not registered. + pub fn target(&self, id: &str, context: &str) -> Result<&Target> { + self.learning_targets + .get(id) + .ok_or_else(|| Error::Unresolved { + kind: "learning target", + id: id.to_string(), + context: Some(context.to_string()), + }) + } + /// Looks up a lecture, erroring on a dangling reference. /// /// # Arguments @@ -451,30 +1613,277 @@ impl CourseFile { }) } - /// The objective text, or the bare id when unregistered. + /// The objective an id rolls up to. /// - /// Report rendering uses this so a missing objective degrades to a readable - /// label instead of failing a whole report. + /// The single function the rest of the tool goes through to turn an item's + /// tag into a reporting unit, so a target and an objective can be handled by + /// one code path. /// /// # Arguments /// - /// * `id` - the objective id. + /// * `id` - a target id or an objective id. + /// + /// # Returns + /// + /// The objective a target belongs to; otherwise `id` itself, which covers + /// both an objective and an id that is not registered at all. An + /// unregistered id is its own objective because validation reports it + /// elsewhere, and a report that silently dropped it would be worse than one + /// showing a row labeled with the bare id. + /// + /// The result borrows from the course in one branch and from `id` in the + /// other, so the two share a single lifetime rather than taking the elided + /// one from `&self`. + pub fn objective_for<'a>(&'a self, id: &'a str) -> &'a str { + match self.learning_targets.get(id) { + Some(target) if self.learning_objectives.contains_key(&target.objective) => { + target.objective.as_str() + } + _ => self + .learning_objectives + .get_key_value(id) + .map(|(k, _)| k.as_str()) + .or_else(|| { + self.learning_targets + .get_key_value(id) + .map(|(k, _)| k.as_str()) + }) + .unwrap_or(id), + } + } + + /// Whether an id names a registered learning objective. + /// + /// # Arguments + /// + /// * `id` - the id to check. + /// + /// # Returns + /// + /// `true` when it is in the objective registry. + pub fn is_objective(&self, id: &str) -> bool { + self.learning_objectives.contains_key(id) + } + + /// Whether an id names a registered learning target. + /// + /// # Arguments + /// + /// * `id` - the id to check. + /// + /// # Returns + /// + /// `true` when it is in the target registry. + pub fn is_target(&self, id: &str) -> bool { + self.learning_targets.contains_key(id) + } + + /// The targets of an objective, in teaching order. + /// + /// # Arguments + /// + /// * `objective` - the objective's id. + /// + /// # Returns + /// + /// Target ids ordered by [`Target::order`] then by id, empty for an + /// objective that has none. + pub fn targets(&self, objective: &str) -> Vec<&str> { + let mut ids: Vec<&String> = self + .learning_targets + .iter() + .filter(|(_, target)| target.objective == objective) + .map(|(id, _)| id) + .collect(); + // By ceiling, then by id. A target is a question template rather than a + // step in a sequence — an objective's targets are not taught in an + // order, and an exam samples from them — so there is no teaching order + // to print. What there is is depth, and grouping by it puts the + // checklist in the order the study methods apply: recall for a Level 1 + // target, explanation for Level 2, variations and written solutions + // above that. Ties break by id so the list is stable. + ids.sort_by_key(|id| { + ( + self.effective_level_ceiling(id) + .map(|l| l.code()) + .unwrap_or(0), + (*id).clone(), + ) + }); + ids.into_iter().map(String::as_str).collect() + } + + /// The registry's own copy of an id, from whichever tier holds it. + /// + /// Anything that walks [`CourseFile::registry_in_order`] and needs to return + /// borrowed ids goes through this, since that walk yields owned `String`s + /// spanning both registries. + /// + /// # Arguments + /// + /// * `id` - a target id or an objective id. + /// + /// # Returns + /// + /// The key as stored, or `None` when neither registry holds it. + pub fn registry_key(&self, id: &str) -> Option<&str> { + // The two maps hold different value types, so the lookups cannot be + // chained before the key is extracted from each. + if let Some((key, _)) = self.learning_objectives.get_key_value(id) { + return Some(key.as_str()); + } + self.learning_targets + .get_key_value(id) + .map(|(key, _)| key.as_str()) + } + + /// The text of an objective or a target, or the bare id when unregistered. + /// + /// Report rendering uses this so a missing id degrades to a readable label + /// instead of failing a whole report. + /// + /// # Arguments + /// + /// * `id` - a target id or an objective id. /// /// # Returns /// /// The display text. - pub fn objective_text(&self, id: &str) -> String { - self.learning_objectives + pub fn text_for(&self, id: &str) -> String { + if let Some(objective) = self.learning_objectives.get(id) { + return objective.text.clone(); + } + self.learning_targets .get(id) - .map(|o| o.text.clone()) + .map(|t| t.text.clone()) .unwrap_or_else(|| id.to_string()) } - /// Objectives in a stable teaching order: by unit as declared, then by id. + /// The lectures that develop an objective or a target. + /// + /// # Arguments + /// + /// * `id` - a target id or an objective id. /// /// # Returns /// - /// Objective ids in report order. + /// The declared lecture ids, empty when the id is unregistered. + pub fn lectures_for(&self, id: &str) -> &[String] { + if let Some(objective) = self.learning_objectives.get(id) { + return &objective.lectures; + } + match self.learning_targets.get(id) { + Some(target) => &target.lectures, + None => &[], + } + } + + /// The unit an objective or target belongs to. + /// + /// # Arguments + /// + /// * `id` - a target id or an objective id. + /// + /// # Returns + /// + /// The unit id. A target has no unit of its own and takes its objective's, + /// which is the only coherent answer: a target in a different unit from its + /// objective would appear twice in any report ordered by unit. + pub fn objective_unit(&self, id: &str) -> Option<&str> { + if let Some(objective) = self.learning_objectives.get(id) { + return objective.unit.as_deref(); + } + let target = self.learning_targets.get(id)?; + self.learning_objectives + .get(&target.objective)? + .unit + .as_deref() + } + + /// The level an objective or target may be assessed up to. + /// + /// # Arguments + /// + /// * `id` - a target id or an objective id. + /// + /// # Returns + /// + /// A target's own ceiling, else its objective's, else `None` for no ceiling. + pub fn effective_level_ceiling(&self, id: &str) -> Option { + if let Some(objective) = self.learning_objectives.get(id) { + return objective.level_ceiling; + } + let target = self.learning_targets.get(id)?; + if let Some(ceiling) = target.level_ceiling { + return Some(ceiling); + } + self.learning_objectives + .get(&target.objective)? + .level_ceiling + } + + /// Whether an objective or target is assessed. + /// + /// # Arguments + /// + /// * `id` - a target id or an objective id. + /// + /// # Returns + /// + /// `false` when the id itself or the objective above it is marked + /// `assessed: false`, and `false` for an unregistered id. + pub fn is_assessed(&self, id: &str) -> bool { + if let Some(objective) = self.learning_objectives.get(id) { + return objective.assessed; + } + let Some(target) = self.learning_targets.get(id) else { + return false; + }; + target.assessed + && self + .learning_objectives + .get(&target.objective) + .is_some_and(|o| o.assessed) + } + + /// Both registries in a stable teaching order: each objective immediately + /// followed by its own targets. + /// + /// Interleaving the two by `order` would be meaningless, because a target's + /// `order` is a position among its siblings rather than among the course. + /// Grouping targets under their objective is also what a report wants: the + /// claim, then the evidence for it. + /// + /// # Returns + /// + /// Objective and target ids in report order. + pub fn registry_in_order(&self) -> Vec { + let mut out = + Vec::with_capacity(self.learning_objectives.len() + self.learning_targets.len()); + for id in self.objectives_in_order() { + let targets = self.targets(&id); + out.push(id); + out.extend(targets.into_iter().map(str::to_string)); + } + // A target whose objective is missing would otherwise vanish from every + // report. Validation reports the dangling id; this keeps the row. + for (id, target) in &self.learning_targets { + if !self.learning_objectives.contains_key(&target.objective) { + out.push(id.clone()); + } + } + out + } + + /// Objectives in teaching order. + /// + /// This is the list a syllabus prints, a blueprint requires items against, + /// and a report classifies. It is short by construction, which is the point. + /// + /// # Returns + /// + /// Objective ids ordered by unit as declared, then by [`Objective::order`], + /// then by id. pub fn objectives_in_order(&self) -> Vec { let unit_rank: BTreeMap<&str, usize> = self .units @@ -484,17 +1893,254 @@ impl CourseFile { .collect(); let mut ids: Vec<&String> = self.learning_objectives.keys().collect(); ids.sort_by_key(|id| { - let lo = &self.learning_objectives[*id]; - let rank = lo + let objective = &self.learning_objectives[*id]; + let rank = objective .unit .as_deref() .and_then(|u| unit_rank.get(u).copied()) .unwrap_or(usize::MAX); - (rank, (*id).clone()) + (rank, objective.order.unwrap_or(u32::MAX), (*id).clone()) }); ids.into_iter().cloned().collect() } + /// The ids items may be tagged with, in teaching order. + /// + /// Every target, plus any objective that has no targets. An objective that + /// has them is not taggable: the item would land in its denominator without + /// recording which performance the question asked for, and that record is + /// what a report drills into. An objective with none stands as its own + /// target, which is what lets a course adopt the second tier one unit at a + /// time. + /// + /// # Returns + /// + /// Taggable ids in report order. + pub fn targets_in_order(&self) -> Vec { + self.registry_in_order() + .into_iter() + .filter(|id| self.targets(id).is_empty()) + .collect() + } + + /// Every objective and target a lecture covers, in teaching order. + /// + /// # Arguments + /// + /// * `lecture` - the lecture id. + /// + /// # Returns + /// + /// Ids whose `lectures` list names this lecture, in [`registry_in_order`] + /// order. + /// + /// [`registry_in_order`]: CourseFile::registry_in_order + pub fn lecture_entries(&self, lecture: &str) -> Vec<&str> { + let mut out: Vec<&str> = Vec::new(); + for id in self.registry_in_order() { + if self.lectures_for(&id).iter().any(|l| l == lecture) { + // Borrow the key rather than the owned String from the ordering. + if let Some((key, _)) = self.learning_objectives.get_key_value(&id) { + out.push(key.as_str()); + } else if let Some((key, _)) = self.learning_targets.get_key_value(&id) { + out.push(key.as_str()); + } + } + } + out + } + + /// The objectives a lecture covers, in teaching order. + /// + /// This is what a lecture's objectives page lists: four to eight claims, not + /// the forty performances they are built from. + /// + /// # Arguments + /// + /// * `lecture` - the lecture id. + /// + /// # Returns + /// + /// Objective ids the lecture names directly, plus the objectives of any + /// target it names, deduplicated and in teaching order. A lecture that lists + /// only targets still has objectives, and leaving them off its page would + /// make the page depend on whether you happened to tag the objective with + /// the lecture as well. + pub fn lecture_objectives(&self, lecture: &str) -> Vec<&str> { + let mut out: Vec<&str> = Vec::new(); + for id in self.objectives_in_order() { + let Some((key, objective)) = self.learning_objectives.get_key_value(&id) else { + continue; + }; + let taught_directly = objective.lectures.iter().any(|l| l == lecture); + let taught_through_a_target = self + .targets(key) + .into_iter() + .any(|t| self.lectures_for(t).iter().any(|l| l == lecture)); + if taught_directly || taught_through_a_target { + out.push(key.as_str()); + } + } + out + } + + /// The targets a lecture covers, in teaching order. + /// + /// # Arguments + /// + /// * `lecture` - the lecture id. + /// + /// # Returns + /// + /// Ids of the lecture's targets, plus any objective it names that has no + /// targets and so stands as its own. + pub fn lecture_targets(&self, lecture: &str) -> Vec<&str> { + self.lecture_entries(lecture) + .into_iter() + .filter(|id| self.targets(id).is_empty()) + .collect() + } + + /// Every reading that serves an objective, with the lecture it was assigned in. + /// + /// Derived by scanning lectures rather than stored on the objective, for the + /// same reason [`crate::history::History`] derives usage from assessment + /// records: a second copy of an edge is a second thing to keep in step. It also + /// puts the pointer on the volatile side, since a new edition renumbers + /// sections but leaves your objectives alone. + /// + /// # Arguments + /// + /// * `objective` - the objective id. + /// + /// # Returns + /// + /// Pairs of lecture id and reading, in lecture order then assignment order. + pub fn readings_for_objective(&self, objective: &str) -> Vec<(&str, &Reading)> { + let mut out = Vec::new(); + for (lecture_id, lecture) in &self.lectures { + for reading in &lecture.readings { + // A reading cited against a target is a reading for that + // target's objective, which is what lets a student report answer + // "where do I go and read about this" from an objective a report + // classified rather than only from the target a question missed. + if reading + .targets + .iter() + .any(|t| t == objective || self.objective_for(t) == objective) + { + out.push((lecture_id.as_str(), reading)); + } + } + } + out + } + + /// Assessed targets with no reading behind them. + /// + /// These are the things a student report cannot advise on: it can say the + /// target was missed, but not where to go and read about it. + /// + /// Reported at the tier readings are cited against, which is the target, so + /// a course is not told off once per target and again per objective. + /// + /// # Returns + /// + /// Target ids in teaching order. + pub fn targets_without_readings(&self) -> Vec<&str> { + let cited: std::collections::BTreeSet<&str> = self + .lectures + .values() + .flat_map(|l| l.readings.iter()) + .flat_map(|r| r.targets.iter()) + .map(String::as_str) + .collect(); + self.registry_in_order() + .into_iter() + .filter_map(|id| { + // The id may name either registry, so the key has to be + // resolved from both. Looking in only one of them silently + // dropped every row this function exists to report. + let key = self.registry_key(&id)?; + if !self.is_assessed(key) { + return None; + } + let targets = self.targets(key); + let covered = + cited.contains(key) || targets.iter().any(|target| cited.contains(target)); + // An objective whose targets carry the readings is covered, and + // an objective with targets is never itself the place to cite + // one. + (!covered && targets.is_empty()).then_some(key) + }) + .collect() + } + + /// Looks up a reference, erroring on a dangling citation key. + /// + /// # Arguments + /// + /// * `key` - the citation key. + /// * `context` - what cited it, for the error message. + /// + /// # Returns + /// + /// The reference. + /// + /// # Errors + /// + /// Returns [`Error::Unresolved`] when the key is not registered. + pub fn reference(&self, key: &str, context: &str) -> Result<&Reference> { + self.references.get(key).ok_or_else(|| Error::Unresolved { + kind: "reference", + id: key.to_string(), + context: Some(context.to_string()), + }) + } + + /// Expands `{objective-or-target-id}` in a prose field to whatever the + /// caller wants. + /// + /// Reading notes refer to targets in passing ("a worked instance of + /// `{t-vdw-additivity}`"), and a lecture page renders that as a number while + /// a student report renders it as text. Both registries are consulted, since + /// a note in a migrated course names targets almost exclusively. Only a name + /// that resolves to a declared id is treated as a placeholder, so + /// `$U_\text{final}$` passes through untouched; that collision is the reason + /// this is not a general template syntax. + /// + /// # Arguments + /// + /// * `prose` - the field to expand. + /// * `render` - called with each resolved objective id. + /// + /// # Returns + /// + /// The prose with resolved placeholders replaced. + pub fn expand_objective_refs(&self, prose: &str, render: impl Fn(&str) -> String) -> String { + let mut out = String::with_capacity(prose.len()); + let mut rest = prose; + while let Some(open) = rest.find('{') { + let (head, tail) = rest.split_at(open); + out.push_str(head); + let Some(close) = tail.find('}') else { + out.push_str(tail); + return out; + }; + let name = &tail[1..close]; + if self.learning_objectives.contains_key(name) + || self.learning_targets.contains_key(name) + { + out.push_str(&render(name)); + } else { + out.push_str(&tail[..=close]); + } + rest = &tail[close + 1..]; + } + out.push_str(rest); + out + } + /// A skeleton course file for `coursebank init`. /// /// # Arguments @@ -515,6 +2161,7 @@ impl CourseFile { date: None, unit: Some("u-intro".to_string()), slides_url: None, + teaches: Vec::new(), readings: Vec::new(), }, ); @@ -522,15 +2169,33 @@ impl CourseFile { los.insert( "lo-example".to_string(), Objective { - text: "Replace this with an objective stated as a student action.".to_string(), + text: "Replace this with an objective: the claim a report should make.".to_string(), unit: Some("u-intro".to_string()), lectures: vec!["L01".to_string()], + order: Some(1), level_ceiling: Some(Level::Understand), prerequisites: Vec::new(), tags: Vec::new(), assessed: true, }, ); + // One target as well, because the two tiers are easier to understand + // from an example than from the schema. + let mut targets = BTreeMap::new(); + targets.insert( + "t-example".to_string(), + Target { + text: "Replace this with a target: one performance an item can measure." + .to_string(), + objective: "lo-example".to_string(), + lectures: vec!["L01".to_string()], + order: Some(1), + level_ceiling: None, + prerequisites: Vec::new(), + tags: Vec::new(), + assessed: true, + }, + ); CourseFile { schema_version: SCHEMA_VERSION.to_string(), course: Course { @@ -549,7 +2214,10 @@ impl CourseFile { }], lectures, learning_objectives: los, + learning_targets: targets, + references: BTreeMap::new(), stimuli: BTreeMap::new(), + origins: BTreeMap::new(), } } } @@ -618,7 +2286,7 @@ course: term: Spring 2026 "#, ); - assert_eq!(c.schema_version, "1.0"); + assert_eq!(c.schema_version, SCHEMA_VERSION); assert_eq!(c.policy.options_per_item, 4); assert_eq!(c.course.slug(), "biosc-1540"); assert!(c.validate().is_empty()); @@ -697,7 +2365,7 @@ learning_objectives: "#, ); // Declared order wins over alphabetical order of unit ids. - assert_eq!(c.objectives_in_order(), vec!["lo-z", "lo-a"]); + assert_eq!(c.registry_in_order(), vec!["lo-z", "lo-a"]); } #[test] @@ -706,4 +2374,630 @@ learning_objectives: assert_eq!(slugify("Exam 4 -- Final!"), "exam-4-final"); assert_eq!(slugify(" "), ""); } + + /// A course with one reference and two readings, one of them supplemental. + fn with_readings() -> CourseFile { + parse( + r#" +course: { code: X, title: Y, term: Z } +references: + kuriyan2013molecules: + label: KKW + role: required + title: The molecules of life + base_url: https://example.org/kkw/ +lectures: + L1.1: + title: Enthalpy + readings: + - ref: kuriyan2013molecules + locator: '§6.1' + path: '6/A/#1' + targets: [lo-a] + summary: What a system is. + focus: Fix the definitions. + - ref: kuriyan2013molecules + locator: '§1.9' + path: '1/B/#9' + role: supplemental + targets: [lo-b] +learning_objectives: + lo-a: { text: A, lectures: [L1.1], order: 1 } + lo-b: { text: B, lectures: [L1.1], order: 2 } +"#, + ) + } + + #[test] + fn a_structured_reading_parses_and_validates() { + let c = with_readings(); + assert!(c.validate().is_empty(), "{:?}", c.validate()); + let readings = &c.lectures["L1.1"].readings; + assert_eq!( + readings[0].reference.as_deref(), + Some("kuriyan2013molecules") + ); + assert_eq!(readings[0].role, ReadingRole::Assigned); + assert_eq!(readings[1].role, ReadingRole::Supplemental); + } + + #[test] + fn a_reading_url_is_built_from_the_reference_base() { + let c = with_readings(); + let reference = &c.references["kuriyan2013molecules"]; + let reading = &c.lectures["L1.1"].readings[0]; + assert_eq!( + reading.resolve_url(reference).as_deref(), + Some("https://example.org/kkw/6/A/#1") + ); + assert_eq!(reading.cite("kuriyan2013molecules", reference), "KKW §6.1"); + } + + #[test] + fn a_link_resolves_from_the_exact_location_outward() { + let mut reference = Reference { + title: "Basic local alignment search tool".into(), + kind: ReferenceKind::Article, + ..Reference::default() + }; + + // Nothing at all to link to. + assert_eq!(reference.href(None, None), None); + + // A DOI is a link to the work, which beats nothing. + reference.doi = Some("10.1016/S0022-2836(05)80360-2".into()); + assert_eq!( + reference.href(None, None).as_deref(), + Some("https://doi.org/10.1016/S0022-2836(05)80360-2") + ); + + // The work's own URL is more use than its identifier. + reference.url = Some("https://example.org/blast".into()); + assert_eq!( + reference.href(None, None).as_deref(), + Some("https://example.org/blast") + ); + + // A location inside the work beats the work. + reference.base_url = Some("https://example.org/blast/".into()); + assert_eq!( + reference.href(None, Some("/§2")).as_deref(), + Some("https://example.org/blast/§2") + ); + assert_eq!( + reference + .href(Some("https://example.org/exact"), Some("§2")) + .as_deref(), + Some("https://example.org/exact") + ); + } + + #[test] + fn identifiers_are_normalized_before_they_become_links() { + let doi_as_url = Reference { + title: "T".into(), + doi: Some("https://doi.org/10.1/x".into()), + ..Reference::default() + }; + assert_eq!( + doi_as_url.href(None, None).as_deref(), + Some("https://doi.org/10.1/x") + ); + + let preprint = Reference { + title: "T".into(), + kind: ReferenceKind::Preprint, + arxiv: Some("arXiv:2301.00001".into()), + ..Reference::default() + }; + assert_eq!( + preprint.href(None, None).as_deref(), + Some("https://arxiv.org/abs/2301.00001") + ); + + // A bare number is still a PMC id. + let open_access = Reference { + title: "T".into(), + pmcid: Some("3084216".into()), + ..Reference::default() + }; + assert_eq!( + open_access.href(None, None).as_deref(), + Some("https://www.ncbi.nlm.nih.gov/pmc/articles/PMC3084216/") + ); + } + + #[test] + fn something_that_is_not_a_doi_is_reported() { + let c = parse( + r#" +course: { code: X, title: Y, term: Z } +references: + bad: + title: A work + doi: nature12373 + worse: + title: Another work + pmid: PMC3084216 +"#, + ); + let issues = c.validate(); + assert!( + issues.iter().any(|i| i.contains("is not a DOI")), + "{issues:?}" + ); + assert!( + issues.iter().any(|i| i.contains("is not a number")), + "{issues:?}" + ); + } + + #[test] + fn a_bare_string_reading_still_parses_and_round_trips() { + let c = parse( + r#" +course: { code: X, title: Y, term: Z } +lectures: + L01: + title: One + readings: + - 'KKW §6.1: system and surroundings. https://example.org/1' +"#, + ); + let reading = &c.lectures["L01"].readings[0]; + assert!(reading.reference.is_none()); + assert_eq!( + reading.text.as_deref(), + Some("KKW §6.1: system and surroundings. https://example.org/1") + ); + assert!(c.validate().is_empty()); + + // Serializing writes the string back as a string, so a load-and-save cycle + // does not migrate a file the author has not chosen to migrate. + let yaml = serde_yaml_ng::to_string(&c).expect("serializes"); + assert!(yaml.contains("- 'KKW §6.1: system and surroundings. https://example.org/1'")); + assert!(!yaml.contains("text:")); + } + + #[test] + fn readings_resolve_backwards_from_a_target() { + let c = with_readings(); + let found = c.readings_for_objective("lo-a"); + assert_eq!(found.len(), 1); + assert_eq!(found[0].0, "L1.1"); + assert_eq!(found[0].1.locator.as_deref(), Some("§6.1")); + assert!(c.readings_for_objective("lo-nobody").is_empty()); + } + + #[test] + fn an_entry_with_no_reading_is_reported() { + let mut c = with_readings(); + assert!(c.targets_without_readings().is_empty()); + c.learning_objectives.insert( + "lo-orphan".to_string(), + Objective { + text: "Orphan".to_string(), + unit: None, + lectures: vec!["L1.1".to_string()], + order: Some(3), + level_ceiling: None, + prerequisites: Vec::new(), + tags: Vec::new(), + assessed: true, + }, + ); + assert_eq!(c.targets_without_readings(), vec!["lo-orphan"]); + + // Something you teach but do not test is not a gap. + c.learning_objectives + .get_mut("lo-orphan") + .expect("just inserted") + .assessed = false; + assert!(c.targets_without_readings().is_empty()); + } + + #[test] + fn objective_order_beats_id_order_within_a_lecture() { + let c = parse( + r#" +course: { code: X, title: Y, term: Z } +lectures: + L1.1: { title: One } +learning_objectives: + lo-enthalpy: { text: Third, lectures: [L1.1], order: 3 } + lo-first-law: { text: Second, lectures: [L1.1], order: 2 } + lo-system: { text: First, lectures: [L1.1], order: 1 } +"#, + ); + // Alphabetically this is enthalpy, first-law, system, which puts an + // objective ahead of its own prerequisite. + assert_eq!( + c.lecture_objectives("L1.1"), + vec!["lo-system", "lo-first-law", "lo-enthalpy"] + ); + } + + #[test] + fn unknown_references_and_objectives_on_a_reading_are_reported() { + let c = parse( + r#" +course: { code: X, title: Y, term: Z } +references: + known: { title: A book } +lectures: + L01: + title: One + readings: + - { ref: missing, locator: '§1' } + - { ref: known, locator: '§2', path: '2/', targets: [lo-nope] } +"#, + ); + let issues = c.validate(); + assert!( + issues + .iter() + .any(|i| i.contains("unknown reference `missing`")) + ); + assert!( + issues + .iter() + .any(|i| i.contains("unknown learning target `lo-nope`")) + ); + // `path` with no base_url to join it to. + assert!(issues.iter().any(|i| i.contains("base_url"))); + } + + #[test] + fn a_duplicate_label_and_a_duplicate_locator_are_reported() { + let c = parse( + r#" +course: { code: X, title: Y, term: Z } +references: + one: { title: First, label: KKW } + two: { title: Second, label: KKW } +lectures: + L01: + title: One + readings: + - { ref: one, locator: '§1' } + - { ref: one, locator: '§1' } +"#, + ); + let issues = c.validate(); + assert!(issues.iter().any(|i| i.contains("`KKW` is the label of"))); + assert!( + issues + .iter() + .any(|i| i.contains("assigned twice in one lecture")) + ); + } + + /// Two objectives, one with targets and one standing on its own. + fn two_tier() -> CourseFile { + parse( + r#" +course: { code: X, title: Y, term: Z } +units: + - { id: u-1, title: One } +learning_objectives: + lo-binding: + text: Quantify single-site binding. + unit: u-1 + order: 1 + level_ceiling: 4 + lo-standalone: + text: An objective with no targets, reported on directly. + unit: u-1 + order: 2 +learning_targets: + t-kd-expression: + text: Write the expression for the dissociation constant. + objective: lo-binding + order: 2 + t-read-kd-from-plot: + text: Determine the dissociation constant from a plotted isotherm. + objective: lo-binding + order: 1 + level_ceiling: 3 +"#, + ) + } + + #[test] + fn a_two_tier_registry_validates() { + let c = two_tier(); + assert!(c.validate().is_empty(), "{:?}", c.validate()); + } + + #[test] + fn targets_roll_up_and_objectives_stand_alone() { + let c = two_tier(); + assert_eq!(c.objective_for("t-kd-expression"), "lo-binding"); + assert_eq!(c.objective_for("lo-binding"), "lo-binding"); + assert_eq!(c.objective_for("lo-standalone"), "lo-standalone"); + // An unregistered id is its own reporting unit rather than a dropped row. + assert_eq!(c.objective_for("t-typo"), "t-typo"); + + assert!(c.is_objective("lo-binding")); + assert!(!c.is_objective("t-kd-expression")); + assert!(c.is_target("t-kd-expression")); + assert!(!c.is_target("lo-binding")); + assert!(!c.is_target("t-typo"), "an unregistered id is not a target"); + } + + #[test] + fn the_registries_are_separate_sections() { + let c = two_tier(); + assert_eq!(c.learning_objectives.len(), 2, "objectives only"); + assert_eq!(c.learning_targets.len(), 2, "targets only"); + assert_eq!(c.objectives_in_order(), vec!["lo-binding", "lo-standalone"]); + assert_eq!( + c.targets("lo-binding"), + vec!["t-read-kd-from-plot", "t-kd-expression"] + ); + assert!(c.targets("lo-standalone").is_empty()); + // Items are tagged with targets, plus any objective that has none. + assert_eq!( + c.targets_in_order(), + vec!["t-read-kd-from-plot", "t-kd-expression", "lo-standalone"] + ); + } + + #[test] + fn targets_follow_their_own_objective_in_report_order() { + let c = two_tier(); + // `t-kd-expression` has order 2 and `lo-standalone` has order 2, but a + // target sorts among its siblings rather than against the whole course. + assert_eq!( + c.registry_in_order(), + vec![ + "lo-binding", + "t-read-kd-from-plot", + "t-kd-expression", + "lo-standalone" + ] + ); + } + + #[test] + fn a_target_inherits_unit_ceiling_and_assessment() { + let mut c = two_tier(); + // A target has no unit of its own; it takes its objective's. + assert_eq!(c.objective_unit("t-kd-expression"), Some("u-1")); + // Declared on the target, so not inherited. + assert_eq!( + c.effective_level_ceiling("t-read-kd-from-plot"), + Some(Level::Apply) + ); + // Absent on the target, so the objective's applies. + assert_eq!( + c.effective_level_ceiling("t-kd-expression"), + Some(Level::Analyze) + ); + + assert!(c.is_assessed("t-kd-expression")); + c.learning_objectives + .get_mut("lo-binding") + .expect("declared") + .assessed = false; + assert!( + !c.is_assessed("t-kd-expression"), + "parking an objective parks every target under it" + ); + } + + #[test] + fn a_target_id_may_not_be_spelled_like_an_objective() { + // The prefix convention is what lets an id in an item or a report say + // which tier it belongs to without a lookup, so it is enforced. + let c = parse( + r#" +course: { code: X, title: Y, term: Z } +learning_objectives: + lo-a: { text: A } +learning_targets: + lo-b: { text: B, objective: lo-a } +"#, + ); + let issues = c.validate(); + assert!( + issues + .iter() + .any(|i| i.contains("must not start with `lo`")), + "expected a naming complaint, got {issues:?}" + ); + assert!( + issues.iter().any(|i| i.contains("`t-b`")), + "and a suggested id, got {issues:?}" + ); + } + + #[test] + fn an_id_in_both_registries_is_reported() { + let c = parse( + r#" +course: { code: X, title: Y, term: Z } +learning_objectives: + lo-a: { text: A } + t-b: { text: 'Also an objective, confusingly' } +learning_targets: + t-b: { text: B, objective: lo-a } +"#, + ); + let issues = c.validate(); + assert!( + issues + .iter() + .any(|i| i.contains("both a learning objective and a learning target")), + "got {issues:?}" + ); + } + + #[test] + fn a_target_naming_an_unknown_objective_is_reported() { + let c = parse( + r#" +course: { code: X, title: Y, term: Z } +learning_targets: + t-a: { text: A, objective: lo-missing } +"#, + ); + let issues = c.validate(); + assert!( + issues + .iter() + .any(|i| i.contains("unknown objective `lo-missing`")), + "got {issues:?}" + ); + // It still appears in report order rather than vanishing. + assert_eq!(c.registry_in_order(), vec!["t-a"]); + } + + #[test] + fn a_target_may_not_outreach_its_objectives_ceiling() { + let c = parse( + r#" +course: { code: X, title: Y, term: Z } +learning_objectives: + lo-a: { text: A, level_ceiling: 2 } +learning_targets: + t-b: { text: B, objective: lo-a, level_ceiling: 4 } +"#, + ); + let issues = c.validate(); + assert!( + issues.iter().any(|i| i.contains("is above objective")), + "expected a ceiling complaint, got {issues:?}" + ); + } + + #[test] + fn a_reading_is_cited_against_a_target_and_serves_its_objective() { + let c = parse( + r#" +course: { code: X, title: Y, term: Z } +references: + kkw: { title: The molecules of life, label: KKW } +lectures: + L1.1: + title: Binding + readings: + - { ref: kkw, locator: '§6.1', targets: [t-kd] } +learning_objectives: + lo-binding: { text: Quantify binding., order: 1, lectures: [L1.1] } +learning_targets: + t-kd: { text: Write the expression., objective: lo-binding, order: 1, lectures: [L1.1] } + t-isotherm: { text: Read an isotherm., objective: lo-binding, order: 2, lectures: [L1.1] } +"#, + ); + assert!(c.validate().is_empty(), "{:?}", c.validate()); + // The objective is never itself the place to cite a reading, and the + // target that has one is not a gap. Its sibling is. + assert_eq!(c.targets_without_readings(), vec!["t-isotherm"]); + // A reading cited against a target answers "what do I read for this + // objective" too. + assert_eq!(c.readings_for_objective("lo-binding").len(), 1); + assert_eq!(c.readings_for_objective("t-kd").len(), 1); + } + + #[test] + fn a_reading_may_not_cite_an_objective_that_has_targets() { + let c = parse( + r#" +course: { code: X, title: Y, term: Z } +references: + kkw: { title: The molecules of life } +lectures: + L1.1: + title: Binding + readings: + - { ref: kkw, locator: '§6.1', targets: [lo-binding] } +learning_objectives: + lo-binding: { text: Quantify binding. } +learning_targets: + t-kd: { text: Write the expression., objective: lo-binding } +"#, + ); + let issues = c.validate(); + assert!( + issues.iter().any(|i| i.contains("cite the specific")), + "expected a tier complaint, got {issues:?}" + ); + } + + #[test] + fn a_lecture_page_sees_both_tiers() { + let c = parse( + r#" +course: { code: X, title: Y, term: Z } +lectures: + L1.1: { title: Binding } +learning_objectives: + lo-binding: { text: Quantify binding., order: 1, lectures: [L1.1] } + lo-solo: { text: A target-less objective., order: 2, lectures: [L1.1] } +learning_targets: + t-kd: { text: Write the expression., objective: lo-binding, order: 1, lectures: [L1.1] } + t-plot: { text: Read a plot., objective: lo-binding, order: 2, lectures: [L1.1] } +"#, + ); + assert_eq!(c.lecture_objectives("L1.1"), vec!["lo-binding", "lo-solo"]); + assert_eq!( + c.lecture_targets("L1.1"), + vec!["t-kd", "t-plot", "lo-solo"], + "a target-less objective stands as its own target" + ); + } + + #[test] + fn a_lecture_that_lists_only_targets_still_has_objectives() { + // The objective is not tagged with the lecture; only its targets are. + let c = parse( + r#" +course: { code: X, title: Y, term: Z } +lectures: + L1.1: { title: Binding } +learning_objectives: + lo-binding: { text: Quantify binding., order: 1 } +learning_targets: + t-kd: { text: Write the expression., objective: lo-binding, order: 1, lectures: [L1.1] } +"#, + ); + assert_eq!(c.lecture_objectives("L1.1"), vec!["lo-binding"]); + } + + #[test] + fn a_target_id_is_a_placeholder_too() { + // Reading notes in a migrated course name targets, so resolving only + // objectives left every placeholder unexpanded. + let c = parse( + r#" +course: { code: X, title: Y, term: Z } +learning_objectives: + lo-binding: { text: Quantify binding. } +learning_targets: + t-kd: { text: Write the expression., objective: lo-binding } +"#, + ); + let expanded = c.expand_objective_refs( + r"a worked instance of {t-kd}, under {lo-binding}, where $U_\text{final}$ is fixed", + |id| format!("<{id}>"), + ); + assert_eq!( + expanded, + r"a worked instance of , under , where $U_\text{final}$ is fixed" + ); + } + + #[test] + fn only_a_declared_objective_id_is_a_placeholder() { + let c = with_readings(); + let expanded = c.expand_objective_refs( + r"a worked instance of {lo-a}, where $U_\text{final}$ is unchanged, {lo-typo} too", + |id| format!("<{id}>"), + ); + assert_eq!( + expanded, + r"a worked instance of , where $U_\text{final}$ is unchanged, {lo-typo} too" + ); + } } diff --git a/src/model/course/fragment.rs b/src/model/course/fragment.rs new file mode 100644 index 0000000..83a1e5d --- /dev/null +++ b/src/model/course/fragment.rs @@ -0,0 +1,974 @@ +// SPDX-License-Identifier: Prosperity-3.0.0 +// Copyright Scientific Computing Studio +// Source: https://git.scient.ing/education/coursebank + +//! One course, several files. +//! +//! A course of forty lectures does not fit in a file anyone wants to scroll. So +//! the registries [`CourseFile`] holds may be spread across a directory and +//! merged on load: +//! +//! ```text +//! course.yaml course, policy, units +//! references.yaml references +//! lectures/l-1-2.yaml one lecture, its readings, and what it teaches +//! objectives/lo-x.yaml one objective and its targets +//! ``` +//! +//! The merge happens in memory on every command. Nothing is generated on disk +//! and no command depends on a build step, because a generated file that other +//! commands read is a file that goes stale. `coursebank course build` exists to +//! show you the merged result, and nothing reads what it writes. +//! +//! # What this does not relax +//! +//! Splitting a file is only worth doing if it cannot introduce a second +//! definition of the same thing. Two rules keep that true, and both are enforced +//! here rather than left to convention: +//! +//! * **One definition site per id.** Two files defining `lo-read-file-formats` +//! is an error naming both paths. The winner is not the last file loaded, +//! because there is no winner. +//! * **A section belongs to a kind of file.** A file under `lectures/` may not +//! define `learning_objectives`. Otherwise the layout decays into forty files +//! that each might hold anything, which is the same navigation problem in a +//! worse shape. +//! +//! `course.yaml` is exempt from the second rule: a course that has not been +//! split is a single fragment that happens to define everything, and it keeps +//! loading unchanged. +//! +//! # Two derivations +//! +//! Splitting by lecture makes two fields tedious to maintain by hand, so they +//! are derived instead: +//! +//! * A lecture's `teaches:` list adds that lecture to each named objective's +//! `lectures`. You write what a lecture covers while planning the lecture, +//! which is when you know. +//! * A target with no `lectures` of its own inherits its objective's, the same +//! way it already inherits `level_ceiling`. +//! +//! Both are unions and both are idempotent, so declaring a pair on both sides +//! is redundant rather than contradictory. + +use std::collections::BTreeMap; +use std::fmt; +use std::path::{Path, PathBuf}; + +use serde::{Deserialize, Serialize}; + +use super::{ + COURSE_FILE, Course, CourseFile, Lecture, Objective, Policy, Reference, SCHEMA_VERSION, + SUPPORTED_MAJORS, Stimulus, Target, Unit, +}; +use crate::error::{Error, Result}; +use crate::layout::Layout; +use crate::yaml; + +/// The file holding the bibliography when it is kept out of `course.yaml`. +pub const REFERENCES_FILE: &str = "references.yaml"; + +/// One registry section of a course. +/// +/// Used to say which file a fact came from, and to keep a fragment from +/// defining something that belongs somewhere else. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] +pub enum Section { + /// Course identity. + Course, + /// Course-wide policy. + Policy, + /// Units. + Units, + /// Lectures. + Lectures, + /// Learning objectives. + Objectives, + /// Learning targets. + Targets, + /// Works the course cites. + References, + /// Shared stimuli. + Stimuli, +} + +impl Section { + /// Every section, in the order a merged course lists them. + pub const ALL: [Section; 8] = [ + Section::Course, + Section::Policy, + Section::Units, + Section::Lectures, + Section::Objectives, + Section::Targets, + Section::References, + Section::Stimuli, + ]; + + /// The YAML key this section is written under. + pub fn key(self) -> &'static str { + match self { + Section::Course => "course", + Section::Policy => "policy", + Section::Units => "units", + Section::Lectures => "lectures", + Section::Objectives => "learning_objectives", + Section::Targets => "learning_targets", + Section::References => "references", + Section::Stimuli => "stimuli", + } + } + + /// What one of its entries is called in a message. + pub fn noun(self) -> &'static str { + match self { + Section::Course => "course identity", + Section::Policy => "policy", + Section::Units => "unit", + Section::Lectures => "lecture", + Section::Objectives => "objective", + Section::Targets => "target", + Section::References => "reference", + Section::Stimuli => "stimulus", + } + } + + /// Where a file defining this section is expected to live. + pub fn home(self) -> &'static str { + match self { + Section::Course | Section::Policy | Section::Units => COURSE_FILE, + Section::Lectures => "lectures/*.yaml", + Section::Objectives | Section::Targets | Section::Stimuli => "objectives/*.yaml", + Section::References => REFERENCES_FILE, + } + } +} + +impl fmt::Display for Section { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str(self.key()) + } +} + +/// What kind of file a fragment is, which fixes what it may define. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Role { + /// `course.yaml`. May define anything, so an unsplit course still loads. + Root, + /// `references.yaml`. + References, + /// A file under `lectures/`. + Lecture, + /// A file under `objectives/`. + Objective, +} + +impl Role { + /// Whether a file in this role may define a section. + pub fn allows(self, section: Section) -> bool { + match self { + Role::Root => true, + Role::References => section == Section::References, + Role::Lecture => section == Section::Lectures, + Role::Objective => matches!( + section, + Section::Objectives | Section::Targets | Section::Stimuli + ), + } + } + + /// A short name for a message. + pub fn label(self) -> &'static str { + match self { + Role::Root => "course", + Role::References => "references", + Role::Lecture => "lecture", + Role::Objective => "objective", + } + } +} + +/// One file's worth of course registries. +/// +/// Every section is optional, which is what makes this both the fragment schema +/// and — with every section filled in — the schema of an unsplit `course.yaml`. +/// [`CourseFile`] is the resolved model the rest of the crate reads; this is +/// only what one file on disk is allowed to say. +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct Fragment { + /// Schema version this file targets. + #[serde( + default, + deserialize_with = "yaml::flexible_string_opt", + skip_serializing_if = "Option::is_none" + )] + pub schema_version: Option, + + /// Course identity. Exactly one fragment must carry it. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub course: Option, + + /// Course-wide policy. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub policy: Option, + + /// Units, in teaching order. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub units: Vec, + + /// Lectures by id. + #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] + pub lectures: BTreeMap, + + /// Learning objectives by id. + #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] + pub learning_objectives: BTreeMap, + + /// Learning targets by id. + #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] + pub learning_targets: BTreeMap, + + /// Works the course cites, by citation key. + #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] + pub references: BTreeMap, + + /// Shared stimuli by id. + #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] + pub stimuli: BTreeMap, +} + +impl Fragment { + /// Loads one fragment from disk. + /// + /// # Arguments + /// + /// * `path` - the file to read. + /// + /// # Returns + /// + /// The parsed fragment. + /// + /// # Errors + /// + /// Returns [`Error::Io`] if unreadable and [`Error::Yaml`] if it does not + /// match the schema. Unknown keys are errors, so a misspelled section name + /// is caught here rather than silently contributing nothing. + pub fn load(path: &Path) -> Result { + yaml::read(path) + } + + /// Which sections this fragment actually defines. + pub fn sections(&self) -> Vec
{ + let mut out = Vec::new(); + if self.course.is_some() { + out.push(Section::Course); + } + if self.policy.is_some() { + out.push(Section::Policy); + } + if !self.units.is_empty() { + out.push(Section::Units); + } + if !self.lectures.is_empty() { + out.push(Section::Lectures); + } + if !self.learning_objectives.is_empty() { + out.push(Section::Objectives); + } + if !self.learning_targets.is_empty() { + out.push(Section::Targets); + } + if !self.references.is_empty() { + out.push(Section::References); + } + if !self.stimuli.is_empty() { + out.push(Section::Stimuli); + } + out + } +} + +/// The fragment files of a course directory, in load order, with their roles. +/// +/// `course.yaml` is listed whether or not it exists, so a directory that is not +/// a course fails with a message naming the file it wanted rather than an empty +/// merge. Directory contents are sorted, which is what makes the merged course +/// independent of filesystem order. +/// +/// # Arguments +/// +/// * `layout` - the resolved course layout. +/// +/// # Returns +/// +/// Paths paired with what each file is allowed to define. +/// +/// # Errors +/// +/// Returns [`Error::Io`] when a fragment directory exists but cannot be read. +pub fn files(layout: &Layout) -> Result> { + let mut out = vec![(layout.course_file(), Role::Root)]; + + let references = layout.references_file(); + if references.is_file() { + out.push((references, Role::References)); + } + for path in yaml::list_yaml(&layout.lectures())? { + out.push((path, Role::Lecture)); + } + for path in yaml::list_yaml(&layout.objectives())? { + out.push((path, Role::Objective)); + } + Ok(out) +} + +/// Loads every fragment in a course directory and merges them into one course. +/// +/// # Arguments +/// +/// * `root` - the course directory. +/// +/// # Returns +/// +/// The merged course, with [`CourseFile::origins`] recording which file defined +/// each id. +/// +/// # Errors +/// +/// Propagates load errors, and returns [`Error::Invalid`] with every merge +/// problem at once: an id defined twice, a section in the wrong kind of file, a +/// fragment written against another major schema version, or no `course:` +/// section anywhere. +/// +/// Cross-references are *not* checked here. A dangling objective id is a +/// content problem, and content problems are [`CourseFile::validate`]'s, so that +/// they are reported the same way whether or not the course is split. +pub fn assemble(root: &Path) -> Result { + let layout = Layout::new(root); + let mut merge = Merge::default(); + + for (path, role) in files(&layout)? { + let fragment = Fragment::load(&path)?; + let shown = path.strip_prefix(root).unwrap_or(&path).to_path_buf(); + merge.take(&shown, role, fragment); + } + + merge.resolve(); + merge.finish(root) +} + +/// Accumulates fragments, remembering where each id came from. +#[derive(Debug, Default)] +struct Merge { + schema_version: Option, + course: Option, + policy: Option, + units: Vec, + lectures: BTreeMap, + objectives: BTreeMap, + targets: BTreeMap, + references: BTreeMap, + stimuli: BTreeMap, + origins: BTreeMap<(Section, String), PathBuf>, + issues: Vec, +} + +impl Merge { + /// Folds one fragment in. + /// + /// # Arguments + /// + /// * `path` - the fragment's path relative to the course root, for messages. + /// * `role` - what this file is allowed to define. + /// * `fragment` - the parsed fragment. + fn take(&mut self, path: &Path, role: Role, fragment: Fragment) { + for section in fragment.sections() { + if !role.allows(section) { + self.issues.push(format!( + "{}: a {} file may not define `{}`; that section belongs in {}", + path.display(), + role.label(), + section.key(), + section.home() + )); + } + } + + if let Some(declared) = &fragment.schema_version { + if !SUPPORTED_MAJORS.contains(&major(declared)) { + self.issues.push(format!( + "{}: declares schema_version {declared}, which this build cannot read. It \ + writes {SCHEMA_VERSION} and reads {}.", + path.display(), + SUPPORTED_MAJORS + .iter() + .map(|m| format!("{m}.x")) + .collect::>() + .join(" and ") + )); + } + if self.schema_version.is_none() { + self.schema_version = Some(declared.clone()); + } + } + + if role.allows(Section::Course) { + if let Some(course) = fragment.course { + if self.claim(Section::Course, path) { + self.course = Some(course); + } + } + } + if role.allows(Section::Policy) { + if let Some(policy) = fragment.policy { + if self.claim(Section::Policy, path) { + self.policy = Some(policy); + } + } + } + + if role.allows(Section::Units) { + for unit in fragment.units { + let key = (Section::Units, unit.id.clone()); + if let Some(first) = self.origins.get(&key) { + let message = duplicate(Section::Units, &unit.id, first.as_path(), path); + self.issues.push(message); + continue; + } + self.origins.insert(key, path.to_path_buf()); + self.units.push(unit); + } + } + + if role.allows(Section::Lectures) { + absorb( + &mut self.lectures, + fragment.lectures, + Section::Lectures, + path, + &mut self.origins, + &mut self.issues, + ); + } + if role.allows(Section::Objectives) { + absorb( + &mut self.objectives, + fragment.learning_objectives, + Section::Objectives, + path, + &mut self.origins, + &mut self.issues, + ); + } + if role.allows(Section::Targets) { + absorb( + &mut self.targets, + fragment.learning_targets, + Section::Targets, + path, + &mut self.origins, + &mut self.issues, + ); + } + if role.allows(Section::References) { + absorb( + &mut self.references, + fragment.references, + Section::References, + path, + &mut self.origins, + &mut self.issues, + ); + } + if role.allows(Section::Stimuli) { + absorb( + &mut self.stimuli, + fragment.stimuli, + Section::Stimuli, + path, + &mut self.origins, + &mut self.issues, + ); + } + } + + /// Records a section that may only be declared once. + /// + /// # Arguments + /// + /// * `section` - the section being claimed. + /// * `path` - the file claiming it. + /// + /// # Returns + /// + /// Whether the claim was the first, and so whether the caller should store + /// what it parsed. + fn claim(&mut self, section: Section, path: &Path) -> bool { + let key = (section, String::new()); + if let Some(first) = self.origins.get(&key) { + let message = format!( + "`{}` is declared twice: {} and {}. It applies to the whole course, so it has \ + one definition site.", + section.key(), + first.display(), + path.display() + ); + self.issues.push(message); + return false; + } + self.origins.insert(key, path.to_path_buf()); + true + } + + /// Fills in the two fields a split layout would otherwise duplicate. + fn resolve(&mut self) { + // A lecture says what it teaches; the objective's lecture list follows. + for (lecture_id, lecture) in &self.lectures { + for objective_id in &lecture.teaches { + if let Some(objective) = self.objectives.get_mut(objective_id) { + if !objective.lectures.iter().any(|l| l == lecture_id) { + objective.lectures.push(lecture_id.clone()); + } + } + } + } + + // Teaching order, from the two sequences that already declare it: the + // lectures in course order, and each lecture's `teaches` list. This is + // what lets an objective stop carrying a hand-kept integer. + let mut position = 0u32; + let ordered: Vec = self.lectures.keys().cloned().collect(); + for lecture_id in ordered { + let teaches = self + .lectures + .get(&lecture_id) + .map(|l| l.teaches.clone()) + .unwrap_or_default(); + for objective_id in teaches { + position += 1; + if let Some(objective) = self.objectives.get_mut(&objective_id) { + if objective.order.is_none() { + objective.order = Some(position); + } + } + } + } + + // A target with no lecture of its own is taught wherever its objective + // is. Collected first: the read of `objectives` and the write to + // `targets` cannot overlap in one pass. + let inherited: Vec<(String, Vec)> = self + .targets + .iter() + .filter(|(_, target)| target.lectures.is_empty()) + .filter_map(|(id, target)| { + self.objectives + .get(&target.objective) + .map(|objective| (id.clone(), objective.lectures.clone())) + }) + .collect(); + for (id, lectures) in inherited { + if let Some(target) = self.targets.get_mut(&id) { + target.lectures = lectures; + } + } + } + + /// Builds the course, or reports every merge problem at once. + fn finish(self, root: &Path) -> Result { + let mut issues = self.issues; + let course = match self.course { + Some(course) => course, + None => { + issues.push(format!( + "no file in {} declares a `course:` section, so the course has no code, \ + title, or term", + root.display() + )); + return Err(Error::Invalid(issues)); + } + }; + if !issues.is_empty() { + return Err(Error::Invalid(issues)); + } + + Ok(CourseFile { + schema_version: self + .schema_version + .unwrap_or_else(|| SCHEMA_VERSION.to_string()), + course, + policy: self.policy.unwrap_or_default(), + units: self.units, + lectures: self.lectures, + learning_objectives: self.objectives, + learning_targets: self.targets, + references: self.references, + stimuli: self.stimuli, + origins: self.origins, + }) + } +} + +/// Moves one section's entries across, refusing a second definition. +fn absorb( + into: &mut BTreeMap, + from: BTreeMap, + section: Section, + path: &Path, + origins: &mut BTreeMap<(Section, String), PathBuf>, + issues: &mut Vec, +) { + for (id, value) in from { + let key = (section, id.clone()); + if let Some(first) = origins.get(&key) { + issues.push(duplicate(section, &id, first.as_path(), path)); + continue; + } + origins.insert(key, path.to_path_buf()); + into.insert(id, value); + } +} + +/// The message for an id defined in two files. +fn duplicate(section: Section, id: &str, first: &Path, second: &Path) -> String { + format!( + "{} `{id}` is defined in two places: {} and {}. An id has one definition site; delete \ + one or rename it.", + section.noun(), + first.display(), + second.display() + ) +} + +/// The part of a schema version before the first dot. +fn major(version: &str) -> &str { + version.split('.').next().unwrap_or(version) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn tmp(tag: &str) -> PathBuf { + let p = std::env::temp_dir().join(format!("coursebank-frag-{tag}-{}", std::process::id())); + let _ = std::fs::remove_dir_all(&p); + std::fs::create_dir_all(&p).unwrap(); + p + } + + fn write(root: &Path, relative: &str, body: &str) { + let path = root.join(relative); + std::fs::create_dir_all(path.parent().unwrap()).unwrap(); + std::fs::write(path, body).unwrap(); + } + + const ROOT: &str = r#" +course: + code: BIOSC 1540 + title: Computational Biology + term: 2026f +policy: + points_per_item: 1.0 +units: + - id: u1 + title: Search and Similarity +"#; + + #[test] + fn a_split_course_merges_into_one_model() { + let root = tmp("merge"); + write(&root, "course.yaml", ROOT); + write( + &root, + "references.yaml", + "references:\n ismail2023:\n title: Bioinformatics\n", + ); + write( + &root, + "lectures/l-1-2.yaml", + "lectures:\n L1.2:\n title: The Digital Genome\n unit: u1\n \ + teaches: [lo-read-file-formats]\n", + ); + write( + &root, + "objectives/lo-read-file-formats.yaml", + "learning_objectives:\n lo-read-file-formats:\n text: Read the text formats.\n \ + unit: u1\nlearning_targets:\n t-fastq-structure:\n text: Identify the four \ + lines.\n objective: lo-read-file-formats\n", + ); + + let course = assemble(&root).unwrap(); + assert_eq!(course.course.code, "BIOSC 1540"); + assert_eq!(course.units.len(), 1); + assert_eq!(course.references.len(), 1); + assert!(course.validate().is_empty(), "{:?}", course.validate()); + + // Derived: the lecture registered the objective, and the target + // inherited the objective's lecture. + assert_eq!( + course.learning_objectives["lo-read-file-formats"].lectures, + vec!["L1.2".to_string()] + ); + assert_eq!( + course.learning_targets["t-fastq-structure"].lectures, + vec!["L1.2".to_string()] + ); + assert_eq!(course.lecture_targets("L1.2"), vec!["t-fastq-structure"]); + } + + #[test] + fn an_unsplit_course_file_still_loads() { + let root = tmp("monolith"); + write( + &root, + "course.yaml", + &format!( + "{ROOT}lectures:\n L1.2:\n title: The Digital Genome\nlearning_objectives:\n \ + lo-x:\n text: Do the thing.\n lectures: [L1.2]\nlearning_targets:\n \ + t-x:\n text: Do the smaller thing.\n objective: lo-x\nreferences:\n \ + ismail2023:\n title: Bioinformatics\n" + ), + ); + + let course = assemble(&root).unwrap(); + assert!(course.validate().is_empty(), "{:?}", course.validate()); + assert_eq!(course.lecture_objectives("L1.2"), vec!["lo-x"]); + assert_eq!( + course.origin("lo-x").map(|(_, p)| p.to_path_buf()), + Some(PathBuf::from(COURSE_FILE)) + ); + } + + #[test] + fn an_id_defined_twice_names_both_files() { + let root = tmp("dup"); + write(&root, "course.yaml", ROOT); + let body = "learning_objectives:\n lo-x:\n text: Do the thing.\n"; + write(&root, "objectives/lo-x.yaml", body); + write(&root, "objectives/lo-x-old.yaml", body); + + let err = assemble(&root).unwrap_err(); + let message = err.to_string(); + assert!(message.contains("objectives/lo-x.yaml"), "{message}"); + assert!(message.contains("objectives/lo-x-old.yaml"), "{message}"); + assert!(message.contains("one definition site"), "{message}"); + } + + #[test] + fn a_section_in_the_wrong_kind_of_file_is_rejected() { + let root = tmp("misplaced"); + write(&root, "course.yaml", ROOT); + write( + &root, + "lectures/l-1-2.yaml", + "lectures:\n L1.2:\n title: The Digital Genome\nlearning_objectives:\n lo-x:\n \ + text: Do the thing.\n", + ); + + let message = assemble(&root).unwrap_err().to_string(); + assert!( + message.contains("may not define `learning_objectives`"), + "{message}" + ); + assert!(message.contains("objectives/*.yaml"), "{message}"); + } + + #[test] + fn a_course_with_no_identity_says_so() { + let root = tmp("no-course"); + write(&root, "course.yaml", "units:\n - id: u1\n title: One\n"); + let message = assemble(&root).unwrap_err().to_string(); + assert!(message.contains("`course:` section"), "{message}"); + } + + #[test] + fn a_fragment_from_another_major_version_is_refused() { + let root = tmp("version"); + write(&root, "course.yaml", ROOT); + write( + &root, + "objectives/lo-x.yaml", + "schema_version: '9.0'\nlearning_objectives:\n lo-x:\n text: Do the thing.\n", + ); + let message = assemble(&root).unwrap_err().to_string(); + assert!(message.contains("schema_version 9.0"), "{message}"); + } + + #[test] + fn origins_point_at_the_fragment_that_defined_each_id() { + let root = tmp("origins"); + write(&root, "course.yaml", ROOT); + write( + &root, + "lectures/l-1-2.yaml", + "lectures:\n L1.2:\n title: The Digital Genome\n", + ); + write( + &root, + "objectives/lo-x.yaml", + "learning_objectives:\n lo-x:\n text: Do the thing.\n", + ); + + let course = assemble(&root).unwrap(); + let (section, path) = course.origin("lo-x").unwrap(); + assert_eq!(section, Section::Objectives); + assert_eq!(path, Path::new("objectives/lo-x.yaml")); + let (section, path) = course.origin("L1.2").unwrap(); + assert_eq!(section, Section::Lectures); + assert_eq!(path, Path::new("lectures/l-1-2.yaml")); + assert!(course.origin("nothing-like-this").is_none()); + } + + #[test] + fn teaching_order_is_derived_from_the_two_sequences_that_declare_it() { + let root = tmp("order"); + write(&root, "course.yaml", ROOT); + write( + &root, + "lectures/l-1-2.yaml", + "lectures:\n L1.2:\n title: One\n teaches: [lo-second, lo-first]\n", + ); + write( + &root, + "lectures/l-1-3.yaml", + "lectures:\n L1.3:\n title: Two\n teaches: [lo-third]\n", + ); + write( + &root, + "objectives/lo-first.yaml", + r#"learning_objectives: + lo-first: + text: A. + lo-second: + text: B. + lo-third: + text: C. +"#, + ); + + let course = assemble(&root).unwrap(); + // Position in `teaches`, not id order: the lecture lists `lo-second` + // first and that is what teaching it first means. + assert_eq!(course.learning_objectives["lo-second"].order, Some(1)); + assert_eq!(course.learning_objectives["lo-first"].order, Some(2)); + assert_eq!(course.learning_objectives["lo-third"].order, Some(3)); + assert_eq!( + course.lecture_objectives("L1.2"), + vec!["lo-second", "lo-first"] + ); + } + + #[test] + fn targets_print_by_ceiling_rather_than_in_a_sequence() { + let root = tmp("target-order"); + write(&root, "course.yaml", ROOT); + write( + &root, + "lectures/l-1-2.yaml", + "lectures:\n L1.2:\n title: One\n teaches: [lo-x]\n", + ); + write( + &root, + "objectives/lo-x.yaml", + r#"learning_objectives: + lo-x: + text: A. + level_ceiling: 3 +learning_targets: + t-predict: + text: Predict the effect. + objective: lo-x + level_ceiling: 3 + t-define: + text: Define the term. + objective: lo-x + level_ceiling: 1 + t-explain: + text: Explain the mechanism. + objective: lo-x + level_ceiling: 2 +"#, + ); + + let course = assemble(&root).unwrap(); + assert!(course.validate().is_empty(), "{:?}", course.validate()); + + // Shallowest first, which is the order the study methods apply in: + // recall, then explanation, then variations. Not id order, which would + // put `t-define` after `t-predict` for no reason at all. + assert_eq!( + course.targets("lo-x"), + vec!["t-define", "t-explain", "t-predict"] + ); + + // A target with no ceiling of its own inherits the objective's, so it + // sorts where that puts it. + assert_eq!(course.learning_targets["t-define"].order, None); + } + + #[test] + fn teaching_the_same_objective_from_two_lectures_unions() { + let root = tmp("union"); + write(&root, "course.yaml", ROOT); + write( + &root, + "lectures/l-1-2.yaml", + "lectures:\n L1.2:\n title: One\n teaches: [lo-x]\n", + ); + write( + &root, + "lectures/l-1-3.yaml", + "lectures:\n L1.3:\n title: Two\n teaches: [lo-x]\n", + ); + write( + &root, + "objectives/lo-x.yaml", + "learning_objectives:\n lo-x:\n text: Do the thing.\n", + ); + + let course = assemble(&root).unwrap(); + assert_eq!( + course.learning_objectives["lo-x"].lectures, + vec!["L1.2".to_string(), "L1.3".to_string()] + ); + } + + #[test] + fn a_declaration_on_both_sides_is_not_duplicated() { + let root = tmp("both-sides"); + write(&root, "course.yaml", ROOT); + write( + &root, + "lectures/l-1-2.yaml", + "lectures:\n L1.2:\n title: One\n teaches: [lo-x]\n", + ); + write( + &root, + "objectives/lo-x.yaml", + "learning_objectives:\n lo-x:\n text: Do the thing.\n lectures: [L1.2]\n", + ); + + let course = assemble(&root).unwrap(); + assert_eq!( + course.learning_objectives["lo-x"].lectures, + vec!["L1.2".to_string()] + ); + } + + #[test] + fn teaching_an_unknown_objective_is_a_validation_problem_not_a_merge_one() { + let root = tmp("unknown-teaches"); + write(&root, "course.yaml", ROOT); + write( + &root, + "lectures/l-1-2.yaml", + "lectures:\n L1.2:\n title: One\n teaches: [lo-nope]\n", + ); + + let course = assemble(&root).unwrap(); + let issues = course.validate(); + assert!(issues.iter().any(|i| i.contains("lo-nope")), "{issues:?}"); + } +} diff --git a/src/model/item.rs b/src/model/item.rs index b1bd877..a5e8ca8 100644 --- a/src/model/item.rs +++ b/src/model/item.rs @@ -19,8 +19,13 @@ //! it. That keeps bank files readable and reviewable in a pull request while //! still letting statistics accumulate across terms. -use serde::{Deserialize, Serialize}; +use std::fmt; +use serde::de::{self, MapAccess, Visitor}; +use serde::ser::SerializeMap; +use serde::{Deserialize, Deserializer, Serialize, Serializer}; + +use crate::course::Reference; use crate::date::Date; use crate::hash::fingerprint; use crate::taxonomy::{ @@ -40,9 +45,28 @@ pub struct Item { /// Revision counter, bumped whenever the content changes in a way that /// invalidates pooled statistics. - #[serde(default = "one_u32")] + /// Retained only so a pre-2.0 bank still loads. Ignored. + /// + /// A version number on a question answered the wrong question. It recorded + /// that *something* changed without constraining what, which meant an item + /// at version 3 might have a reworded distractor — fair, the statistics + /// still describe the same question — or a reworded stem, which makes it a + /// different question wearing the same id. Since 2.0 the stem *is* the + /// identity: reword it and you have a new item, with a new id and + /// [`Item::supersedes`] pointing back. [`Item::stem_digest`] is what + /// enforces that, against the seals of every administration. + /// + /// `coursebank migrate stems` removes it. + #[serde(default, skip_serializing)] pub version: u32, + /// The item this one replaces, when it is a rewording of an earlier stem. + /// + /// Lineage rather than versioning: both items stay in the bank, each with + /// its own statistics, and a report can say which one a cohort answered. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub supersedes: Option, + /// Workflow state; only [`Status::Approved`] items may be assembled. pub status: Status, @@ -79,11 +103,38 @@ pub struct Item { /// The answer options in canonical order. Shuffling happens at export time /// per form, never here, so the bank stays diffable. + /// + /// Empty for a [`Format::OpenResponse`] item, which is answered in free text + /// and graded from its [`Solution`] instead. A choice format must still supply + /// at least two, which [`crate::bank::BankFile::validate`] enforces; leaving + /// them out is reported there, with every other problem, rather than failing + /// the parse on its own. + #[serde(default, skip_serializing_if = "Vec::is_empty")] pub options: Vec, - /// Objectives this item measures, as ids into the course registry. - #[serde(default, skip_serializing_if = "Vec::is_empty")] - pub learning_objectives: Vec, + /// The worked solution: a model answer, an explanation a student can learn + /// from, and, for an open-response item, the rubric it is graded against. + /// + /// This is what the solutions document renders and what a paper answer key + /// prints. It is withheld from any question paper and from the exam payload, + /// the same way an option's `correct` flag is, so a document built for the + /// student cannot leak it. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub solution: Option, + + /// The learning targets this item measures, as ids into the course + /// registry. + /// + /// Targets rather than objectives: an item measures one specific + /// performance, and recording which one is what lets a report explain an + /// objective's result instead of only stating it. The objective follows from + /// the target, so it is never recorded twice. + #[serde( + default, + alias = "learning_objectives", + skip_serializing_if = "Vec::is_empty" + )] + pub learning_targets: Vec, /// Where the material was taught. #[serde(default, skip_serializing_if = "Vec::is_empty")] @@ -93,7 +144,7 @@ pub struct Item { #[serde(default, skip_serializing_if = "Vec::is_empty")] pub topics: Vec, - /// Item ids or objective ids a student needs before this is fair. + /// Item ids or registry ids a student needs before this is fair. #[serde(default, skip_serializing_if = "Vec::is_empty")] pub prerequisites: Vec, @@ -106,7 +157,14 @@ pub struct Item { pub design: Option, /// What the evidence says, accumulated across administrations. - #[serde(default, skip_serializing_if = "Option::is_none")] + /// What the statistics say, filled in from `analysis/` at load. + /// + /// Read from the store and never written back: `skip_serializing` means a + /// bank file cannot acquire a `calibration:` block by being round-tripped + /// through this type. A bank is a reviewed artifact whose diff should be a + /// change of intent, and every grading run would otherwise produce a diff + /// on it. See [`crate::calibration`]. + #[serde(default, skip_serializing)] pub calibration: Option, /// The last review decision recorded for this item. @@ -114,7 +172,16 @@ pub struct Item { pub review: Option, /// Append-only change log. - #[serde(default, skip_serializing_if = "Vec::is_empty")] + /// Retained only so a pre-2.0 bank still loads. Ignored. + /// + /// A hand-maintained change log inside a version-controlled file, every + /// entry of which duplicated what `git log -p` already knew, with no + /// guarantee of agreeing with it. What git cannot express is a claim about + /// the item rather than a record of an edit, and that has its own fields: + /// [`Item::retired`] and [`Item::supersedes`]. + /// + /// `coursebank migrate stems` removes it. + #[serde(default, skip_serializing)] pub history: Vec, /// The author of record. @@ -139,7 +206,17 @@ pub struct Item { #[derive(Debug, Clone, Serialize, Deserialize)] #[serde(deny_unknown_fields)] pub struct Choice { - /// Option letter, `A` through `H`. + /// The option's id, unique within its item: `o-fourth-line`. + /// + /// A name rather than a position. Until 2.0 this was a letter, which put a + /// position in a field that pooled statistics, student feedback, and + /// `credit_overrides` all join on — so reordering a YAML block silently + /// moved the misconception recorded against one option onto another. The + /// letter a student sees is derived per form from the form's seed and lives + /// in the seal; see [`crate::seal::printed_letter`]. + /// + /// A single letter `A` through `H` still loads, so a bank migrates when you + /// run `coursebank migrate options` rather than when you upgrade. pub id: String, /// The option text. @@ -187,6 +264,17 @@ pub struct Choice { /// Your a priori guess at how often this option is chosen. #[serde(default, skip_serializing_if = "Option::is_none")] pub selection_rate_expected: Option, + + /// Why this option is no longer drawn, when it is not. + /// + /// A retired option stays in the file forever. It has to: a seal and four + /// terms of response rows refer to it by id, and deleting it would turn + /// every one of those references into a dangling one. What retirement does + /// is take it out of the pool an assessment draws from, with the reason + /// attached — "selected by 1 of 96 across two administrations" is a finding + /// about the option, and the place for it is next to the option. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub retired: Option, } impl Choice { @@ -239,6 +327,227 @@ pub struct Source { pub recording_seconds: Option, } +/// A pointer from an item into the course reference registry: where to read more, +/// or what to revisit after missing the item. +/// +/// It holds a citation key and a locator rather than a restated citation, so a +/// reference is written once in `course.yaml` and a changed edition is a single +/// edit. The exporters resolve it against +/// [`crate::course::CourseFile::references`] into a short label such as `KKW §6.1`, +/// linked when the location resolves to a URL. This is the same pointer a lecture +/// [`crate::course::Reading`] uses, kept lean here because an item cites a reading; +/// it does not restate one. +/// +/// A citation may also be written as a bare string, which lands unparsed in `text` +/// and serializes back out as a string, so a bank that stored readings as plain +/// strings keeps loading and round-trips byte-for-byte. +#[derive(Debug, Clone, Default, PartialEq, Eq)] +pub struct Citation { + /// Citation key into the course reference registry. + pub reference: Option, + /// Where inside the work: `§6.1`, `pp. 212-219`, `fig. 4`. + pub locator: Option, + /// Appended to the reference's `base_url` to reach this location. + pub path: Option, + /// A full URL, when the location is not under the reference's `base_url`. + pub url: Option, + /// A citation written as a bare string, held unparsed. + pub text: Option, +} + +impl Citation { + /// A short display string that needs no reference lookup. + /// + /// Prefers the unparsed `text`, then the key and locator. A caller that holds + /// the course, such as an exporter, can resolve a nicer label and a link; this + /// is the fallback for one that does not. + /// + /// # Returns + /// + /// The display string, empty when the citation carries nothing. + pub fn display(&self) -> String { + if let Some(text) = &self.text { + return text.clone(); + } + match (&self.reference, &self.locator) { + (Some(k), Some(l)) => format!("{k} {l}"), + (Some(k), None) => k.clone(), + (None, Some(l)) => l.clone(), + (None, None) => String::new(), + } + } + + /// The link for this location, resolved against the work it points into. + /// + /// # Arguments + /// + /// * `reference` - the work, looked up from the citation key. + /// + /// # Returns + /// + /// The most specific link available. See [`Reference::href`]. + pub fn href(&self, reference: &Reference) -> Option { + reference.href(self.url.as_deref(), self.path.as_deref()) + } +} + +/// Writes a citation as a mapping, or as a bare string when that is all it holds. +impl Serialize for Citation { + fn serialize(&self, s: S) -> std::result::Result { + if let Some(text) = &self.text { + if self.reference.is_none() + && self.locator.is_none() + && self.path.is_none() + && self.url.is_none() + { + return s.serialize_str(text); + } + } + let mut map = s.serialize_map(None)?; + if let Some(v) = &self.reference { + map.serialize_entry("ref", v)?; + } + if let Some(v) = &self.locator { + map.serialize_entry("locator", v)?; + } + if let Some(v) = &self.path { + map.serialize_entry("path", v)?; + } + if let Some(v) = &self.url { + map.serialize_entry("url", v)?; + } + if let Some(v) = &self.text { + map.serialize_entry("text", v)?; + } + map.end() + } +} + +/// Accepts a citation written either as a mapping or as a bare string. +impl<'de> Deserialize<'de> for Citation { + fn deserialize>(d: D) -> std::result::Result { + /// The mapping form, with the field set kept in one place. + #[derive(Deserialize)] + #[serde(deny_unknown_fields)] + struct Mapping { + #[serde(rename = "ref", default)] + reference: Option, + #[serde(default)] + locator: Option, + #[serde(default)] + path: Option, + #[serde(default)] + url: Option, + #[serde(default)] + text: Option, + } + + struct V; + impl<'a> Visitor<'a> for V { + type Value = Citation; + + fn expecting(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str("a citation mapping with a `ref`, or a plain citation string") + } + + fn visit_str(self, v: &str) -> std::result::Result { + Ok(Citation { + text: Some(v.to_string()), + ..Citation::default() + }) + } + + fn visit_map>( + self, + map: M, + ) -> std::result::Result { + let m = Mapping::deserialize(de::value::MapAccessDeserializer::new(map))?; + Ok(Citation { + reference: m.reference, + locator: m.locator, + path: m.path, + url: m.url, + text: m.text, + }) + } + } + d.deserialize_any(V) + } +} + +/// One line of a grading rubric for an open-response item. +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct RubricCriterion { + /// What earns the points, e.g. "states H = U + PV" or "compares to ~2.5 kJ/mol". + pub description: String, + /// Points for this line. Absent lets a grader decide; when present, the lines + /// are meant to sum to the item's point value. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub points: Option, +} + +/// The worked solution to an item: what the answer is, why, and how it is graded. +/// +/// One place, versioned with the question, holds everything a student learns from +/// after the fact and everything a grader marks an open response against. For a +/// choice item the per-option [`Choice::explanation`] says why each option is right +/// or wrong; the solution adds the single worked line of reasoning a solutions +/// document leads with. For an [`Format::OpenResponse`] item the solution is the +/// whole answer, because there are no options to annotate. +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct Solution { + /// The model answer, in the authoring markup. For an open-response item this is + /// the response a full-credit student would write; for a choice item it is an + /// optional one-line statement of the key in words. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub model_answer: Option, + /// The worked reasoning a student can learn from: the derivation, the estimate, + /// the argument for the key over its neighbours. This is the body of the + /// solutions document. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub explanation: Option, + /// How an open response is graded, one criterion per line. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub rubric: Vec, + /// Responses a short constructed answer would be accepted as. Shown in the + /// solutions document as accepted answers, and the hook for automated grading + /// later. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub accepted: Vec, + /// Where to look again after missing this item, as citations into the course + /// reference registry. Resolved and linked by the exporters. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub review: Vec, +} + +impl Solution { + /// Whether the solution carries anything worth rendering. + /// + /// Used to decide whether a solutions entry has a body to print, so an item + /// with an empty `solution:` block is treated as having none. + pub fn is_empty(&self) -> bool { + self.model_answer.is_none() + && self.explanation.is_none() + && self.rubric.is_empty() + && self.accepted.is_empty() + && self.review.is_empty() + } + + /// Total of the rubric line points, when every line carries one. + /// + /// # Returns + /// + /// The sum, or `None` if any line omits its points or the rubric is empty. + pub fn rubric_points(&self) -> Option { + if self.rubric.is_empty() { + return None; + } + self.rubric.iter().map(|c| c.points).sum::>() + } +} + /// A figure or data file reproduced with an item. #[derive(Debug, Clone, Serialize, Deserialize)] #[serde(deny_unknown_fields)] @@ -314,6 +623,89 @@ pub struct Calibration { /// Machine-detected problems. #[serde(default, skip_serializing_if = "Vec::is_empty")] pub flags: Vec, + + /// One record per option set ever administered. + /// + /// What the flat fields above cannot express once options are a pool. A + /// stem shown with distractors `{third, second, first}` is a measurably + /// easier item than the same stem with `{third, plus-line, line-two}`, so a + /// p-value pooled across both is the average of two different questions. + /// Statistics are computed and compared per variant; the flat fields remain + /// as the pre-2.0 summary, and for an item whose pool is its form the two + /// agree. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub variants: Vec, + + /// One record per option, pooled across every set it appeared in. + /// + /// The capability the pool is worth the trouble for. Selection rates are + /// shares of a fixed set, so they are only comparable *within* a variant — + /// which means this view supports exactly one kind of claim, and it is the + /// useful one: this option draws nobody, anywhere. That is the evidence + /// that retires a distractor, and one administration cannot supply it. + #[serde(default, skip_serializing_if = "std::collections::BTreeMap::is_empty")] + pub options: std::collections::BTreeMap, +} + +/// Statistics for one option set, as administered. +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct VariantCalibration { + /// The digest this record describes. See [`Item::variant_digest`]. + pub variant: String, + /// The option ids keyed correct. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub key: Vec, + /// The option ids offered alongside them. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub distractors: Vec, + /// The administrations pooled into these numbers. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub administrations: Vec, + /// Examinees pooled. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub n_examinees: Option, + /// Proportion correct. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub p_value: Option, + /// Corrected item-total point-biserial correlation. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub point_biserial: Option, + /// Upper-minus-lower-group discrimination index. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub discrimination_index: Option, + /// Per-option behaviour within this set, keyed by option id. + #[serde(default, skip_serializing_if = "std::collections::BTreeMap::is_empty")] + pub option_stats: std::collections::BTreeMap, + /// Fitted item response theory parameters for this set. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub irt: Option, + /// Machine-detected problems with this set. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub flags: Vec, +} + +/// What one option has done across every set it has appeared in. +/// +/// Deliberately coarse. Averaging selection rates across variants is not +/// meaningful — each is a share of a different set — so `mean_selection_rate` +/// is a summary for reading, not a statistic to act on. `never_chosen` is the +/// one field that carries weight, and it needs several administrations to earn. +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct OptionHistory { + /// How many distinct variants this option has appeared in. + #[serde(default)] + pub appearances: usize, + /// Examinees who saw it, summed across those variants. + #[serde(default)] + pub n_examinees: usize, + /// Mean of its within-variant selection rates. For reading only. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub mean_selection_rate: Option, + /// Whether it has never been chosen, anywhere. + #[serde(default, skip_serializing_if = "is_false")] + pub never_chosen: bool, } /// How one option behaved. @@ -422,7 +814,7 @@ pub struct Retirement { #[derive(Debug, Clone, Serialize, Deserialize)] #[serde(deny_unknown_fields)] pub struct HistoryEntry { - /// The version this change produced. + /// The version this change produced. Ignored since 2.0. pub version: u32, /// When it was made. pub date: Date, @@ -433,6 +825,32 @@ pub struct HistoryEntry { pub change: String, } +/// The separator a pre-2.0 bank-qualified item id used: `b-1-2::q-fastq-line`. +pub const LEGACY_QUALIFIER: &str = "::"; + +/// An item id with any pre-2.0 bank qualifier removed. +/// +/// Until 2.0 an item was named `bank::item`, which made the file it happened to +/// live in part of its identity — and therefore part of the join key on every +/// row of response data ever collected. Moving a question between banks renamed +/// it. Since 2.0 the id names the item course-wide and the bank is only where it +/// is kept, so anything reading an old id strips the qualifier rather than +/// failing to match. +/// +/// # Arguments +/// +/// * `id` - an item id in either form. +/// +/// # Returns +/// +/// The part after the qualifier, or the whole id when there is none. +pub fn canonical_id(id: &str) -> &str { + match id.split_once(LEGACY_QUALIFIER) { + Some((_, rest)) => rest, + None => id, + } +} + impl Item { /// Builds a draft item with everything optional left empty. /// @@ -442,7 +860,7 @@ impl Item { /// the only way to get a half-built `Item` is deliberately. /// /// The result is `Status::Draft` and deliberately will not pass - /// [`Item::is_assemblable`] — it still needs learning objectives, sources, and + /// [`Item::is_assemblable`] — it still needs learning targets, sources, and /// a cognitive process before it can be drawn onto an assessment. /// /// # Arguments @@ -473,7 +891,8 @@ impl Item { stimulus: None, stem: stem.to_string(), options, - learning_objectives: Vec::new(), + solution: None, + learning_targets: Vec::new(), sources: Vec::new(), topics: Vec::new(), prerequisites: Vec::new(), @@ -485,6 +904,7 @@ impl Item { author: None, notes_private: None, retired: None, + supersedes: None, } } @@ -518,19 +938,38 @@ impl Item { out } - /// Looks up an option by letter. + /// Looks up an option by id. /// /// # Arguments /// - /// * `letter` - the option id, case insensitive. + /// * `id` - the option id. A pre-2.0 letter matches case-insensitively, + /// which a slug never needs but a hand-typed `d` does. /// /// # Returns /// /// The option, or `None`. - pub fn option(&self, letter: &str) -> Option<&Choice> { + pub fn option(&self, id: &str) -> Option<&Choice> { self.options .iter() - .find(|o| o.id.eq_ignore_ascii_case(letter)) + .find(|o| o.id == id) + .or_else(|| self.options.iter().find(|o| o.id.eq_ignore_ascii_case(id))) + } + + /// Whether an option id is a pre-2.0 letter rather than a name. + /// + /// # Arguments + /// + /// * `id` - the option id. + /// + /// # Returns + /// + /// `true` for `A` through `H`. + pub fn is_legacy_option_id(id: &str) -> bool { + id.len() == 1 + && id + .chars() + .next() + .is_some_and(|c| c.is_ascii_uppercase() && c <= 'H') } /// Whether the item keys more than one option. @@ -538,6 +977,14 @@ impl Item { self.key_indices().len() > 1 } + /// Whether the item presents selectable options, per its [`Format`]. + /// + /// `false` for an [`Format::OpenResponse`] item. Callers that would otherwise + /// index `options` or read a key should branch on this first. + pub fn has_options(&self) -> bool { + self.format.has_options() + } + /// The display title, falling back to a truncated stem. /// /// # Returns @@ -560,7 +1007,7 @@ impl Item { /// A content fingerprint over everything that affects what a student sees. /// - /// Metadata deliberately does not contribute: retagging an objective must not + /// Metadata deliberately does not contribute: retagging a target must not /// invalidate pooled statistics, but rewording an option must. /// /// # Returns @@ -586,6 +1033,159 @@ impl Item { fingerprint(parts.iter().map(|s| s.as_str())) } + /// The options an assessment administers, in the order the bank declares + /// them. + /// + /// Since 2.0 `options` is a *pool*: it may hold several defensible keys and + /// more distractors than any one form shows, and which of them a student + /// saw is a property of the placement rather than of the item. Everything + /// that renders, seals, decodes, or scores an administration has to work + /// from this rather than from `options`, or the paper and the key disagree. + /// + /// Bank order, not administered order: the per-form permutation is + /// [`crate::select::option_order`]'s business, and keeping the two separate + /// is what lets one item appear on three forms with one set of statistics. + /// + /// # Arguments + /// + /// * `key` - the option ids keyed correct for this administration. + /// * `distractors` - the option ids offered alongside them. + /// + /// # Returns + /// + /// The named options, or the whole live pool when `distractors` is empty. + /// + /// `distractors` is what says the set was chosen, not `key`. A pre-2.0 + /// record names its key and nothing else — `key: [D]` with no distractor + /// list — and it means "all of them, and D is the right one". Reading that + /// as "administer D alone" would print a one-option paper for every + /// assessment ever recorded. + pub fn administered(&self, key: &[String], distractors: &[String]) -> Vec<&Choice> { + if distractors.is_empty() { + return self + .options + .iter() + .filter(|o| o.retired.is_none()) + .collect(); + } + self.options + .iter() + .filter(|o| key.contains(&o.id) || distractors.contains(&o.id)) + .collect() + } + + /// The options that may still be drawn. + /// + /// # Returns + /// + /// Every option not retired, split into candidate keys and distractors. + pub fn pool(&self) -> (Vec<&Choice>, Vec<&Choice>) { + let live = || self.options.iter().filter(|o| o.retired.is_none()); + ( + live().filter(|o| o.correct).collect(), + live().filter(|o| !o.correct).collect(), + ) + } + + /// A digest of the item as one administration showed it. + /// + /// The pooling key for statistics, and the reason + /// [`Item::fingerprint`] cannot be. A stem with distractors + /// `{third, second, first}` is a measurably easier item than the same stem + /// with `{third, plus-line, line-two}`, so pooling a p-value across both is + /// averaging two different questions. Covers the stem, the administered + /// options, and which of them was keyed — the last because the same option + /// set with a different key is again a different item. + /// + /// # Arguments + /// + /// * `key` - the option ids keyed correct for this administration. + /// * `distractors` - the option ids offered alongside them. + /// + /// # Returns + /// + /// The digest as hex. + pub fn variant_digest(&self, key: &[String], distractors: &[String]) -> String { + let mut parts = vec![self.stem_digest()]; + let mut shown: Vec<&Choice> = self.administered(key, distractors); + shown.sort_by(|a, b| a.id.cmp(&b.id)); + for option in shown { + let keyed = if key.is_empty() { + option.correct + } else { + key.contains(&option.id) + }; + parts.push(format!( + "{}|{}|{}", + option.id, + if keyed { "1" } else { "0" }, + option.text.trim() + )); + } + fingerprint(parts.iter().map(|s| s.as_str())) + } + + /// A digest of what the item asks, without its options. + /// + /// The identity check. [`Item::fingerprint`] covers the options too, which + /// is right for calibration — reword a distractor and the pooled selection + /// rates no longer describe what students saw — but wrong for identity, + /// because a question whose distractors changed is still the same question. + /// This covers the stem and the stimulus, and nothing else. + /// + /// Compared against the digest each seal recorded, which is what makes + /// "a reworded stem is a new stem" a rule the tool enforces rather than a + /// convention that decays. + /// + /// # Returns + /// + /// The digest as hex. + pub fn stem_digest(&self) -> String { + let mut parts: Vec = vec![self.stem.trim().to_string()]; + if let Some(s) = &self.stimulus { + parts.push(format!("stimulus:{s}")); + } + fingerprint(parts.iter().map(|s| s.as_str())) + } + + /// The calibration recorded for one option set. + /// + /// # Arguments + /// + /// * `variant` - the digest from [`Item::variant_digest`]. + /// + /// # Returns + /// + /// The record, or `None` when this set has not been calibrated. + pub fn calibration_for(&self, variant: &str) -> Option<&VariantCalibration> { + self.calibration + .as_ref()? + .variants + .iter() + .find(|v| v.variant == variant) + } + + /// Whether a variant's recorded statistics still describe it. + /// + /// Staleness gets *narrower* with a pool rather than wider: rewording one + /// distractor used to invalidate the item's whole calibration, and now it + /// invalidates only the sets that distractor appeared in. + /// + /// # Arguments + /// + /// * `variant` - the digest to check. + /// + /// # Returns + /// + /// `false` only when a record exists for that digest and the digest no + /// longer matches what the option ids now say. + pub fn variant_is_current(&self, variant: &str) -> bool { + match self.calibration_for(variant) { + Some(record) => variant == self.variant_digest(&record.key, &record.distractors), + None => true, + } + } + /// Whether the recorded calibration matches the current content. /// /// # Returns @@ -641,16 +1241,20 @@ impl Item { } } - /// Appends a change-log entry and bumps the version. + /// Appends a change-log entry. + /// + /// Kept for the pre-2.0 banks that still carry a `history:` block, so + /// reading one and writing it back does not silently drop entries. New + /// entries belong in a commit message. /// /// # Arguments /// /// * `change` - a description of what changed. /// * `author` - who made the change. pub fn record_change(&mut self, change: &str, author: Option<&str>) { - self.version += 1; + let version = self.history.iter().map(|h| h.version).max().unwrap_or(0) + 1; self.history.push(HistoryEntry { - version: self.version, + version, date: Date::today(), author: author.map(|a| a.to_string()), change: change.to_string(), @@ -658,9 +1262,6 @@ impl Item { } } -fn one_u32() -> u32 { - 1 -} fn default_format() -> Format { Format::SingleBestAnswer } @@ -690,7 +1291,6 @@ options: #[test] fn minimal_item_parses_with_defaults() { let it = item(MINIMAL); - assert_eq!(it.version, 1); assert_eq!(it.format, Format::SingleBestAnswer); assert!(!it.bonus); assert_eq!(it.key_letters(), vec!["A"]); @@ -726,7 +1326,7 @@ options: let base = item(MINIMAL); let mut retagged = base.clone(); retagged.topics = vec!["kinetics".into()]; - retagged.learning_objectives = vec!["lo-a".into()]; + retagged.learning_targets = vec!["lo-a".into()]; retagged.author = Some("someone".into()); assert_eq!( base.fingerprint(), @@ -762,6 +1362,101 @@ options: assert_eq!(a.fingerprint(), b.fingerprint()); } + #[test] + fn a_pool_administers_a_subset_and_defaults_to_everything() { + let mut it = item(MINIMAL); + let all: Vec = it.options.iter().map(|o| o.id.clone()).collect(); + + // Unstated means the whole pool, which is what a pre-2.0 record meant. + assert_eq!(it.administered(&[], &[]).len(), all.len()); + + // A retired option leaves the pool but not the file. + it.options[1].retired = Some(Retirement { + on: Date::new(2026, 9, 20).unwrap(), + reason: "chosen by 1 of 96 across two administrations".into(), + replaced_by: None, + }); + let shown = it.administered(&[], &[]); + assert_eq!(shown.len(), all.len() - 1); + assert!(!shown.iter().any(|o| o.id == all[1])); + // Still resolvable: a seal and four terms of rows refer to it. + assert!(it.option(&all[1]).is_some()); + + // A key with no distractor list is a pre-2.0 record, and it means all + // of them. Reading it as "administer the key alone" would print a + // one-option paper for every assessment already recorded. + assert_eq!(it.administered(&[all[2].clone()], &[]).len(), all.len() - 1); + + // Named explicitly, bank order is kept whatever order the lists are in. + let shown = it.administered(&[all[2].clone()], &[all[0].clone()]); + assert_eq!( + shown.iter().map(|o| o.id.clone()).collect::>(), + vec![all[0].clone(), all[2].clone()] + ); + } + + #[test] + fn the_variant_digest_tracks_the_option_set_and_the_stem_does_not() { + let it = item(MINIMAL); + let ids: Vec = it.options.iter().map(|o| o.id.clone()).collect(); + + let one = it.variant_digest(&[ids[0].clone()], &[ids[1].clone()]); + let two = it.variant_digest(&[ids[0].clone()], &[ids[2].clone()]); + // A different distractor is a different item: same stem, different + // difficulty, so pooling a p-value across both would average two + // questions. + assert_ne!(one, two, "a swapped distractor is a new variant"); + // The stem is unmoved by any of it. + assert_eq!(it.stem_digest(), item(MINIMAL).stem_digest()); + + // Order of the lists is not part of the identity. + assert_eq!( + it.variant_digest(&[ids[0].clone()], &[ids[2].clone(), ids[1].clone()]), + it.variant_digest(&[ids[0].clone()], &[ids[1].clone(), ids[2].clone()]) + ); + } + + #[test] + fn a_variant_goes_stale_alone_rather_than_taking_the_item_with_it() { + let mut it = item(MINIMAL); + let ids: Vec = it.options.iter().map(|o| o.id.clone()).collect(); + let one = it.variant_digest(&[ids[0].clone()], &[ids[1].clone()]); + let two = it.variant_digest(&[ids[0].clone()], &[ids[2].clone()]); + + it.calibration = Some(Calibration { + variants: vec![ + VariantCalibration { + variant: one.clone(), + key: vec![ids[0].clone()], + distractors: vec![ids[1].clone()], + n_examinees: Some(96), + p_value: Some(0.84), + ..VariantCalibration::default() + }, + VariantCalibration { + variant: two.clone(), + key: vec![ids[0].clone()], + distractors: vec![ids[2].clone()], + n_examinees: Some(32), + ..VariantCalibration::default() + }, + ], + ..Calibration::default() + }); + + assert_eq!(it.calibration_for(&one).unwrap().n_examinees, Some(96)); + assert!(it.calibration_for("nothing-like-this").is_none()); + assert!(it.variant_is_current(&one)); + assert!(it.variant_is_current(&two)); + + // Rewording the option that only the second set used leaves the first + // set's numbers standing. Before the pool, one distractor edit + // invalidated every statistic the item had. + it.options[2].text = "a different distractor".into(); + assert!(it.variant_is_current(&one), "the first set never showed it"); + assert!(!it.variant_is_current(&two)); + } + #[test] fn stale_calibration_is_detectable() { let mut it = item(MINIMAL); @@ -805,12 +1500,24 @@ options: } #[test] - fn record_change_bumps_version_and_logs() { + fn the_stem_is_the_identity_rather_than_a_version_number() { let mut it = item(MINIMAL); + let before = it.stem_digest(); + + // A change log entry numbers itself and leaves the item alone: since + // 2.0 nothing reads `version`, and rewording a stem is not a version + // bump but a new item. it.record_change("clarified the stem", Some("Alex")); - assert_eq!(it.version, 2); + assert_eq!(it.version, 0); assert_eq!(it.history.len(), 1); - assert_eq!(it.history[0].version, 2); + assert_eq!(it.history[0].version, 1); + assert_eq!(it.stem_digest(), before); + + // The options are the fingerprint's business, not the stem's. + it.options[1].text = "a different distractor".into(); + assert_eq!(it.stem_digest(), before); + it.stem = "What is y?".into(); + assert_ne!(it.stem_digest(), before); } #[test] @@ -821,4 +1528,77 @@ options: IrtModel::ThreePl ); } + + #[test] + fn open_response_item_parses_without_options() { + let it = item( + r#" +id: q-enthalpy-op-001 +status: draft +level: 2 +format: open_response +stem: Explain why, at constant pressure, the heat exchanged equals the enthalpy change. +solution: + model_answer: >- + At constant pressure the P dV expansion work is folded into H = U + PV, so the + heat q_p equals the change in H. + rubric: + - { description: "states H = U + PV", points: 1 } + - { description: "identifies q_p with the enthalpy change", points: 1 } + review: + - { ref: kuriyan2012molecules, locator: "§6.4", path: "6/A/#4" } +"#, + ); + assert_eq!(it.format, Format::OpenResponse); + assert!(it.options.is_empty()); + assert!(!it.has_options()); + assert!(it.key_letters().is_empty()); + let sol = it.solution.as_ref().expect("has a solution"); + assert!(!sol.is_empty()); + assert_eq!(sol.rubric_points(), Some(2.0)); + assert_eq!(sol.review.len(), 1); + assert_eq!( + sol.review[0].reference.as_deref(), + Some("kuriyan2012molecules") + ); + } + + #[test] + fn a_solution_serializes_only_what_it_holds() { + let it = item( + r#" +id: q-demo-op-002 +status: draft +level: 2 +format: open_response +stem: State the first law. +solution: + model_answer: The total energy of an isolated system is constant. +"#, + ); + let yaml = serde_yaml_ng::to_string(&it).expect("serializes"); + // Open-response items carry no options key, and an empty rubric is omitted. + assert!(!yaml.contains("options:"), "no options key:\n{yaml}"); + assert!(!yaml.contains("rubric"), "empty rubric omitted:\n{yaml}"); + assert!(yaml.contains("model_answer:")); + } + + #[test] + fn a_citation_round_trips_as_string_or_mapping() { + // A bare string stays a bare string. + let bare: Citation = serde_yaml_ng::from_str("\"KKW §6.4 (course reserve)\"").unwrap(); + assert_eq!(bare.text.as_deref(), Some("KKW §6.4 (course reserve)")); + assert_eq!(bare.display(), "KKW §6.4 (course reserve)"); + let back = serde_yaml_ng::to_string(&bare).unwrap(); + assert_eq!(back.trim(), "KKW §6.4 (course reserve)"); + + // A mapping keeps its fields, and `ref` is the key's YAML spelling. + let mapped: Citation = + serde_yaml_ng::from_str("{ ref: kuriyan2012molecules, locator: \"§6.4\" }").unwrap(); + assert_eq!(mapped.reference.as_deref(), Some("kuriyan2012molecules")); + assert_eq!(mapped.display(), "kuriyan2012molecules §6.4"); + let back = serde_yaml_ng::to_string(&mapped).unwrap(); + assert!(back.contains("ref: kuriyan2012molecules")); + assert!(back.contains("locator:")); + } } diff --git a/src/model/layout.rs b/src/model/layout.rs index de454cf..632b097 100644 --- a/src/model/layout.rs +++ b/src/model/layout.rs @@ -3,6 +3,11 @@ // Source: https://git.scient.ing/education/coursebank //! The on-disk layout of a course directory. +//! +//! Two of these directories hold fragments of the course file rather than files +//! of their own kind: `lectures/` and `objectives/` are merged into one +//! [`crate::course::CourseFile`] on load, along with `references.yaml`. See +//! [`crate::course::fragment`]. use std::path::PathBuf; @@ -35,11 +40,48 @@ impl Layout { self.root.join(COURSE_FILE) } + /// Path to `references.yaml`, the bibliography when it is kept out of + /// `course.yaml`. + /// + /// Optional: absent means the course keeps its `references:` section in the + /// course file, which is how an unsplit course is arranged. + pub fn references_file(&self) -> PathBuf { + self.root.join(crate::course::fragment::REFERENCES_FILE) + } + + /// Directory holding one file per lecture. + pub fn lectures(&self) -> PathBuf { + self.root.join("lectures") + } + + /// Directory holding one file per learning objective. + pub fn objectives(&self) -> PathBuf { + self.root.join("objectives") + } + /// Directory holding item bank YAML files. pub fn banks(&self) -> PathBuf { self.root.join("banks") } + /// Directory holding the statistics, which are kept out of the banks. + /// + /// Committed, unlike [`Layout::data`]: everything under here is a cohort + /// aggregate with no student in it. See [`crate::calibration`]. + pub fn analysis(&self) -> PathBuf { + self.root.join("analysis") + } + + /// The pooled per-item calibration store. + pub fn calibration_file(&self) -> PathBuf { + self.analysis().join(crate::calibration::CALIBRATION_FILE) + } + + /// Directory holding one immutable record per administration. + pub fn measurements(&self) -> PathBuf { + self.analysis().join("administrations") + } + /// Directory holding assessment records. pub fn assessments(&self) -> PathBuf { self.root.join("assessments") @@ -76,6 +118,15 @@ impl Layout { self.root.join("templates") } + /// Directory holding sealed administrations. + /// + /// Deliberately not `assessments/`: + /// [`crate::assessment::AssessmentFile::load_all`] parses every `.yaml` in + /// that directory, and a seal is not an assessment record. + pub fn seals(&self) -> PathBuf { + self.root.join(crate::seal::SEAL_DIR) + } + /// Creates every directory in the layout. /// /// # Errors @@ -84,6 +135,10 @@ impl Layout { pub fn create_all(&self) -> Result<()> { for dir in [ self.root.clone(), + self.analysis(), + self.measurements(), + self.lectures(), + self.objectives(), self.banks(), self.assessments(), self.data(), diff --git a/src/model/seal.rs b/src/model/seal.rs new file mode 100644 index 0000000..a6d760f --- /dev/null +++ b/src/model/seal.rs @@ -0,0 +1,1435 @@ +// SPDX-License-Identifier: Prosperity-3.0.0 +// Copyright Scientific Computing Studio +// Source: https://git.scient.ing/education/coursebank + +//! The sealed administration: what the students actually held in their hands. +//! +//! An assessment record says which item sat at question 14. It does not say what +//! that item *said*, and it does not say that on form B question 14's option C was +//! the bank's option A. Both facts are derivable — the first from the bank, the +//! second from the form seed — right up until the bank changes, and then they are +//! silently derivable to the wrong answer. +//! +//! A seal freezes them. It is written once, before the exam is administered, and +//! it holds three things the record does not: +//! +//! * **The content as administered.** Stem and option text, verbatim, with a +//! fingerprint on each. Reword a distractor next term and the seal still shows +//! what this cohort was asked. +//! * **The permutation, expanded.** For every form and every printed question, the +//! map from printed letter to bank letter, plus the key as printed. This is what +//! turns "the student chose C" into "the student chose the bank's option A", +//! which is the only form of that sentence worth storing. +//! * **Digests over both.** One per form, one per item, and one over the whole +//! file. Re-derive any of them and compare; a mismatch names what moved. +//! +//! The seal is not a cache. Nothing reads it for speed. It exists so that the +//! question "is what I am about to analyze the same thing the students saw?" has +//! an answer that does not depend on the bank having stayed still. +//! +//! # Where it lives +//! +//! `seals/.yaml`, in its own directory rather than beside the +//! assessment record, because [`crate::assessment::AssessmentFile::load_all`] +//! parses every `.yaml` in `assessments/` and a seal is not an assessment. +//! +//! # The order of operations +//! +//! ```text +//! assemble ──▶ export typst ──▶ seal ──▶ print ──▶ administer ──▶ ingest +//! │ │ +//! └────────── verified against ──────┘ +//! ``` +//! +//! Seal after exporting and before printing. Sealing earlier is harmless; +//! sealing after ingest is not, because a seal written from a drifted bank +//! records the drift as truth. + +use std::collections::{BTreeMap, BTreeSet}; +use std::path::{Path, PathBuf}; + +use serde::{Deserialize, Serialize}; + +use crate::assessment::{AssessmentFile, Form, Placement}; +use crate::catalog::{Catalog, Severity}; +use crate::course::SCHEMA_VERSION; +use crate::date::Date; +use crate::error::{Error, Result}; +use crate::hash::{fingerprint, hex, sha256}; +use crate::item::Item; +use crate::layout::Layout; +use crate::select; +use crate::taxonomy::Level; +use crate::yaml; + +/// The directory holding seals, relative to the course root. +pub const SEAL_DIR: &str = "seals"; + +/// The prefix on a digest string, naming the algorithm so a future change is +/// visible in the file rather than inferred from the length. +const DIGEST_PREFIX: &str = "sha256:"; + +/// A whole seal file. +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct SealFile { + /// Schema version this file targets. + #[serde( + default = "default_version", + deserialize_with = "yaml::flexible_string" + )] + pub schema_version: String, + + /// Identity, provenance, and the digest over everything else. + pub seal: SealMeta, + + /// The items as administered, by recorded question number. + #[serde(default)] + pub items: Vec, + + /// One entry per form, holding that form's expanded permutation. + #[serde(default)] + pub forms: Vec, +} + +/// Identity and provenance. +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct SealMeta { + /// The assessment id this seals. + pub assessment: String, + /// The assessment title, denormalized so the file reads standalone. + pub title: String, + /// The course code. + pub course: String, + /// The term. + pub term: String, + /// The administration date, when the record carried one. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub date: Option, + /// The day the seal was written. + pub sealed_on: Date, + /// The tool and version that wrote it. + pub generator: Generator, + /// Whether option and stem text were stored, or only their fingerprints. + #[serde(default = "yes")] + pub content: bool, + /// Recorded question numbers that were already dropped when the seal was + /// written, and therefore were not printed. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub dropped: Vec, + /// The digest over every item and form below. + /// + /// Written last and checked first. [`SealFile::recompute_digest`] rebuilds it + /// from the file's own contents; [`SealFile::verify`] rebuilds a seal from the + /// live course and compares this against it. + pub digest: String, + + /// The digests this file had before it was rewritten, oldest first. + /// + /// A seal is meant to be tamper-evident, which puts a migration that + /// renames an id inside it in an awkward position: the id is part of + /// [`SealFile::digest_input`], so rewriting the file invalidates the digest + /// that vouched for it, and recomputing it quietly would produce a record + /// that looks untouched and is not. + /// + /// So a migration does both — recomputes the digest and leaves the old one + /// here. What the seal then says is the honest thing: these are the + /// contents, this is what they hash to, and here is what the file hashed to + /// before each rewrite. Anyone holding an earlier copy, a backup or a git + /// revision, can check it against the right entry. + /// + /// Deliberately outside [`SealFile::digest_input`]: a field that recorded + /// past digests and was itself covered by the current one could not be + /// appended to without invalidating what it describes. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub superseded_digests: Vec, +} + +/// What wrote a seal. +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct Generator { + /// The tool name. + pub tool: String, + /// The tool version. + pub version: String, +} + +/// One item, frozen as administered. +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct SealedItem { + /// The recorded question number, which is the join key to grading exports. + pub number: u32, + /// The item's global id. + pub item: String, + /// The item version as administered. + /// Retained only so a pre-2.0 seal still loads. Ignored when reasoning + /// about the item, but still written: [`SealFile::digest_input`] covers it, + /// so dropping it on a round trip would invalidate the digest of every + /// administration sealed before 2.0. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub version: Option, + + /// The stem's digest as administered. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub stem_digest: Option, + /// The item's content fingerprint, the same one + /// [`crate::item::Item::fingerprint`] computes, so a seal and a bank can be + /// compared without re-hashing either by hand. + pub fingerprint: String, + /// Points as administered. + pub points: f64, + /// Whether it was scored as bonus. + #[serde(default, skip_serializing_if = "is_false")] + pub bonus: bool, + /// The cognitive level as administered. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub level: Option, + /// Objectives as administered. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub learning_targets: Vec, + /// The keyed letters in the bank's own lettering, before any shuffle. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub key: Vec, + /// The stem as administered. Absent when the seal was written without content. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub stem: Option, + /// The options in bank order. + #[serde(default)] + pub options: Vec, +} + +/// One option, frozen as administered. +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct SealedOption { + /// The letter this option carries in the bank. + pub letter: String, + /// Whether it was keyed correct. + #[serde(default, skip_serializing_if = "is_false")] + pub correct: bool, + /// Credit it earned, when the bank set one explicitly. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub credit: Option, + /// A fingerprint of the option text, so drift is detectable even when the + /// seal was written without content. + pub digest: String, + /// The option text as administered. Absent when the seal was written without + /// content. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub text: Option, +} + +/// One form, with its permutation expanded rather than implied by a seed. +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct SealedForm { + /// The form id, e.g. `A`. + pub id: String, + /// The seed the permutation came from, kept so it can be re-derived. + pub seed: u64, + /// Whether item order was permuted. + #[serde(default)] + pub shuffle_items: bool, + /// Whether option order was permuted. + #[serde(default)] + pub shuffle_options: bool, + /// A digest over this form's question list alone, so a single form can be + /// checked without reading the rest of the file. + pub digest: String, + /// The printed questions, in printed order. + #[serde(default)] + pub questions: Vec, +} + +/// One question as printed on one form. +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct SealedQuestion { + /// Where it sat on the page, counting from 1. This is what a grading export + /// names its file after. + pub position: u32, + /// The recorded question number, which is what the assessment record and the + /// response store use. + pub number: u32, + /// The item's global id, repeated here so a form block reads on its own. + pub item: String, + /// The keyed letters *as printed on this form*. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub printed_key: Vec, + /// The letter map, in printed order: `printed` is what the student saw, + /// `canonical` is the bank's letter for the same option. + #[serde(default)] + pub options: Vec, +} + +/// One printed letter and the bank letter it stands for. +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct LetterMap { + /// The letter as printed on this form. + pub printed: String, + /// The letter the same option carries in the bank. + pub canonical: String, +} + +/// What to put in a seal. +#[derive(Debug, Clone)] +pub struct Options { + /// Whether to store stem and option text, not just fingerprints. + /// + /// On by default. The whole point of a seal is to answer "what did they + /// actually see" after the bank has moved on, and a fingerprint answers only + /// "not this". + pub content: bool, + /// Which forms to seal. Empty means every form the record declares, or a + /// single unshuffled form `A` when it declares none. + pub forms: Vec, +} + +impl Default for Options { + fn default() -> Options { + Options { + content: true, + forms: Vec::new(), + } + } +} + +/// One way a live course has drifted from a seal. +#[derive(Debug, Clone)] +pub struct Drift { + /// How much it matters. + pub severity: Severity, + /// A machine-readable code, so a check can be silenced or grepped. + pub code: &'static str, + /// What moved, in a sentence. + pub message: String, +} + +impl Drift { + /// Builds a drift finding. + /// + /// # Arguments + /// + /// * `severity` - how much it matters. + /// * `code` - the stable code. + /// * `message` - the explanation. + /// + /// # Returns + /// + /// The finding. + fn new(severity: Severity, code: &'static str, message: impl Into) -> Drift { + Drift { + severity, + code, + message: message.into(), + } + } + + /// Whether this finding should stop an analysis rather than annotate it. + pub fn is_blocking(&self) -> bool { + self.severity == Severity::High + } +} + +impl std::fmt::Display for Drift { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "[{}] {}", self.code, self.message) + } +} + +/// The printed label for a zero-based option position. +/// +/// This mirrors [`crate::typst::config::LetterStyle::Upper`], which is the +/// lettering every bundled exam template uses and the only lettering a Gradescope +/// rubric column can carry. A course that prints numeric or Roman option labels +/// must map them back to letters before ingest; the seal always speaks letters. +/// +/// # Arguments +/// +/// * `position` - the printed position, counting from zero. +/// +/// # Returns +/// +/// `A`, `B`, ... `Z`, `AA`. +pub fn printed_letter(position: usize) -> String { + let mut n = position; + let mut letters = Vec::new(); + loop { + letters.push((b'A' + (n % 26) as u8) as char); + if n < 26 { + break; + } + n = n / 26 - 1; + } + letters.iter().rev().collect() +} + +/// The forms to seal, resolving an empty request and a record with no forms. +/// +/// # Arguments +/// +/// * `record` - the assessment record. +/// * `wanted` - form ids, or empty for all of them. +/// +/// # Returns +/// +/// The forms, in the record's order. +/// +/// # Errors +/// +/// Returns [`Error::Usage`] when a named form is not declared. +fn forms_to_seal(record: &AssessmentFile, wanted: &[String]) -> Result> { + if record.forms.is_empty() { + // A record with no declared forms was still printed once, and that single + // printing is a form: unshuffled, seed zero. Sealing it costs nothing and + // means the ingest path has one shape rather than two. + return Ok(vec![Form { + id: "A".to_string(), + seed: 0, + shuffle_items: false, + shuffle_options: false, + }]); + } + if wanted.is_empty() { + return Ok(record.forms.clone()); + } + let mut out = Vec::new(); + for id in wanted { + let form = record + .forms + .iter() + .find(|f| f.id.eq_ignore_ascii_case(id)) + .ok_or_else(|| { + Error::usage(format!( + "no form `{id}` on this assessment; it declares {}", + record + .forms + .iter() + .map(|f| f.id.as_str()) + .collect::>() + .join(", ") + )) + })?; + out.push(form.clone()); + } + Ok(out) +} + +/// Builds a seal from the live course. +/// +/// # Arguments +/// +/// * `catalog` - the loaded course. +/// * `record` - the assessment record. +/// * `opts` - what to include. +/// +/// # Returns +/// +/// The seal, with every digest computed. +/// +/// # Errors +/// +/// Returns [`Error::Unresolved`] when a placement references an item that is not +/// in the bank, and [`Error::Usage`] when a requested form is not declared. +pub fn build(catalog: &Catalog, record: &AssessmentFile, opts: &Options) -> Result { + let default_points = catalog.course.policy.points_per_item; + let forms = forms_to_seal(record, &opts.forms)?; + + let mut items = Vec::new(); + for placement in &record.items { + let entry = catalog.require(&placement.item)?; + items.push(sealed_item(placement, &entry.item, default_points, opts)); + } + items.sort_by_key(|i| i.number); + + let mut sealed_forms = Vec::new(); + for form in &forms { + sealed_forms.push(sealed_form(catalog, record, form)?); + } + + let dropped: Vec = record + .items + .iter() + .filter(|p| p.dropped) + .map(|p| p.number) + .collect(); + + let mut file = SealFile { + schema_version: SCHEMA_VERSION.to_string(), + seal: SealMeta { + assessment: record.assessment.id.clone(), + title: record.assessment.title.clone(), + course: catalog.course.course.code.clone(), + term: record + .assessment + .term + .clone() + .unwrap_or_else(|| catalog.course.course.term.clone()), + date: record.assessment.date, + sealed_on: Date::today(), + generator: Generator { + tool: "coursebank".to_string(), + version: crate::VERSION.to_string(), + }, + content: opts.content, + dropped, + digest: String::new(), + superseded_digests: Vec::new(), + }, + items, + forms: sealed_forms, + }; + file.seal.digest = file.recompute_digest(); + Ok(file) +} + +/// Freezes one item. +fn sealed_item( + placement: &Placement, + item: &Item, + default_points: f64, + opts: &Options, +) -> SealedItem { + let options: Vec = item + .options + .iter() + .map(|choice| SealedOption { + letter: choice.id.clone(), + correct: choice.correct, + credit: choice.credit, + digest: fingerprint([choice.id.as_str(), choice.text.trim()]), + text: if opts.content { + Some(choice.text.clone()) + } else { + None + }, + }) + .collect(); + + SealedItem { + number: placement.number, + item: placement.item.clone(), + version: None, + stem_digest: Some(item.stem_digest()), + fingerprint: item.fingerprint(), + points: placement + .points + .unwrap_or_else(|| item.points(default_points)), + bonus: placement.bonus, + level: placement.level.or(Some(item.level)), + learning_targets: if placement.learning_targets.is_empty() { + item.learning_targets.clone() + } else { + placement.learning_targets.clone() + }, + key: if placement.key.is_empty() { + item.key_letters() + } else { + placement.key.clone() + }, + stem: if opts.content { + Some(item.stem.clone()) + } else { + None + }, + options, + } +} + +/// Expands one form's permutation. +/// +/// The expansion comes from [`select::layout`] and [`select::option_order`], the +/// same two functions every export calls, so a seal cannot describe a paper the +/// exporter would not have produced. +fn sealed_form(catalog: &Catalog, record: &AssessmentFile, form: &Form) -> Result { + let mut questions = Vec::new(); + + // Only an item pulled before printing takes no printed position. An item + // dropped from scoring after the exam was on the page and keeps its place, + // so re-sealing after a drop describes the same paper the students held + // rather than renumbering everything after it. + let printed: Vec = select::layout(record, form) + .into_iter() + .filter(|p| p.was_printed()) + .collect(); + + for (index, placement) in printed.iter().enumerate() { + let entry = catalog.require(&placement.item)?; + let item = &entry.item; + // The pool is not the paper: seal what this placement administered. + let shown = item.administered(&placement.key, &placement.distractors); + let n = shown.len(); + let order = select::option_order(form, &placement.item, n); + + let canonical_key: BTreeSet = if placement.key.is_empty() { + item.key_letters().into_iter().collect() + } else { + placement.key.iter().cloned().collect() + }; + + let mut options = Vec::with_capacity(n); + let mut printed_key = Vec::new(); + for (position, source_index) in order.iter().enumerate() { + let canonical = shown + .get(*source_index) + .map(|c| c.id.clone()) + .unwrap_or_else(|| printed_letter(*source_index)); + let printed = printed_letter(position); + if canonical_key.contains(&canonical) { + printed_key.push(printed.clone()); + } + options.push(LetterMap { printed, canonical }); + } + + questions.push(SealedQuestion { + position: index as u32 + 1, + number: placement.number, + item: placement.item.clone(), + printed_key, + options, + }); + } + + let mut sealed = SealedForm { + id: form.id.clone(), + seed: form.seed, + shuffle_items: form.shuffle_items, + shuffle_options: form.shuffle_options, + digest: String::new(), + questions, + }; + sealed.digest = form_digest(&sealed); + Ok(sealed) +} + +/// The digest over one form's question list. +fn form_digest(form: &SealedForm) -> String { + let mut buf = String::new(); + buf.push_str(&format!( + "form\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\n", + form.id, form.seed, form.shuffle_items, form.shuffle_options + )); + for q in &form.questions { + buf.push_str(&format!( + "q\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}", + q.position, + q.number, + q.item, + q.printed_key.join(",") + )); + for map in &q.options { + buf.push_str(&format!("{}>{};", map.printed, map.canonical)); + } + buf.push('\n'); + } + digest_of(&buf) +} + +/// A digest string over some canonical text. +fn digest_of(text: &str) -> String { + format!("{DIGEST_PREFIX}{}", hex(&sha256(text.as_bytes()))) +} + +impl SealFile { + /// The path a seal lives at. + /// + /// # Arguments + /// + /// * `layout` - the course layout. + /// * `assessment_id` - the assessment id. + /// + /// # Returns + /// + /// `seals/.yaml` under the course root. + pub fn path(layout: &Layout, assessment_id: &str) -> PathBuf { + layout + .root + .join(SEAL_DIR) + .join(format!("{assessment_id}.yaml")) + } + + /// Loads a seal. + /// + /// # Arguments + /// + /// * `path` - the YAML file. + /// + /// # Returns + /// + /// The parsed seal. + /// + /// # Errors + /// + /// Returns a load error. + pub fn load(path: &Path) -> Result { + yaml::read(path) + } + + /// Loads every seal in a directory, oldest administration first. + /// + /// # Arguments + /// + /// * `dir` - the seals directory. + /// + /// # Returns + /// + /// The seals, empty when the directory does not exist. + /// + /// # Errors + /// + /// Propagates load failures, including a seal that does not parse. + pub fn load_all(dir: &Path) -> Result> { + let mut out = Vec::new(); + for path in crate::yaml::list_yaml(dir)? { + out.push(SealFile::load(&path)?); + } + out.sort_by(|a, b| { + a.seal + .date + .cmp(&b.seal.date) + .then(a.seal.assessment.cmp(&b.seal.assessment)) + }); + Ok(out) + } + + /// Loads the seal for an assessment, if one has been written. + /// + /// Absence is not an error. A course that has never sealed anything should + /// still ingest, with the permutation derived from the record instead. + /// + /// # Arguments + /// + /// * `layout` - the course layout. + /// * `assessment_id` - the assessment id. + /// + /// # Returns + /// + /// The seal, or `None`. + /// + /// # Errors + /// + /// Returns a load error when the file exists but does not parse. + pub fn find(layout: &Layout, assessment_id: &str) -> Result> { + let path = SealFile::path(layout, assessment_id); + if path.is_file() { + Ok(Some(SealFile::load(&path)?)) + } else { + Ok(None) + } + } + + /// Writes the seal out. + /// + /// # Arguments + /// + /// * `path` - the destination. + /// + /// # Errors + /// + /// Returns [`Error::Io`] on a write failure. + pub fn save(&self, path: &Path) -> Result<()> { + yaml::write(path, self) + } + + /// One form's block. + /// + /// # Arguments + /// + /// * `id` - the form id, matched case-insensitively. + /// + /// # Returns + /// + /// The form, or `None`. + pub fn form(&self, id: &str) -> Option<&SealedForm> { + self.forms.iter().find(|f| f.id.eq_ignore_ascii_case(id)) + } + + /// One item's block, by recorded question number. + /// + /// # Arguments + /// + /// * `number` - the recorded number. + /// + /// # Returns + /// + /// The item, or `None`. + pub fn item(&self, number: u32) -> Option<&SealedItem> { + self.items.iter().find(|i| i.number == number) + } + + /// The canonical text this file's digest is computed over. + /// + /// Built field by field rather than by serializing the struct, so a change to + /// serde attributes, key order, or YAML quoting cannot invalidate every seal + /// ever written. Unit separators keep concatenation unambiguous. + /// + /// # Returns + /// + /// The digest input. + pub fn digest_input(&self) -> String { + let mut buf = String::new(); + buf.push_str(&format!( + "seal\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\n", + self.schema_version, self.seal.assessment, self.seal.course, self.seal.term + )); + for item in &self.items { + // Appended only when present: a seal written before 2.0 has to keep + // producing the input it was digested from, or every older + // administration fails verification. + if let Some(stem) = &item.stem_digest { + buf.push_str(&format!("stem\u{1f}{}\u{1f}{stem}\n", item.number)); + } + buf.push_str(&format!( + "item\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}{}\u{1f}", + item.number, + item.item, + item.version.unwrap_or(0), + item.fingerprint, + item.points, + item.bonus, + item.level.map(|l| l.code()).unwrap_or(0), + item.key.join(",") + )); + for option in &item.options { + buf.push_str(&format!( + "{}:{}:{};", + option.letter, option.correct, option.digest + )); + } + buf.push('\n'); + } + for form in &self.forms { + buf.push_str(&format!("formref\u{1f}{}\u{1f}{}\n", form.id, form.digest)); + } + buf + } + + /// Renames the items this seal froze, keeping the digest honest. + /// + /// The only sanctioned way to rewrite a seal. It recomputes the digest, + /// because the ids are part of what the digest covers, and records the + /// previous one in [`SealMeta::superseded_digests`], because a recomputed + /// digest with no trace of the recompute is a record that claims never to + /// have been touched. + /// + /// # Arguments + /// + /// * `rename` - a map from old item id to new. + /// + /// # Returns + /// + /// How many placements were renamed. Zero leaves the file alone, digest + /// included. + pub fn rename_items(&mut self, rename: &BTreeMap) -> usize { + let mut renamed = 0; + for item in &mut self.items { + if let Some(new) = rename.get(&item.item) { + item.item = new.clone(); + renamed += 1; + } + } + if renamed == 0 { + return 0; + } + let previous = std::mem::take(&mut self.seal.digest); + self.seal.superseded_digests.push(previous); + self.seal.digest = self.recompute_digest(); + renamed + } + + /// Recomputes this file's digest from its own contents. + /// + /// # Returns + /// + /// The digest string. + pub fn recompute_digest(&self) -> String { + digest_of(&self.digest_input()) + } + + /// Checks that the file has not been edited since it was written. + /// + /// This catches a hand-edit of the seal itself, which is a different failure + /// from the bank drifting and deserves a different message. + /// + /// # Returns + /// + /// Findings, empty when the file is internally consistent. + pub fn check_self(&self) -> Vec { + let mut out = Vec::new(); + let expected = self.recompute_digest(); + if expected != self.seal.digest { + out.push(Drift::new( + Severity::High, + "seal-digest", + format!( + "the seal's own digest does not match its contents ({} recorded, {} \ + computed); the file was edited after it was written", + short(&self.seal.digest), + short(&expected) + ), + )); + } + for form in &self.forms { + let expected = form_digest(form); + if expected != form.digest { + out.push(Drift::new( + Severity::High, + "seal-form-digest", + format!( + "form {}'s permutation block was edited after sealing", + form.id + ), + )); + } + } + out + } + + /// Compares the seal against the live course. + /// + /// Everything this reports is a reason to distrust an analysis that pools the + /// sealed administration with anything else, and most of it is a reason to + /// distrust per-option feedback sent to a student. + /// + /// # Arguments + /// + /// * `catalog` - the loaded course, as it stands now. + /// * `record` - the assessment record, as it stands now. + /// + /// # Returns + /// + /// Findings, worst first, empty when nothing has moved. + pub fn verify(&self, catalog: &Catalog, record: &AssessmentFile) -> Vec { + let mut out = self.check_self(); + + let rebuilt = build( + catalog, + record, + &Options { + content: self.seal.content, + forms: self.forms.iter().map(|f| f.id.clone()).collect(), + }, + ); + let rebuilt = match rebuilt { + Ok(r) => r, + Err(e) => { + out.push(Drift::new( + Severity::High, + "seal-unbuildable", + format!("the course no longer produces this assessment at all: {e}"), + )); + return out; + } + }; + + if rebuilt.seal.digest == self.seal.digest { + return out; + } + + // The digests differ, so say what differs rather than that they do. + let sealed: BTreeMap = self.items.iter().map(|i| (i.number, i)).collect(); + let live: BTreeMap = + rebuilt.items.iter().map(|i| (i.number, i)).collect(); + + for (number, was) in &sealed { + let Some(now) = live.get(number) else { + out.push(Drift::new( + Severity::High, + "item-removed", + format!( + "question {number} ({}) is no longer on the assessment record", + was.item + ), + )); + continue; + }; + if was.item != now.item { + out.push(Drift::new( + Severity::High, + "item-replaced", + format!("question {number} was {} and is now {}", was.item, now.item), + )); + continue; + } + if was.fingerprint != now.fingerprint { + out.push(Drift::new( + Severity::High, + "item-edited", + format!( + "question {number} ({}) was edited after administration; the students saw \ + a different version of the stem or options", + was.item + ), + )); + } + if was.key != now.key { + out.push(Drift::new( + Severity::High, + "key-changed", + format!( + "question {number} ({}) was keyed {} and is now keyed {}", + was.item, + was.key.join(""), + now.key.join("") + ), + )); + } + if was.options.len() != now.options.len() { + out.push(Drift::new( + Severity::High, + "option-count-changed", + format!( + "question {number} ({}) had {} options and now has {}, so every printed \ + letter on every form means something else", + was.item, + was.options.len(), + now.options.len() + ), + )); + } + if was.points != now.points { + out.push(Drift::new( + Severity::Medium, + "points-changed", + format!( + "question {number} was worth {} point(s) and is now worth {}", + was.points, now.points + ), + )); + } + if was.learning_targets != now.learning_targets { + out.push(Drift::new( + Severity::Low, + "objectives-retagged", + format!( + "question {number} ({}) was retagged; per-objective results for this \ + administration were computed against the old tags", + was.item + ), + )); + } + if was.level != now.level { + out.push(Drift::new( + Severity::Low, + "level-changed", + format!("question {number} ({}) changed level", was.item), + )); + } + } + + for number in live.keys() { + if !sealed.contains_key(number) { + out.push(Drift::new( + Severity::Medium, + "item-added", + format!("question {number} was added to the record after sealing"), + )); + } + } + + for form in &self.forms { + let Some(now) = rebuilt.form(&form.id) else { + out.push(Drift::new( + Severity::High, + "form-removed", + format!("form {} is no longer declared on the record", form.id), + )); + continue; + }; + if now.digest != form.digest { + let reason = if now.seed != form.seed { + format!( + "its seed changed from {} to {}, so every option moved", + form.seed, now.seed + ) + } else if now.shuffle_options != form.shuffle_options + || now.shuffle_items != form.shuffle_items + { + "its shuffle settings changed".to_string() + } else { + "the items behind it changed".to_string() + }; + out.push(Drift::new( + Severity::High, + "form-permutation-changed", + format!( + "form {} no longer permutes the way it did when it was printed: {reason}", + form.id + ), + )); + } + } + + out.sort_by_key(|d| std::cmp::Reverse(d.severity)); + out + } +} + +/// How a form's printed answer key is distributed. +/// +/// A shuffle is uniform per item and says nothing about the sequence it produces +/// across a form. Over thirty-odd questions it will occasionally produce a run +/// students notice: six consecutive questions keying to `A` is unremarkable +/// probabilistically and very remarkable at a desk, where it reads as a mistake +/// and makes a student who is sure of all six doubt the last three. +/// +/// Sealing is the moment to look, because it is the last moment before printing. +/// Changing a form's seed reshuffles it; nothing else has to change. +#[derive(Debug, Clone)] +pub struct Balance { + /// The form id. + pub form: String, + /// How many questions key to each printed letter. + pub counts: BTreeMap, + /// The longest run of consecutive questions sharing a printed key, as + /// `(letter, first number, last number)`. + pub longest_run: Option<(String, u32, u32)>, + /// How long that run was. Stored rather than derived from the two numbers, + /// which are recorded numbers and need not be contiguous. + pub run: usize, + /// How many questions were counted. + pub total: usize, +} + +impl Balance { + /// The length of the longest run. + pub fn run_length(&self) -> usize { + self.run + } + + /// Advisories worth printing before the form is printed. + /// + /// Deliberately quiet. Both thresholds are set where the pattern stops being + /// something only a statistician would see: a run of five, and a letter + /// carrying more than twice its share. + /// + /// # Returns + /// + /// One message per observation, empty when the form looks unremarkable. + pub fn notes(&self) -> Vec { + let mut out = Vec::new(); + if self.total == 0 { + return out; + } + + if self.run_length() >= 5 { + if let Some((letter, first, last)) = &self.longest_run { + out.push(format!( + "form {}: questions {first} through {last} all key to {letter}. That is {} in \ + a row, which students read as an error in the key. Re-seed the form if you \ + have not printed it yet", + self.form, + self.run_length() + )); + } + } + + let share = self.total as f64 / self.counts.len().max(1) as f64; + for (letter, count) in &self.counts { + if *count as f64 > share * 2.0 && *count >= 4 { + out.push(format!( + "form {}: {count} of {} questions key to {letter}", + self.form, self.total + )); + } + } + out + } +} + +/// Measures one form's printed key. +/// +/// # Arguments +/// +/// * `form` - the sealed form. +/// +/// # Returns +/// +/// The balance. Questions with no single keyed letter, such as a +/// multiple-response item, are skipped rather than guessed at. +pub fn balance(form: &SealedForm) -> Balance { + let mut counts: BTreeMap = BTreeMap::new(); + let mut sequence: Vec<(u32, String)> = Vec::new(); + + for question in &form.questions { + if question.printed_key.len() != 1 { + continue; + } + let letter = question.printed_key[0].clone(); + *counts.entry(letter.clone()).or_insert(0) += 1; + sequence.push((question.number, letter)); + } + + let mut longest: Option<(String, u32, u32)> = None; + let mut best = 0usize; + let mut run = 1usize; + for i in 1..sequence.len() { + if sequence[i].1 == sequence[i - 1].1 { + run += 1; + if run > best { + best = run; + longest = Some(( + sequence[i].1.clone(), + sequence[i + 1 - run].0, + sequence[i].0, + )); + } + } else { + run = 1; + } + } + + Balance { + form: form.id.clone(), + total: sequence.len(), + counts, + longest_run: longest, + run: best.max(if sequence.is_empty() { 0 } else { 1 }), + } +} + +/// The first eight hex characters of a digest, for messages. +/// +/// # Arguments +/// +/// * `digest` - the full digest string. +/// +/// # Returns +/// +/// A short form such as `sha256:1a2b3c4d`. +pub fn short(digest: &str) -> String { + match digest.split_once(':') { + Some((algorithm, hex)) => { + format!("{algorithm}:{}", &hex[..hex.len().min(8)]) + } + None => digest.chars().take(8).collect(), + } +} + +fn default_version() -> String { + SCHEMA_VERSION.to_string() +} +fn yes() -> bool { + true +} +fn is_false(b: &bool) -> bool { + !*b +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn printed_letters_run_past_z() { + assert_eq!(printed_letter(0), "A"); + assert_eq!(printed_letter(3), "D"); + assert_eq!(printed_letter(25), "Z"); + assert_eq!(printed_letter(26), "AA"); + } + + fn sample() -> SealFile { + let mut file = SealFile { + schema_version: "1.0".into(), + seal: SealMeta { + assessment: "e1".into(), + title: "Exam 1".into(), + course: "BIOSC 1000".into(), + term: "2026f".into(), + date: None, + sealed_on: Date::today(), + generator: Generator { + tool: "coursebank".into(), + version: "test".into(), + }, + content: true, + dropped: Vec::new(), + digest: String::new(), + superseded_digests: Vec::new(), + }, + items: vec![SealedItem { + number: 1, + item: "b::q-1".into(), + version: Some(1), + stem_digest: None, + fingerprint: "abc".into(), + points: 1.0, + bonus: false, + level: Some(Level::Remember), + learning_targets: vec!["lo-a".into()], + key: vec!["B".into()], + stem: Some("Stem.".into()), + options: vec![ + SealedOption { + letter: "A".into(), + correct: false, + credit: None, + digest: "d1".into(), + text: Some("First".into()), + }, + SealedOption { + letter: "B".into(), + correct: true, + credit: None, + digest: "d2".into(), + text: Some("Second".into()), + }, + ], + }], + forms: vec![SealedForm { + id: "A".into(), + seed: 7, + shuffle_items: false, + shuffle_options: true, + digest: String::new(), + questions: vec![SealedQuestion { + position: 1, + number: 1, + item: "b::q-1".into(), + printed_key: vec!["A".into()], + options: vec![ + LetterMap { + printed: "A".into(), + canonical: "B".into(), + }, + LetterMap { + printed: "B".into(), + canonical: "A".into(), + }, + ], + }], + }], + }; + file.forms[0].digest = form_digest(&file.forms[0]); + file.seal.digest = file.recompute_digest(); + file + } + + #[test] + fn a_fresh_seal_is_internally_consistent() { + assert!(sample().check_self().is_empty()); + } + + #[test] + fn editing_the_file_breaks_its_digest() { + let mut file = sample(); + file.items[0].key = vec!["A".into()]; + let findings = file.check_self(); + assert!( + findings.iter().any(|d| d.code == "seal-digest"), + "{findings:?}" + ); + } + + #[test] + fn editing_a_permutation_breaks_that_forms_digest() { + let mut file = sample(); + file.forms[0].questions[0].options[0].canonical = "A".into(); + let findings = file.check_self(); + assert!( + findings.iter().any(|d| d.code == "seal-form-digest"), + "{findings:?}" + ); + } + + #[test] + fn the_digest_is_order_sensitive() { + let file = sample(); + let mut swapped = file.clone(); + swapped.items[0].options.swap(0, 1); + assert_ne!(file.recompute_digest(), swapped.recompute_digest()); + } + + fn form_keyed(letters: &[&str]) -> SealedForm { + SealedForm { + id: "B".into(), + seed: 1, + shuffle_items: false, + shuffle_options: true, + digest: String::new(), + questions: letters + .iter() + .enumerate() + .map(|(i, letter)| SealedQuestion { + position: i as u32 + 1, + number: i as u32 + 1, + item: format!("b::q-{i}"), + printed_key: vec![letter.to_string()], + options: Vec::new(), + }) + .collect(), + } + } + + #[test] + fn a_long_run_of_one_letter_is_named_with_its_question_numbers() { + // Form B of a real exam did this: questions 2 through 7 all keyed to A. + let form = form_keyed(&["C", "A", "A", "A", "A", "A", "A", "C", "B"]); + let balance = balance(&form); + assert_eq!(balance.run_length(), 6); + assert_eq!( + balance.longest_run, + Some(("A".to_string(), 2, 7)), + "the run is reported by question number, not by index" + ); + let notes = balance.notes(); + assert!(notes.iter().any(|n| n.contains("2 through 7")), "{notes:?}"); + } + + #[test] + fn an_ordinary_form_says_nothing() { + let form = form_keyed(&["A", "B", "C", "D", "A", "C", "B", "D", "C", "A", "D", "B"]); + assert!(balance(&form).notes().is_empty()); + } + + #[test] + fn a_form_with_no_questions_does_not_panic() { + let balance = balance(&form_keyed(&[])); + assert_eq!(balance.run_length(), 0); + assert!(balance.notes().is_empty()); + } + + #[test] + fn short_digests_keep_the_algorithm() { + assert_eq!(short("sha256:0123456789abcdef"), "sha256:01234567"); + } + + #[test] + fn renaming_an_item_recomputes_the_digest_and_says_it_did() { + let mut seal = sample(); + seal.seal.digest = seal.recompute_digest(); + let original = seal.seal.digest.clone(); + assert!(seal.check_self().is_empty()); + + let mut rename = BTreeMap::new(); + rename.insert(seal.items[0].item.clone(), "q-renamed".to_string()); + assert_eq!(seal.rename_items(&rename), 1); + + // The digest covers the ids, so it has to move — and the file has to + // admit that it moved rather than looking untouched. + assert_eq!(seal.items[0].item, "q-renamed"); + assert_ne!(seal.seal.digest, original); + assert_eq!(seal.seal.superseded_digests, vec![original]); + assert!(seal.check_self().is_empty(), "{:?}", seal.check_self()); + + // A rename that matches nothing leaves the file entirely alone. + let steady = seal.seal.digest.clone(); + assert_eq!(seal.rename_items(&BTreeMap::new()), 0); + assert_eq!(seal.seal.digest, steady); + assert_eq!(seal.seal.superseded_digests.len(), 1); + } + + #[test] + fn round_trips_through_yaml() { + let file = sample(); + let text = serde_yaml_ng::to_string(&file).unwrap(); + let back: SealFile = serde_yaml_ng::from_str(&text).unwrap(); + assert_eq!(back.seal.digest, file.seal.digest); + assert!(back.check_self().is_empty()); + assert_eq!(back.form("a").map(|f| f.seed), Some(7)); + } +} diff --git a/src/model/taxonomy.rs b/src/model/taxonomy.rs index 8b717b2..51f52cd 100644 --- a/src/model/taxonomy.rs +++ b/src/model/taxonomy.rs @@ -9,6 +9,44 @@ use std::fmt; use serde::{Deserialize, Serialize}; +/// Which tier of the objective registry an entry or a reported row belongs to. +/// +/// The two words come from the assessment literature and name two different +/// jobs, not two sizes of the same thing. A **learning objective** is what a +/// syllabus lists and what a report classifies as met: the unit a claim is made +/// about. A **learning target** is the specific performance an item is written +/// against and tagged to: what a student aims at in one class, and the evidence +/// a claim about an objective rests on. +/// +/// It lives here rather than beside the registry because every layer that +/// reports needs it, and a `bool` named for one of the two tiers leaves the +/// reader guessing which way round it points. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum Tier { + /// A learning objective: the tier a mastery claim is made about. + Objective, + /// A learning target: the tier items are tagged to, and evidence for the + /// objective above it. + Target, +} + +impl Tier { + /// The snake_case token used in YAML and in flat storage. + pub fn as_str(self) -> &'static str { + match self { + Tier::Objective => "objective", + Tier::Target => "target", + } + } +} + +impl fmt::Display for Tier { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str(self.as_str()) + } +} + /// Cognitive demand, following the revised Bloom taxonomy. /// /// Serialized as the integers 1 through 5 so YAML reads `level: 3`. The derived @@ -423,17 +461,53 @@ pub enum Format { MultipleResponse, /// Two options, True and False. TrueFalse, + /// A free-text answer the student writes rather than selects. + /// + /// It carries no options and is not machine-scored. What a grader marks it + /// against, and what a solutions document shows, lives in the item's + /// [`crate::item::Solution`]: a model answer and, when the item is worth more + /// than a point, a rubric. This is the format for "explain", "derive", and + /// "estimate" prompts that a set of distractors would trivialize. + OpenResponse, } impl Format { + /// Every response format. + pub const ALL: [Format; 4] = [ + Format::SingleBestAnswer, + Format::MultipleResponse, + Format::TrueFalse, + Format::OpenResponse, + ]; + /// The QTI question type Canvas expects for this format. pub fn qti_type(self) -> &'static str { match self { Format::SingleBestAnswer => "multiple_choice_question", Format::MultipleResponse => "multiple_answers_question", Format::TrueFalse => "true_false_question", + Format::OpenResponse => "essay_question", } } + + /// The snake_case token used in YAML. + pub fn as_str(self) -> &'static str { + match self { + Format::SingleBestAnswer => "single_best_answer", + Format::MultipleResponse => "multiple_response", + Format::TrueFalse => "true_false", + Format::OpenResponse => "open_response", + } + } + + /// Whether items in this format present selectable options. + /// + /// `false` only for [`Format::OpenResponse`]. Validation, assembly, and the + /// exporters branch on this rather than on the variant, so the day a second + /// free-text format is added it inherits the no-options handling for free. + pub fn has_options(self) -> bool { + !matches!(self, Format::OpenResponse) + } } /// How strongly an item is expected to separate strong from weak students. @@ -633,4 +707,18 @@ mod tests { Status::InReview ); } + + #[test] + fn open_response_is_the_only_format_without_options() { + assert_eq!( + serde_json::from_str::("\"open_response\"").unwrap(), + Format::OpenResponse + ); + assert_eq!(Format::OpenResponse.as_str(), "open_response"); + assert_eq!(Format::OpenResponse.qti_type(), "essay_question"); + assert!(!Format::OpenResponse.has_options()); + for f in Format::ALL { + assert_eq!(f.has_options(), f != Format::OpenResponse); + } + } } diff --git a/src/util.rs b/src/util.rs index 83a30bb..2bdca5c 100644 --- a/src/util.rs +++ b/src/util.rs @@ -9,6 +9,7 @@ //! replaces a dependency that would otherwise need to keep working for as long as //! a course repository needs to stay readable. +pub mod citation; pub mod date; pub mod hash; pub mod markup; diff --git a/src/util/citation.rs b/src/util/citation.rs new file mode 100644 index 0000000..c4f5af2 --- /dev/null +++ b/src/util/citation.rs @@ -0,0 +1,330 @@ +// SPDX-License-Identifier: Prosperity-3.0.0 +// Copyright Scientific Computing Studio +// Source: https://git.scient.ing/education/coursebank + +//! Pulling a citation apart when it was written as prose. +//! +//! A bibliography assembled by hand tends to collect entries like +//! +//! ```text +//! note: 'Nucleic Acids Res 25:3389-3402. doi:10.1093/nar/25.17.3389' +//! ``` +//! +//! which is a complete citation in a field that means "anything else worth +//! saying". Nothing can use it: a reading list cannot link the DOI, an export to +//! Hayagriva or BibTeX has no journal to put in `parent` or `journal`, and the +//! `container`, `volume`, `pages`, and `doi` fields sit empty beside it. +//! +//! [`parse`] takes such a note apart. What it cannot account for it leaves in +//! the note, which is the important half of the contract: a note reading +//! `'Bioinformatics 18:440-445. Origin of spaced seeds.'` yields the journal, +//! the volume, the pages, and a note that still says where spaced seeds came +//! from. Nothing is discarded and nothing is invented — an issue number that was +//! never written down stays absent, even when a publisher's DOI happens to +//! encode one. + +/// The parts of a citation recovered from a note. +#[derive(Debug, Clone, Default, PartialEq, Eq)] +pub struct Parsed { + /// The journal, proceedings, or book the work appeared in. + pub container: Option, + /// The volume. + pub volume: Option, + /// The page range, as `first-last`. + pub pages: Option, + /// The DOI, bare. + pub doi: Option, + /// What the note still says after the citation is removed. + pub note: Option, +} + +/// Takes a citation apart, leaving the rest of the note alone. +/// +/// # Arguments +/// +/// * `note` - the note as written. +/// * `year` - the record's year, which is how a trailing year is recognized as +/// part of a conference name rather than part of the title of the venue. +/// +/// # Returns +/// +/// The parts found. Every field is independently optional: a note that carries +/// only a DOI yields only a DOI. +pub fn parse(note: &str, year: Option) -> Parsed { + let mut out = Parsed::default(); + let mut rest = note.trim().to_string(); + + if let Some((container, volume, pages, tail)) = citation(&rest) { + out.container = Some(container); + out.volume = Some(volume); + out.pages = Some(pages); + rest = tail; + } + + if let Some((doi, tail)) = doi(&rest) { + out.doi = Some(doi); + rest = tail; + } + + // A venue with no volume or pages — a conference, usually — is named by the + // clause that ends in the year the work was published. + if out.container.is_none() { + if let Some(y) = year { + if let Some((container, tail)) = venue(&rest, y) { + out.container = Some(container); + rest = tail; + } + } + } + + // The year belongs to the record, not to the name of the venue. + if let (Some(container), Some(y)) = (&out.container, year) { + let suffix = format!(" {y}"); + if let Some(trimmed) = container.strip_suffix(&suffix) { + out.container = Some(trimmed.trim_end().to_string()); + } + } + + let rest = rest.trim().trim_start_matches('.').trim().to_string(); + out.note = (!rest.is_empty()).then_some(rest); + out +} + +/// Finds `Journal 25:3389-3402` or `Journal 48, 443-453` at the start. +/// +/// The volume is the first digit run that follows a space and is followed by a +/// separator and a page range. Requiring the whole shape is what keeps a year in +/// a conference name (`Proc. FOCS 2000.`) from being read as a volume. +/// +/// # Returns +/// +/// The container, volume, page range, and whatever followed. +fn citation(text: &str) -> Option<(String, String, String, String)> { + let bytes = text.as_bytes(); + let mut at = 0; + + while at < bytes.len() { + // A volume follows a space, so that a digit inside a name is not one. + if !(bytes[at].is_ascii_digit() && at > 0 && bytes[at - 1] == b' ') { + at += 1; + continue; + } + let volume_start = at; + let volume_end = digits(bytes, volume_start); + let mut cursor = spaces(bytes, volume_end); + + // The separator between volume and pages is a colon or a comma. + if cursor < bytes.len() && (bytes[cursor] == b':' || bytes[cursor] == b',') { + cursor = spaces(bytes, cursor + 1); + let first_start = cursor; + let first_end = digits(bytes, first_start); + if first_end > first_start { + let dash = text[first_end..] + .strip_prefix('-') + .or_else(|| text[first_end..].strip_prefix('\u{2013}')); + if let Some(after_dash) = dash { + let last_offset = text.len() - after_dash.len(); + let last_end = digits(bytes, last_offset); + if last_end > last_offset { + let container = text[..volume_start].trim_end_matches([' ', ',']); + if !container.is_empty() { + return Some(( + container.to_string(), + text[volume_start..volume_end].to_string(), + format!( + "{}-{}", + &text[first_start..first_end], + &text[last_offset..last_end] + ), + text[last_end..].to_string(), + )); + } + } + } + } + } + at = volume_end; + } + None +} + +/// Finds a `doi:10.…` anywhere in the text. +/// +/// # Returns +/// +/// The DOI and the text with it removed. +fn doi(text: &str) -> Option<(String, String)> { + let lower = text.to_ascii_lowercase(); + let at = lower.find("doi:")?; + let after = text[at + 4..].trim_start(); + let offset = text.len() - after.len(); + let end = after + .find(char::is_whitespace) + .map(|n| offset + n) + .unwrap_or(text.len()); + + let doi = text[offset..end].trim_end_matches('.'); + if !doi.starts_with("10.") { + return None; + } + let mut remainder = String::from(text[..at].trim_end()); + let tail = text[end..].trim(); + if !tail.is_empty() { + if !remainder.is_empty() { + remainder.push(' '); + } + remainder.push_str(tail); + } + Some((doi.to_string(), remainder)) +} + +/// Finds a leading clause ending in the publication year: `Proc. FOCS 2000.` +/// +/// # Returns +/// +/// The clause without its trailing period, and whatever followed. +fn venue(text: &str, year: u32) -> Option<(String, String)> { + let needle = format!("{year}."); + let at = text.find(&needle)?; + let clause = text[..at + needle.len() - 1].trim(); + if clause.is_empty() { + return None; + } + Some(( + clause.to_string(), + text[at + needle.len()..].trim().to_string(), + )) +} + +/// The end of a run of ASCII digits starting at `from`. +fn digits(bytes: &[u8], from: usize) -> usize { + let mut at = from; + while at < bytes.len() && bytes[at].is_ascii_digit() { + at += 1; + } + at +} + +/// The end of a run of spaces starting at `from`. +fn spaces(bytes: &[u8], from: usize) -> usize { + let mut at = from; + while at < bytes.len() && bytes[at] == b' ' { + at += 1; + } + at +} + +/// Whether a note still looks like it is carrying a citation. +/// +/// Used by validation to say so, rather than leaving a note that a reading list +/// cannot link and an export cannot use. +/// +/// # Arguments +/// +/// * `note` - the note as written. +/// +/// # Returns +/// +/// `true` when a volume and page range, or a DOI, can be found in it. +pub fn looks_like_a_citation(note: &str) -> bool { + citation(note.trim()).is_some() || doi(note.trim()).is_some() +} + +#[cfg(test)] +mod tests { + use super::*; + + /// Every article note in a real course bibliography, which is where the + /// shapes below come from. Two separator styles, DOIs in three positions, + /// a conference with no volume, and prose that has to survive. + #[test] + fn a_journal_citation_comes_apart() { + let p = parse( + "Nucleic Acids Res 25:3389-3402. doi:10.1093/nar/25.17.3389", + Some(1997), + ); + assert_eq!(p.container.as_deref(), Some("Nucleic Acids Res")); + assert_eq!(p.volume.as_deref(), Some("25")); + assert_eq!(p.pages.as_deref(), Some("3389-3402")); + assert_eq!(p.doi.as_deref(), Some("10.1093/nar/25.17.3389")); + assert_eq!(p.note, None); + // The DOI encodes volume 25, issue 17. Nothing infers the issue from + // it: a field nobody wrote down stays empty. + } + + #[test] + fn an_abbreviation_keeps_its_final_period() { + let p = parse("J. Mol. Biol. 48, 443-453.", Some(1970)); + assert_eq!(p.container.as_deref(), Some("J. Mol. Biol.")); + assert_eq!(p.volume.as_deref(), Some("48")); + assert_eq!(p.pages.as_deref(), Some("443-453")); + assert_eq!(p.note, None); + } + + #[test] + fn prose_after_a_citation_stays_in_the_note() { + let p = parse( + "Bioinformatics 18:440-445. Origin of spaced seeds.", + Some(2002), + ); + assert_eq!(p.container.as_deref(), Some("Bioinformatics")); + assert_eq!(p.note.as_deref(), Some("Origin of spaced seeds.")); + + // A caveat the author wrote is the last thing to throw away. + let p = parse("J Mol Biol 215:403-410. Verify before use.", Some(1990)); + assert_eq!(p.note.as_deref(), Some("Verify before use.")); + + let p = parse( + "Bioinformatics 25:2078-2079. doi:10.1093/bioinformatics/btp352. Author list is the \ + core set plus the 1000 Genomes Data Processing Subgroup; verify.", + Some(2009), + ); + assert_eq!(p.doi.as_deref(), Some("10.1093/bioinformatics/btp352")); + assert_eq!( + p.note.as_deref(), + Some( + "Author list is the core set plus the 1000 Genomes Data Processing Subgroup; \ + verify." + ) + ); + } + + #[test] + fn a_conference_has_a_year_where_a_volume_would_be() { + let p = parse( + "Proc. FOCS 2000. doi:10.1109/SFCS.2000.892127. The FM-index. Theory background.", + Some(2000), + ); + assert_eq!(p.container.as_deref(), Some("Proc. FOCS")); + // No volume and no pages were written, so none are invented — and the + // 2000 in the DOI is not mistaken for either. + assert_eq!(p.volume, None); + assert_eq!(p.pages, None); + assert_eq!(p.doi.as_deref(), Some("10.1109/SFCS.2000.892127")); + assert_eq!(p.note.as_deref(), Some("The FM-index. Theory background.")); + } + + #[test] + fn a_note_with_nothing_to_find_is_left_whole() { + let p = parse("Origin of the MAPQ score.", Some(2008)); + assert_eq!(p.container, None); + assert_eq!(p.note.as_deref(), Some("Origin of the MAPQ score.")); + } + + #[test] + fn a_multi_word_journal_is_not_cut_at_a_number() { + let p = parse("Advances in Mathematics 20, 367-387.", Some(1976)); + assert_eq!(p.container.as_deref(), Some("Advances in Mathematics")); + assert_eq!(p.volume.as_deref(), Some("20")); + } + + #[test] + fn what_validation_looks_for() { + assert!(looks_like_a_citation("Nat Methods 12:59-60.")); + assert!(looks_like_a_citation("doi:10.1038/nmeth.3176")); + assert!(!looks_like_a_citation("Origin of minimizers.")); + assert!(!looks_like_a_citation("Verify before use.")); + // A page range with no volume is not a citation shape. + assert!(!looks_like_a_citation("see pages 12-14")); + } +} diff --git a/src/util/markup.rs b/src/util/markup.rs index 6588c38..d0d6b1d 100644 --- a/src/util/markup.rs +++ b/src/util/markup.rs @@ -113,10 +113,53 @@ pub fn to_plain(src: &str) -> String { out.trim().to_string() } -/// Passes authoring markup through for Typst. +/// Converts authoring markup to Pandoc-flavoured Markdown, for a Quarto document. /// -/// The markup is already a Typst subset, so this only normalizes whitespace and -/// escapes the few characters Typst treats specially in content mode. +/// Subscripts and superscripts become Pandoc's `~x~` and `^x^`, the symbol table +/// renders as Unicode, and the bold, italic, and inline-code spans are already +/// Markdown, so they pass through unchanged. Paragraph breaks are kept. Nothing is +/// HTML-escaped, because the consumer is a Markdown renderer rather than a page. +/// +/// # Arguments +/// +/// * `src` - the authoring source. +/// +/// # Returns +/// +/// Pandoc Markdown, trimmed, with paragraph breaks preserved. +pub fn to_markdown(src: &str) -> String { + let symbolized = apply_symbols(src, false); + let mut out = wrap_bracket(&symbolized, "#sub[", "~", "~"); + out = wrap_bracket(&out, "#sup[", "^", "^"); + out.lines() + .map(|l| l.trim_end()) + .collect::>() + .join("\n") + .trim() + .to_string() +} + +/// Passes authoring markup through for Typst, translating inline LaTeX math on +/// the way. +/// +/// Outside math this only escapes the characters Typst treats specially in +/// content mode: a bare `@` or `<` starts a reference or label. +/// +/// Math is different. Authors write ordinary LaTeX between `$...$`, and Typst's +/// own math grammar is not LaTeX's — a backslash escapes the next character +/// rather than naming a symbol, so `\Delta`, `\times`, `\ln` compile without +/// error and print wrong. Each `$...$` span is instead handed whole to +/// mitex's `mi`, which parses LaTeX grammar on purpose: `$\phi$` becomes +/// `#mi("\\phi")`. mhchem's `\ce{...}` is the one exception, and goes to +/// whalogen's `ce` instead, via `push_math_span`. The `@`/`<`/`>` escaping +/// above is skipped for anything +/// inside the span, since it reaches Typst as a string argument, not as +/// markup — escaping `<` there would corrupt the LaTeX rather than protect +/// anything. +/// +/// `\$` is left alone, matching LaTeX's own convention for a literal dollar +/// sign. A `$` with no matching close is escaped the same way rather than left +/// to open Typst's own math mode on a stray price or a malformed source line. /// /// # Arguments /// @@ -126,17 +169,286 @@ pub fn to_plain(src: &str) -> String { /// /// Typst content-mode markup. pub fn to_typst(src: &str) -> String { + let src = src.trim(); + let chars: Vec<(usize, char)> = src.char_indices().collect(); let mut out = String::with_capacity(src.len()); - for ch in src.trim().chars() { - match ch { - // A bare `@` or `<` starts a Typst reference or label. + let mut i = 0; + while i < chars.len() { + let (_, c) = chars[i]; + + // A `\ce{...}` written outside math is still chemistry and still cannot + // reach Typst as a backslash: `\c` is an escape in content mode. Checked + // before the escaped-pair rule below, which would otherwise copy `\c` + // through and leave `e{...}` as literal text. The cheap two-character + // guard keeps a stem full of `\Delta` and `\times` from rescanning the + // remainder at every backslash. + if c == '\\' + && chars.get(i + 1).map(|c| c.1) == Some('c') + && chars.get(i + 2).map(|c| c.1) == Some('e') + { + if let Some(span) = find_ce(&src[chars[i].0..]) { + if span.start == 0 { + let at = chars[i].0; + push_ce(&mut out, &src[at + span.arg.0..at + span.arg.1]); + i = char_index(&chars, at + span.end); + continue; + } + } + } + + // An escaped pair is copied verbatim and never reconsidered, so `\$` + // can't be mistaken for the start of math and an `\@`/`\<`/`\>` an + // author already wrote is not escaped a second time. + if c == '\\' && i + 1 < chars.len() { + out.push('\\'); + out.push(chars[i + 1].1); + i += 2; + continue; + } + + if c == '$' { + match find_math_close(&chars, i) { + Some(close) => { + let start = chars[i + 1].0; + let end = chars[close].0; + push_math_span(&mut out, &src[start..end]); + i = close + 1; + continue; + } + None => { + out.push_str("\\$"); + i += 1; + continue; + } + } + } + + match c { '@' => out.push_str("\\@"), '<' => out.push_str("\\<"), '>' => out.push_str("\\>"), + _ => out.push(c), + } + i += 1; + } + out +} + +/// Finds the index into `chars` of the `$` matching the opener at `open`. A +/// backslash-escaped pair is skipped as a unit, so a `\$` inside the math span +/// doesn't close it early. +fn find_math_close(chars: &[(usize, char)], open: usize) -> Option { + let mut j = open + 1; + while j < chars.len() { + match chars[j].1 { + '\\' if j + 1 < chars.len() => j += 2, + '$' => return Some(j), + _ => j += 1, + } + } + None +} + +/// The import line a template needs before it can receive a `#ce(...)` call. +/// +/// mitex renders the LaTeX in a stem, but it ships no package support, so +/// mhchem's `\ce` is simply an unknown command to it. whalogen is a Typst port +/// of mhchem, and `ce` is the function this module emits chemistry as. +pub const CHEM_IMPORT: &str = "#import \"@preview/whalogen:0.3.0\": ce"; + +/// Whether `body` needs [`CHEM_IMPORT`] and `template` does not provide it. +/// +/// A template is checked for the package name rather than for the exact import +/// line, so a template that pins a different whalogen version, imports the +/// package under an alias, or defines its own `ce` is left alone. +/// +/// # Arguments +/// +/// * `template` - the template source the body will be injected into. +/// * `body` - the generated Typst source. +/// +/// # Returns +/// +/// Whether the pair would fail to compile for want of the import. +pub fn needs_chem_import(template: &str, body: &str) -> bool { + body.contains("#ce(") && !template.contains("whalogen") && !template.contains("let ce") +} + +/// A `\ce{...}` command located in a source fragment, as byte offsets from the +/// start of that fragment. +struct CeSpan { + /// Offset of the backslash. + start: usize, + /// Offset just past the closing brace. + end: usize, + /// The argument, without its braces. + arg: (usize, usize), +} + +/// Finds the first `\ce{...}` in `s`. +/// +/// Three details matter, and all three come from the same place: the scan has to +/// agree with what LaTeX itself would read. +/// +/// A backslash-escaped pair is skipped as a unit, so the line break `\\` +/// followed by the letters `ce` is not read as the command. `$a \\ ce{x}$` is a +/// break and then literal text; mhchem's own command is a single backslash. +/// +/// The name must end at the `e`, so `\cellcolor` and `\century` are left for +/// mitex rather than half-consumed here. +/// +/// The argument is matched on brace depth rather than on the first `}`, so +/// `\ce{Fe^{2+}}` keeps its superscript. An unbalanced argument yields `None`: +/// there is nothing safe to convert, and mitex's own error names the line. +fn find_ce(s: &str) -> Option { + let chars: Vec<(usize, char)> = s.char_indices().collect(); + let mut i = 0; + while i < chars.len() { + if chars[i].1 != '\\' { + i += 1; + continue; + } + if matches!(chars.get(i + 1), Some((_, '\\'))) { + i += 2; + continue; + } + let named = chars.get(i + 1).map(|c| c.1) == Some('c') + && chars.get(i + 2).map(|c| c.1) == Some('e') + && !matches!(chars.get(i + 3), Some((_, c)) if c.is_ascii_alphabetic()); + if !named { + i += 1; + continue; + } + + // LaTeX allows whitespace between a control word and its argument. + let mut open = i + 3; + while matches!(chars.get(open), Some((_, c)) if c.is_whitespace()) { + open += 1; + } + if !matches!(chars.get(open), Some((_, '{'))) { + i += 1; + continue; + } + + let mut depth = 1usize; + let mut k = open + 1; + while k < chars.len() { + match chars[k].1 { + '\\' => k += 2, + '{' => { + depth += 1; + k += 1; + } + '}' => { + depth -= 1; + if depth == 0 { + return Some(CeSpan { + start: chars[i].0, + end: chars[k].0 + 1, + arg: (chars[open].0 + 1, chars[k].0), + }); + } + k += 1; + } + _ => k += 1, + } + } + return None; + } + None +} + +/// The index into `chars` of the entry at byte offset `byte`, or the length when +/// the offset is past the end. +fn char_index(chars: &[(usize, char)], byte: usize) -> usize { + chars + .iter() + .position(|(at, _)| *at >= byte) + .unwrap_or(chars.len()) +} + +/// Emits one `$...$` span as Typst content. +/// +/// Most of a span goes to mitex's `mi`, which parses LaTeX on purpose. The +/// exception is mhchem: `\ce{...}` is not a LaTeX primitive but a package +/// command with its own character-level grammar, and mitex implements no +/// packages, so the plugin aborts the whole document with `unknown command: +/// \ce` rather than degrading to something printable. Those runs are handed to +/// whalogen's `ce` instead and the rest of the span still goes to `mi`, so an +/// equation that mixes chemistry with ordinary math renders both. +/// +/// A span with no chemistry in it produces exactly one `mi` call, as it always +/// has. +/// +/// # Arguments +/// +/// * `out` - the buffer to append to. +/// * `latex` - the LaTeX between the dollar signs. +fn push_math_span(out: &mut String, latex: &str) { + if find_ce(latex).is_none() { + push_mi(out, latex); + return; + } + + let mut rest = latex; + while let Some(span) = find_ce(rest) { + push_run(out, &rest[..span.start]); + push_ce(out, &rest[span.arg.0..span.arg.1]); + rest = &rest[span.end..]; + } + push_run(out, rest); +} + +/// Emits a non-chemistry run of a math span, dropping an empty one and keeping a +/// whitespace-only one as the single space it separates two calls with. +fn push_run(out: &mut String, latex: &str) { + if latex.is_empty() { + return; + } + if latex.trim().is_empty() { + out.push(' '); + return; + } + push_mi(out, latex); +} + +/// Emits a `#mi(...)` call wrapping LaTeX math. +fn push_mi(out: &mut String, latex: &str) { + out.push_str("#mi("); + push_typst_string(out, latex); + out.push(')'); +} + +/// Emits a `#ce(...)` call wrapping an mhchem argument. +/// +/// The argument is passed through as written. whalogen reads the same formula, +/// charge, bond, and arrow syntax mhchem does, so `H2O`, `<=>`, `[AgCl2]-`, and +/// `->[H2O]` need no translation. Its isotope and oxidation-number spellings do +/// differ (`@Th,227,90@` against mhchem's `^{227}_{90}Th`), and so do `\pu` and +/// `\bond`, which whalogen has no equivalent for. Those print oddly rather than +/// failing the build, so a bank that uses them needs its own pass. +fn push_ce(out: &mut String, argument: &str) { + out.push_str("#ce("); + push_typst_string(out, argument.trim()); + out.push(')'); +} + +/// Writes `s` as a quoted Typst string. Kept local, duplicating the five-case +/// match in `typst::value::write_string`, rather than reaching into the +/// Typst-specific value writer for one small helper. +fn push_typst_string(out: &mut String, s: &str) { + out.push('"'); + for ch in s.chars() { + match ch { + '"' => out.push_str("\\\""), + '\\' => out.push_str("\\\\"), + '\n' => out.push_str("\\n"), + '\r' => out.push_str("\\r"), + '\t' => out.push_str("\\t"), _ => out.push(ch), } } - out + out.push('"'); } /// Escapes the five XML-significant characters. @@ -173,7 +485,7 @@ pub fn escape_html(s: &str) -> String { /// # Returns /// /// The substituted text. -fn apply_symbols(s: &str, html: bool) -> String { +pub(crate) fn apply_symbols(s: &str, html: bool) -> String { let mut out = s.to_string(); for (token, entity, plain) in SYMBOLS { if out.contains(token) { @@ -192,7 +504,7 @@ fn apply_symbols(s: &str, html: bool) -> String { /// # Returns /// /// The text with inline markup converted. -fn apply_inline(s: &str) -> String { +pub(crate) fn apply_inline(s: &str) -> String { let mut out = s.to_string(); // Bracketed forms first: their contents may contain other markup characters. out = wrap_bracket(&out, "#sub[", "", ""); @@ -379,4 +691,138 @@ mod tests { assert_eq!(to_typst("a @ b"), "a \\@ b"); assert_eq!(to_typst("x < y"), "x \\< y"); } + + #[test] + fn markdown_uses_pandoc_scripts_and_unicode_symbols() { + assert_eq!(to_markdown("H#sub[2]O"), "H~2~O"); + assert_eq!(to_markdown("x#sup[2]"), "x^2^"); + assert_eq!( + to_markdown("K#sub[m] #sym.approx 5 mM"), + "K~m~ \u{2248} 5 mM" + ); + // Bold, italic, and code are already Markdown. + assert_eq!( + to_markdown("**bold** and *em* and `code`"), + "**bold** and *em* and `code`" + ); + // Paragraph breaks survive. + assert_eq!(to_markdown("one\n\ntwo"), "one\n\ntwo"); + } +} + +#[test] +fn latex_math_becomes_a_mitex_call() { + assert_eq!( + to_typst("angle $\\phi$ (phi)"), + "angle #mi(\"\\\\phi\") (phi)" + ); +} + +#[test] +fn comparison_operators_inside_math_are_not_escaped() { + assert_eq!(to_typst("$\\Delta H < 0$"), "#mi(\"\\\\Delta H < 0\")"); +} + +#[test] +fn reference_starters_outside_math_are_still_escaped() { + assert_eq!(to_typst("see @fig:x and x < y"), "see \\@fig:x and x \\< y"); +} + +#[test] +fn escaped_and_unmatched_dollar_signs_are_left_or_escaped() { + assert_eq!(to_typst("costs \\$5 total"), "costs \\$5 total"); + assert_eq!(to_typst("just $5"), "just \\$5"); +} + +#[test] +fn mhchem_goes_to_whalogen_rather_than_mitex() { + // The regression: mitex implements no LaTeX packages, so `\ce` reached the + // plugin as an unknown command and failed the whole document rather than + // printing badly. + assert_eq!(to_typst("$\\ce{H2O}$"), "#ce(\"H2O\")"); + assert_eq!( + to_typst("water ionizes, $\\ce{H2O <=> H+ + OH-}$, giving"), + "water ionizes, #ce(\"H2O <=> H+ + OH-\"), giving" + ); +} + +#[test] +fn a_span_can_mix_chemistry_with_ordinary_math() { + // The chemistry leaves the span and the rest of it still reaches mitex, so + // an equation that needs both renders both. + assert_eq!(to_typst("$K_w = \\ce{H2O}$"), "#mi(\"K_w = \")#ce(\"H2O\")"); + // The space between the two runs stays inside the `mi` call rather than + // becoming content-mode whitespace, so the spacing is TeX's to decide. + assert_eq!( + to_typst("$\\ce{H2O} \\to \\Delta H$"), + "#ce(\"H2O\")#mi(\" \\\\to \\\\Delta H\")" + ); +} + +#[test] +fn mhchem_arguments_keep_their_nested_braces() { + // Matched on brace depth, not on the first `}`, or the charge is orphaned + // and the remaining `}` closes the `#ce(` call early. + assert_eq!(to_typst("$\\ce{Fe^{2+}}$"), "#ce(\"Fe^{2+}\")"); + assert_eq!( + to_typst("$\\ce{SO4^{2-} + Ba^{2+}}$"), + "#ce(\"SO4^{2-} + Ba^{2+}\")" + ); +} + +#[test] +fn a_latex_line_break_is_not_read_as_the_chemistry_command() { + // `\\` is an escaped backslash followed by the letters `ce`, not `\ce`. + // Scanning that skips escaped pairs as a unit keeps the two apart; scanning + // that does not would convert a line break into a formula. + assert_eq!(to_typst("$a \\\\ce{x}$"), "#mi(\"a \\\\\\\\ce{x}\")"); +} + +#[test] +fn commands_that_merely_start_with_ce_are_left_to_mitex() { + // The name has to end at the `e`, or `\cellcolor` is half-consumed and the + // conversion invents a formula out of its argument. + assert_eq!( + to_typst("$\\cellcolor{red} x$"), + "#mi(\"\\\\cellcolor{red} x\")" + ); +} + +#[test] +fn unbalanced_chemistry_is_left_for_mitex_to_report() { + // Nothing safe to convert. mitex's own error names the file and line, which + // is more useful than a silently truncated formula. + assert_eq!(to_typst("$\\ce{H2O$"), "#mi(\"\\\\ce{H2O\")"); +} + +#[test] +fn chemistry_outside_math_is_converted_too() { + // A bare `\ce` never reaches mitex at all, and `\c` is an escape in Typst + // content mode, so leaving it alone produces a document that either fails + // or prints the letters. + assert_eq!( + to_typst("the backbone \\ce{-NH} group"), + "the backbone #ce(\"-NH\") group" + ); +} + +#[test] +fn a_template_without_the_chemistry_import_is_named() { + let body = "#question((stem: [#ce(\"H2O\")]))"; + assert!(needs_chem_import( + "#import \"@preview/mitex:0.2.7\": mi", + body + )); + // An import of any whalogen version, or a template's own `ce`, is enough. + assert!(!needs_chem_import(CHEM_IMPORT, body)); + assert!(!needs_chem_import( + "#import \"@preview/whalogen:0.2.0\": ce as ce", + body + )); + assert!(!needs_chem_import("#let ce(f) = f", body)); + // No chemistry in the body, nothing to warn about. + assert!(!needs_chem_import( + "#import \"@preview/mitex:0.2.7\": mi", + "#mi(\"x\")" + )); } diff --git a/src/util/yaml.rs b/src/util/yaml.rs index 4e801ac..da97e63 100644 --- a/src/util/yaml.rs +++ b/src/util/yaml.rs @@ -16,7 +16,7 @@ use std::fs; use std::path::Path; use serde::de::{self, DeserializeOwned, Visitor}; -use serde::{Deserializer, Serialize}; +use serde::{Deserialize, Deserializer, Serialize}; use crate::error::{Error, Result}; @@ -61,6 +61,26 @@ pub fn write(path: &Path, value: &T) -> Result<()> { fs::write(path, text).map_err(|e| Error::io(path, e)) } +/// Serializes a value to a YAML string. +/// +/// Used where the caller needs to put something in front of the document, such +/// as the banner on a generated file. +/// +/// # Arguments +/// +/// * `value` - the value to serialize. +/// +/// # Returns +/// +/// The YAML text. +/// +/// # Errors +/// +/// Returns [`Error::Other`] if the value cannot be represented as YAML. +pub fn to_string(value: &T) -> Result { + serde_yaml_ng::to_string(value).map_err(Error::other) +} + /// Deserializes a JSON file into any type. /// /// Used only for importing legacy banks and for reading emitted schemas back in @@ -202,6 +222,33 @@ where d.deserialize_any(V) } +/// Deserializes an optional scalar as a string, quoted or not. +/// +/// The [`flexible_string`] of a field that may be absent, which is what a +/// fragment's `schema_version` is: one file in a course declares it and the +/// rest inherit. +/// +/// # Arguments +/// +/// * `d` - the deserializer. +/// +/// # Returns +/// +/// The value as a string, or `None`. +/// +/// # Errors +/// +/// Returns a deserialization error for non-scalar input. +pub fn flexible_string_opt<'de, D>(d: D) -> std::result::Result, D::Error> +where + D: Deserializer<'de>, +{ + #[derive(serde::Deserialize)] + struct Wrapper(#[serde(deserialize_with = "flexible_string")] String); + + Ok(Option::::deserialize(d)?.map(|w| w.0)) +} + #[cfg(test)] mod tests { use super::*;