// SPDX-License-Identifier: Prosperity-3.0.0 // Copyright Scientific Computing Studio // Source: https://git.scient.ing/education/coursebank //! One course, several files. //! //! A course of forty lectures does not fit in a file anyone wants to scroll. So //! the registries [`CourseFile`] holds may be spread across a directory and //! merged on load: //! //! ```text //! course.yaml course, policy, units //! references.yaml references //! lectures/l-1-2.yaml one lecture, its readings, and what it teaches //! objectives/lo-x.yaml one objective and its targets //! ``` //! //! The merge happens in memory on every command. Nothing is generated on disk //! and no command depends on a build step, because a generated file that other //! commands read is a file that goes stale. `coursebank course build` exists to //! show you the merged result, and nothing reads what it writes. //! //! # What this does not relax //! //! Splitting a file is only worth doing if it cannot introduce a second //! definition of the same thing. Two rules keep that true, and both are enforced //! here rather than left to convention: //! //! * **One definition site per id.** Two files defining `lo-read-file-formats` //! is an error naming both paths. The winner is not the last file loaded, //! because there is no winner. //! * **A section belongs to a kind of file.** A file under `lectures/` may not //! define `learning_objectives`. Otherwise the layout decays into forty files //! that each might hold anything, which is the same navigation problem in a //! worse shape. //! //! `course.yaml` is exempt from the second rule: a course that has not been //! split is a single fragment that happens to define everything, and it keeps //! loading unchanged. //! //! # Two derivations //! //! Splitting by lecture makes two fields tedious to maintain by hand, so they //! are derived instead: //! //! * A lecture's `teaches:` list adds that lecture to each named objective's //! `lectures`. You write what a lecture covers while planning the lecture, //! which is when you know. //! * A target with no `lectures` of its own inherits its objective's, the same //! way it already inherits `level_ceiling`. //! //! Both are unions and both are idempotent, so declaring a pair on both sides //! is redundant rather than contradictory. use std::collections::BTreeMap; use std::fmt; use std::path::{Path, PathBuf}; use serde::{Deserialize, Serialize}; use super::{ COURSE_FILE, Course, CourseFile, Lecture, Objective, Policy, Reference, SCHEMA_VERSION, SUPPORTED_MAJORS, Stimulus, Target, Unit, }; use crate::error::{Error, Result}; use crate::layout::Layout; use crate::yaml; /// The file holding the bibliography when it is kept out of `course.yaml`. pub const REFERENCES_FILE: &str = "references.yaml"; /// One registry section of a course. /// /// Used to say which file a fact came from, and to keep a fragment from /// defining something that belongs somewhere else. #[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] pub enum Section { /// Course identity. Course, /// Course-wide policy. Policy, /// Units. Units, /// Lectures. Lectures, /// Learning objectives. Objectives, /// Learning targets. Targets, /// Works the course cites. References, /// Shared stimuli. Stimuli, } impl Section { /// Every section, in the order a merged course lists them. pub const ALL: [Section; 8] = [ Section::Course, Section::Policy, Section::Units, Section::Lectures, Section::Objectives, Section::Targets, Section::References, Section::Stimuli, ]; /// The YAML key this section is written under. pub fn key(self) -> &'static str { match self { Section::Course => "course", Section::Policy => "policy", Section::Units => "units", Section::Lectures => "lectures", Section::Objectives => "learning_objectives", Section::Targets => "learning_targets", Section::References => "references", Section::Stimuli => "stimuli", } } /// What one of its entries is called in a message. pub fn noun(self) -> &'static str { match self { Section::Course => "course identity", Section::Policy => "policy", Section::Units => "unit", Section::Lectures => "lecture", Section::Objectives => "objective", Section::Targets => "target", Section::References => "reference", Section::Stimuli => "stimulus", } } /// Where a file defining this section is expected to live. pub fn home(self) -> &'static str { match self { Section::Course | Section::Policy | Section::Units => COURSE_FILE, Section::Lectures => "lectures/*.yaml", Section::Objectives | Section::Targets | Section::Stimuli => "objectives/*.yaml", Section::References => REFERENCES_FILE, } } } impl fmt::Display for Section { fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { f.write_str(self.key()) } } /// What kind of file a fragment is, which fixes what it may define. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum Role { /// `course.yaml`. May define anything, so an unsplit course still loads. Root, /// `references.yaml`. References, /// A file under `lectures/`. Lecture, /// A file under `objectives/`. Objective, } impl Role { /// Whether a file in this role may define a section. pub fn allows(self, section: Section) -> bool { match self { Role::Root => true, Role::References => section == Section::References, Role::Lecture => section == Section::Lectures, Role::Objective => matches!( section, Section::Objectives | Section::Targets | Section::Stimuli ), } } /// A short name for a message. pub fn label(self) -> &'static str { match self { Role::Root => "course", Role::References => "references", Role::Lecture => "lecture", Role::Objective => "objective", } } } /// One file's worth of course registries. /// /// Every section is optional, which is what makes this both the fragment schema /// and — with every section filled in — the schema of an unsplit `course.yaml`. /// [`CourseFile`] is the resolved model the rest of the crate reads; this is /// only what one file on disk is allowed to say. #[derive(Debug, Clone, Default, Serialize, Deserialize)] #[serde(deny_unknown_fields)] pub struct Fragment { /// Schema version this file targets. #[serde( default, deserialize_with = "yaml::flexible_string_opt", skip_serializing_if = "Option::is_none" )] pub schema_version: Option, /// Course identity. Exactly one fragment must carry it. #[serde(default, skip_serializing_if = "Option::is_none")] pub course: Option, /// Course-wide policy. #[serde(default, skip_serializing_if = "Option::is_none")] pub policy: Option, /// Units, in teaching order. #[serde(default, skip_serializing_if = "Vec::is_empty")] pub units: Vec, /// Lectures by id. #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] pub lectures: BTreeMap, /// Learning objectives by id. #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] pub learning_objectives: BTreeMap, /// Learning targets by id. #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] pub learning_targets: BTreeMap, /// Works the course cites, by citation key. #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] pub references: BTreeMap, /// Shared stimuli by id. #[serde(default, skip_serializing_if = "BTreeMap::is_empty")] pub stimuli: BTreeMap, } impl Fragment { /// Loads one fragment from disk. /// /// # Arguments /// /// * `path` - the file to read. /// /// # Returns /// /// The parsed fragment. /// /// # Errors /// /// Returns [`Error::Io`] if unreadable and [`Error::Yaml`] if it does not /// match the schema. Unknown keys are errors, so a misspelled section name /// is caught here rather than silently contributing nothing. pub fn load(path: &Path) -> Result { yaml::read(path) } /// Which sections this fragment actually defines. pub fn sections(&self) -> Vec
{ let mut out = Vec::new(); if self.course.is_some() { out.push(Section::Course); } if self.policy.is_some() { out.push(Section::Policy); } if !self.units.is_empty() { out.push(Section::Units); } if !self.lectures.is_empty() { out.push(Section::Lectures); } if !self.learning_objectives.is_empty() { out.push(Section::Objectives); } if !self.learning_targets.is_empty() { out.push(Section::Targets); } if !self.references.is_empty() { out.push(Section::References); } if !self.stimuli.is_empty() { out.push(Section::Stimuli); } out } } /// The fragment files of a course directory, in load order, with their roles. /// /// `course.yaml` is listed whether or not it exists, so a directory that is not /// a course fails with a message naming the file it wanted rather than an empty /// merge. Directory contents are sorted, which is what makes the merged course /// independent of filesystem order. /// /// # Arguments /// /// * `layout` - the resolved course layout. /// /// # Returns /// /// Paths paired with what each file is allowed to define. /// /// # Errors /// /// Returns [`Error::Io`] when a fragment directory exists but cannot be read. pub fn files(layout: &Layout) -> Result> { let mut out = vec![(layout.course_file(), Role::Root)]; let references = layout.references_file(); if references.is_file() { out.push((references, Role::References)); } for path in yaml::list_yaml(&layout.lectures())? { out.push((path, Role::Lecture)); } for path in yaml::list_yaml(&layout.objectives())? { out.push((path, Role::Objective)); } Ok(out) } /// Loads every fragment in a course directory and merges them into one course. /// /// # Arguments /// /// * `root` - the course directory. /// /// # Returns /// /// The merged course, with [`CourseFile::origins`] recording which file defined /// each id. /// /// # Errors /// /// Propagates load errors, and returns [`Error::Invalid`] with every merge /// problem at once: an id defined twice, a section in the wrong kind of file, a /// fragment written against another major schema version, or no `course:` /// section anywhere. /// /// Cross-references are *not* checked here. A dangling objective id is a /// content problem, and content problems are [`CourseFile::validate`]'s, so that /// they are reported the same way whether or not the course is split. pub fn assemble(root: &Path) -> Result { let layout = Layout::new(root); let mut merge = Merge::default(); for (path, role) in files(&layout)? { let fragment = Fragment::load(&path)?; let shown = path.strip_prefix(root).unwrap_or(&path).to_path_buf(); merge.take(&shown, role, fragment); } merge.resolve(); merge.finish(root) } /// Accumulates fragments, remembering where each id came from. #[derive(Debug, Default)] struct Merge { schema_version: Option, course: Option, policy: Option, units: Vec, lectures: BTreeMap, objectives: BTreeMap, targets: BTreeMap, references: BTreeMap, stimuli: BTreeMap, origins: BTreeMap<(Section, String), PathBuf>, issues: Vec, } impl Merge { /// Folds one fragment in. /// /// # Arguments /// /// * `path` - the fragment's path relative to the course root, for messages. /// * `role` - what this file is allowed to define. /// * `fragment` - the parsed fragment. fn take(&mut self, path: &Path, role: Role, fragment: Fragment) { for section in fragment.sections() { if !role.allows(section) { self.issues.push(format!( "{}: a {} file may not define `{}`; that section belongs in {}", path.display(), role.label(), section.key(), section.home() )); } } if let Some(declared) = &fragment.schema_version { if !SUPPORTED_MAJORS.contains(&major(declared)) { self.issues.push(format!( "{}: declares schema_version {declared}, which this build cannot read. It \ writes {SCHEMA_VERSION} and reads {}.", path.display(), SUPPORTED_MAJORS .iter() .map(|m| format!("{m}.x")) .collect::>() .join(" and ") )); } if self.schema_version.is_none() { self.schema_version = Some(declared.clone()); } } if role.allows(Section::Course) { if let Some(course) = fragment.course { if self.claim(Section::Course, path) { self.course = Some(course); } } } if role.allows(Section::Policy) { if let Some(policy) = fragment.policy { if self.claim(Section::Policy, path) { self.policy = Some(policy); } } } if role.allows(Section::Units) { for unit in fragment.units { let key = (Section::Units, unit.id.clone()); if let Some(first) = self.origins.get(&key) { let message = duplicate(Section::Units, &unit.id, first.as_path(), path); self.issues.push(message); continue; } self.origins.insert(key, path.to_path_buf()); self.units.push(unit); } } if role.allows(Section::Lectures) { absorb( &mut self.lectures, fragment.lectures, Section::Lectures, path, &mut self.origins, &mut self.issues, ); } if role.allows(Section::Objectives) { absorb( &mut self.objectives, fragment.learning_objectives, Section::Objectives, path, &mut self.origins, &mut self.issues, ); } if role.allows(Section::Targets) { absorb( &mut self.targets, fragment.learning_targets, Section::Targets, path, &mut self.origins, &mut self.issues, ); } if role.allows(Section::References) { absorb( &mut self.references, fragment.references, Section::References, path, &mut self.origins, &mut self.issues, ); } if role.allows(Section::Stimuli) { absorb( &mut self.stimuli, fragment.stimuli, Section::Stimuli, path, &mut self.origins, &mut self.issues, ); } } /// Records a section that may only be declared once. /// /// # Arguments /// /// * `section` - the section being claimed. /// * `path` - the file claiming it. /// /// # Returns /// /// Whether the claim was the first, and so whether the caller should store /// what it parsed. fn claim(&mut self, section: Section, path: &Path) -> bool { let key = (section, String::new()); if let Some(first) = self.origins.get(&key) { let message = format!( "`{}` is declared twice: {} and {}. It applies to the whole course, so it has \ one definition site.", section.key(), first.display(), path.display() ); self.issues.push(message); return false; } self.origins.insert(key, path.to_path_buf()); true } /// Fills in the two fields a split layout would otherwise duplicate. fn resolve(&mut self) { // A lecture says what it teaches; the objective's lecture list follows. for (lecture_id, lecture) in &self.lectures { for objective_id in &lecture.teaches { if let Some(objective) = self.objectives.get_mut(objective_id) { if !objective.lectures.iter().any(|l| l == lecture_id) { objective.lectures.push(lecture_id.clone()); } } } } // A target with no lecture of its own is taught wherever its objective // is. Collected first: the read of `objectives` and the write to // `targets` cannot overlap in one pass. let inherited: Vec<(String, Vec)> = self .targets .iter() .filter(|(_, target)| target.lectures.is_empty()) .filter_map(|(id, target)| { self.objectives .get(&target.objective) .map(|objective| (id.clone(), objective.lectures.clone())) }) .collect(); for (id, lectures) in inherited { if let Some(target) = self.targets.get_mut(&id) { target.lectures = lectures; } } } /// Builds the course, or reports every merge problem at once. fn finish(self, root: &Path) -> Result { let mut issues = self.issues; let course = match self.course { Some(course) => course, None => { issues.push(format!( "no file in {} declares a `course:` section, so the course has no code, \ title, or term", root.display() )); return Err(Error::Invalid(issues)); } }; if !issues.is_empty() { return Err(Error::Invalid(issues)); } Ok(CourseFile { schema_version: self .schema_version .unwrap_or_else(|| SCHEMA_VERSION.to_string()), course, policy: self.policy.unwrap_or_default(), units: self.units, lectures: self.lectures, learning_objectives: self.objectives, learning_targets: self.targets, references: self.references, stimuli: self.stimuli, origins: self.origins, }) } } /// Moves one section's entries across, refusing a second definition. fn absorb( into: &mut BTreeMap, from: BTreeMap, section: Section, path: &Path, origins: &mut BTreeMap<(Section, String), PathBuf>, issues: &mut Vec, ) { for (id, value) in from { let key = (section, id.clone()); if let Some(first) = origins.get(&key) { issues.push(duplicate(section, &id, first.as_path(), path)); continue; } origins.insert(key, path.to_path_buf()); into.insert(id, value); } } /// The message for an id defined in two files. fn duplicate(section: Section, id: &str, first: &Path, second: &Path) -> String { format!( "{} `{id}` is defined in two places: {} and {}. An id has one definition site; delete \ one or rename it.", section.noun(), first.display(), second.display() ) } /// The part of a schema version before the first dot. fn major(version: &str) -> &str { version.split('.').next().unwrap_or(version) } #[cfg(test)] mod tests { use super::*; fn tmp(tag: &str) -> PathBuf { let p = std::env::temp_dir().join(format!("coursebank-frag-{tag}-{}", std::process::id())); let _ = std::fs::remove_dir_all(&p); std::fs::create_dir_all(&p).unwrap(); p } fn write(root: &Path, relative: &str, body: &str) { let path = root.join(relative); std::fs::create_dir_all(path.parent().unwrap()).unwrap(); std::fs::write(path, body).unwrap(); } const ROOT: &str = r#" course: code: BIOSC 1540 title: Computational Biology term: 2026f policy: points_per_item: 1.0 units: - id: u1 title: Search and Similarity "#; #[test] fn a_split_course_merges_into_one_model() { let root = tmp("merge"); write(&root, "course.yaml", ROOT); write( &root, "references.yaml", "references:\n ismail2023:\n title: Bioinformatics\n", ); write( &root, "lectures/l-1-2.yaml", "lectures:\n L1.2:\n title: The Digital Genome\n unit: u1\n \ teaches: [lo-read-file-formats]\n", ); write( &root, "objectives/lo-read-file-formats.yaml", "learning_objectives:\n lo-read-file-formats:\n text: Read the text formats.\n \ unit: u1\nlearning_targets:\n t-fastq-structure:\n text: Identify the four \ lines.\n objective: lo-read-file-formats\n", ); let course = assemble(&root).unwrap(); assert_eq!(course.course.code, "BIOSC 1540"); assert_eq!(course.units.len(), 1); assert_eq!(course.references.len(), 1); assert!(course.validate().is_empty(), "{:?}", course.validate()); // Derived: the lecture registered the objective, and the target // inherited the objective's lecture. assert_eq!( course.learning_objectives["lo-read-file-formats"].lectures, vec!["L1.2".to_string()] ); assert_eq!( course.learning_targets["t-fastq-structure"].lectures, vec!["L1.2".to_string()] ); assert_eq!(course.lecture_targets("L1.2"), vec!["t-fastq-structure"]); } #[test] fn an_unsplit_course_file_still_loads() { let root = tmp("monolith"); write( &root, "course.yaml", &format!( "{ROOT}lectures:\n L1.2:\n title: The Digital Genome\nlearning_objectives:\n \ lo-x:\n text: Do the thing.\n lectures: [L1.2]\nlearning_targets:\n \ t-x:\n text: Do the smaller thing.\n objective: lo-x\nreferences:\n \ ismail2023:\n title: Bioinformatics\n" ), ); let course = assemble(&root).unwrap(); assert!(course.validate().is_empty(), "{:?}", course.validate()); assert_eq!(course.lecture_objectives("L1.2"), vec!["lo-x"]); assert_eq!( course.origin("lo-x").map(|(_, p)| p.to_path_buf()), Some(PathBuf::from(COURSE_FILE)) ); } #[test] fn an_id_defined_twice_names_both_files() { let root = tmp("dup"); write(&root, "course.yaml", ROOT); let body = "learning_objectives:\n lo-x:\n text: Do the thing.\n"; write(&root, "objectives/lo-x.yaml", body); write(&root, "objectives/lo-x-old.yaml", body); let err = assemble(&root).unwrap_err(); let message = err.to_string(); assert!(message.contains("objectives/lo-x.yaml"), "{message}"); assert!(message.contains("objectives/lo-x-old.yaml"), "{message}"); assert!(message.contains("one definition site"), "{message}"); } #[test] fn a_section_in_the_wrong_kind_of_file_is_rejected() { let root = tmp("misplaced"); write(&root, "course.yaml", ROOT); write( &root, "lectures/l-1-2.yaml", "lectures:\n L1.2:\n title: The Digital Genome\nlearning_objectives:\n lo-x:\n \ text: Do the thing.\n", ); let message = assemble(&root).unwrap_err().to_string(); assert!( message.contains("may not define `learning_objectives`"), "{message}" ); assert!(message.contains("objectives/*.yaml"), "{message}"); } #[test] fn a_course_with_no_identity_says_so() { let root = tmp("no-course"); write(&root, "course.yaml", "units:\n - id: u1\n title: One\n"); let message = assemble(&root).unwrap_err().to_string(); assert!(message.contains("`course:` section"), "{message}"); } #[test] fn a_fragment_from_another_major_version_is_refused() { let root = tmp("version"); write(&root, "course.yaml", ROOT); write( &root, "objectives/lo-x.yaml", "schema_version: '9.0'\nlearning_objectives:\n lo-x:\n text: Do the thing.\n", ); let message = assemble(&root).unwrap_err().to_string(); assert!(message.contains("schema_version 9.0"), "{message}"); } #[test] fn origins_point_at_the_fragment_that_defined_each_id() { let root = tmp("origins"); write(&root, "course.yaml", ROOT); write( &root, "lectures/l-1-2.yaml", "lectures:\n L1.2:\n title: The Digital Genome\n", ); write( &root, "objectives/lo-x.yaml", "learning_objectives:\n lo-x:\n text: Do the thing.\n", ); let course = assemble(&root).unwrap(); let (section, path) = course.origin("lo-x").unwrap(); assert_eq!(section, Section::Objectives); assert_eq!(path, Path::new("objectives/lo-x.yaml")); let (section, path) = course.origin("L1.2").unwrap(); assert_eq!(section, Section::Lectures); assert_eq!(path, Path::new("lectures/l-1-2.yaml")); assert!(course.origin("nothing-like-this").is_none()); } #[test] fn teaching_the_same_objective_from_two_lectures_unions() { let root = tmp("union"); write(&root, "course.yaml", ROOT); write( &root, "lectures/l-1-2.yaml", "lectures:\n L1.2:\n title: One\n teaches: [lo-x]\n", ); write( &root, "lectures/l-1-3.yaml", "lectures:\n L1.3:\n title: Two\n teaches: [lo-x]\n", ); write( &root, "objectives/lo-x.yaml", "learning_objectives:\n lo-x:\n text: Do the thing.\n", ); let course = assemble(&root).unwrap(); assert_eq!( course.learning_objectives["lo-x"].lectures, vec!["L1.2".to_string(), "L1.3".to_string()] ); } #[test] fn a_declaration_on_both_sides_is_not_duplicated() { let root = tmp("both-sides"); write(&root, "course.yaml", ROOT); write( &root, "lectures/l-1-2.yaml", "lectures:\n L1.2:\n title: One\n teaches: [lo-x]\n", ); write( &root, "objectives/lo-x.yaml", "learning_objectives:\n lo-x:\n text: Do the thing.\n lectures: [L1.2]\n", ); let course = assemble(&root).unwrap(); assert_eq!( course.learning_objectives["lo-x"].lectures, vec!["L1.2".to_string()] ); } #[test] fn teaching_an_unknown_objective_is_a_validation_problem_not_a_merge_one() { let root = tmp("unknown-teaches"); write(&root, "course.yaml", ROOT); write( &root, "lectures/l-1-2.yaml", "lectures:\n L1.2:\n title: One\n teaches: [lo-nope]\n", ); let course = assemble(&root).unwrap(); let issues = course.validate(); assert!(issues.iter().any(|i| i.contains("lo-nope")), "{issues:?}"); } }