feat: splitting

This commit is contained in:
2026-09-26 01:15:12 -04:00
parent 220363d4d3
commit eabc98ad31
29 changed files with 4843 additions and 234 deletions
+866
View File
@@ -0,0 +1,866 @@
// SPDX-License-Identifier: Prosperity-3.0.0
// Copyright Scientific Computing Studio
// Source: https://git.scient.ing/education/coursebank
//! One course, several files.
//!
//! A course of forty lectures does not fit in a file anyone wants to scroll. So
//! the registries [`CourseFile`] holds may be spread across a directory and
//! merged on load:
//!
//! ```text
//! course.yaml course, policy, units
//! references.yaml references
//! lectures/l-1-2.yaml one lecture, its readings, and what it teaches
//! objectives/lo-x.yaml one objective and its targets
//! ```
//!
//! The merge happens in memory on every command. Nothing is generated on disk
//! and no command depends on a build step, because a generated file that other
//! commands read is a file that goes stale. `coursebank course build` exists to
//! show you the merged result, and nothing reads what it writes.
//!
//! # What this does not relax
//!
//! Splitting a file is only worth doing if it cannot introduce a second
//! definition of the same thing. Two rules keep that true, and both are enforced
//! here rather than left to convention:
//!
//! * **One definition site per id.** Two files defining `lo-read-file-formats`
//! is an error naming both paths. The winner is not the last file loaded,
//! because there is no winner.
//! * **A section belongs to a kind of file.** A file under `lectures/` may not
//! define `learning_objectives`. Otherwise the layout decays into forty files
//! that each might hold anything, which is the same navigation problem in a
//! worse shape.
//!
//! `course.yaml` is exempt from the second rule: a course that has not been
//! split is a single fragment that happens to define everything, and it keeps
//! loading unchanged.
//!
//! # Two derivations
//!
//! Splitting by lecture makes two fields tedious to maintain by hand, so they
//! are derived instead:
//!
//! * A lecture's `teaches:` list adds that lecture to each named objective's
//! `lectures`. You write what a lecture covers while planning the lecture,
//! which is when you know.
//! * A target with no `lectures` of its own inherits its objective's, the same
//! way it already inherits `level_ceiling`.
//!
//! Both are unions and both are idempotent, so declaring a pair on both sides
//! is redundant rather than contradictory.
use std::collections::BTreeMap;
use std::fmt;
use std::path::{Path, PathBuf};
use serde::{Deserialize, Serialize};
use super::{
COURSE_FILE, Course, CourseFile, Lecture, Objective, Policy, Reference, SCHEMA_VERSION,
SUPPORTED_MAJORS, Stimulus, Target, Unit,
};
use crate::error::{Error, Result};
use crate::layout::Layout;
use crate::yaml;
/// The file holding the bibliography when it is kept out of `course.yaml`.
pub const REFERENCES_FILE: &str = "references.yaml";
/// One registry section of a course.
///
/// Used to say which file a fact came from, and to keep a fragment from
/// defining something that belongs somewhere else.
#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)]
pub enum Section {
/// Course identity.
Course,
/// Course-wide policy.
Policy,
/// Units.
Units,
/// Lectures.
Lectures,
/// Learning objectives.
Objectives,
/// Learning targets.
Targets,
/// Works the course cites.
References,
/// Shared stimuli.
Stimuli,
}
impl Section {
/// Every section, in the order a merged course lists them.
pub const ALL: [Section; 8] = [
Section::Course,
Section::Policy,
Section::Units,
Section::Lectures,
Section::Objectives,
Section::Targets,
Section::References,
Section::Stimuli,
];
/// The YAML key this section is written under.
pub fn key(self) -> &'static str {
match self {
Section::Course => "course",
Section::Policy => "policy",
Section::Units => "units",
Section::Lectures => "lectures",
Section::Objectives => "learning_objectives",
Section::Targets => "learning_targets",
Section::References => "references",
Section::Stimuli => "stimuli",
}
}
/// What one of its entries is called in a message.
pub fn noun(self) -> &'static str {
match self {
Section::Course => "course identity",
Section::Policy => "policy",
Section::Units => "unit",
Section::Lectures => "lecture",
Section::Objectives => "objective",
Section::Targets => "target",
Section::References => "reference",
Section::Stimuli => "stimulus",
}
}
/// Where a file defining this section is expected to live.
pub fn home(self) -> &'static str {
match self {
Section::Course | Section::Policy | Section::Units => COURSE_FILE,
Section::Lectures => "lectures/*.yaml",
Section::Objectives | Section::Targets | Section::Stimuli => "objectives/*.yaml",
Section::References => REFERENCES_FILE,
}
}
}
impl fmt::Display for Section {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
f.write_str(self.key())
}
}
/// What kind of file a fragment is, which fixes what it may define.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Role {
/// `course.yaml`. May define anything, so an unsplit course still loads.
Root,
/// `references.yaml`.
References,
/// A file under `lectures/`.
Lecture,
/// A file under `objectives/`.
Objective,
}
impl Role {
/// Whether a file in this role may define a section.
pub fn allows(self, section: Section) -> bool {
match self {
Role::Root => true,
Role::References => section == Section::References,
Role::Lecture => section == Section::Lectures,
Role::Objective => matches!(
section,
Section::Objectives | Section::Targets | Section::Stimuli
),
}
}
/// A short name for a message.
pub fn label(self) -> &'static str {
match self {
Role::Root => "course",
Role::References => "references",
Role::Lecture => "lecture",
Role::Objective => "objective",
}
}
}
/// One file's worth of course registries.
///
/// Every section is optional, which is what makes this both the fragment schema
/// and — with every section filled in — the schema of an unsplit `course.yaml`.
/// [`CourseFile`] is the resolved model the rest of the crate reads; this is
/// only what one file on disk is allowed to say.
#[derive(Debug, Clone, Default, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct Fragment {
/// Schema version this file targets.
#[serde(
default,
deserialize_with = "yaml::flexible_string_opt",
skip_serializing_if = "Option::is_none"
)]
pub schema_version: Option<String>,
/// Course identity. Exactly one fragment must carry it.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub course: Option<Course>,
/// Course-wide policy.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub policy: Option<Policy>,
/// Units, in teaching order.
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub units: Vec<Unit>,
/// Lectures by id.
#[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
pub lectures: BTreeMap<String, Lecture>,
/// Learning objectives by id.
#[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
pub learning_objectives: BTreeMap<String, Objective>,
/// Learning targets by id.
#[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
pub learning_targets: BTreeMap<String, Target>,
/// Works the course cites, by citation key.
#[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
pub references: BTreeMap<String, Reference>,
/// Shared stimuli by id.
#[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
pub stimuli: BTreeMap<String, Stimulus>,
}
impl Fragment {
/// Loads one fragment from disk.
///
/// # Arguments
///
/// * `path` - the file to read.
///
/// # Returns
///
/// The parsed fragment.
///
/// # Errors
///
/// Returns [`Error::Io`] if unreadable and [`Error::Yaml`] if it does not
/// match the schema. Unknown keys are errors, so a misspelled section name
/// is caught here rather than silently contributing nothing.
pub fn load(path: &Path) -> Result<Fragment> {
yaml::read(path)
}
/// Which sections this fragment actually defines.
pub fn sections(&self) -> Vec<Section> {
let mut out = Vec::new();
if self.course.is_some() {
out.push(Section::Course);
}
if self.policy.is_some() {
out.push(Section::Policy);
}
if !self.units.is_empty() {
out.push(Section::Units);
}
if !self.lectures.is_empty() {
out.push(Section::Lectures);
}
if !self.learning_objectives.is_empty() {
out.push(Section::Objectives);
}
if !self.learning_targets.is_empty() {
out.push(Section::Targets);
}
if !self.references.is_empty() {
out.push(Section::References);
}
if !self.stimuli.is_empty() {
out.push(Section::Stimuli);
}
out
}
}
/// The fragment files of a course directory, in load order, with their roles.
///
/// `course.yaml` is listed whether or not it exists, so a directory that is not
/// a course fails with a message naming the file it wanted rather than an empty
/// merge. Directory contents are sorted, which is what makes the merged course
/// independent of filesystem order.
///
/// # Arguments
///
/// * `layout` - the resolved course layout.
///
/// # Returns
///
/// Paths paired with what each file is allowed to define.
///
/// # Errors
///
/// Returns [`Error::Io`] when a fragment directory exists but cannot be read.
pub fn files(layout: &Layout) -> Result<Vec<(PathBuf, Role)>> {
let mut out = vec![(layout.course_file(), Role::Root)];
let references = layout.references_file();
if references.is_file() {
out.push((references, Role::References));
}
for path in yaml::list_yaml(&layout.lectures())? {
out.push((path, Role::Lecture));
}
for path in yaml::list_yaml(&layout.objectives())? {
out.push((path, Role::Objective));
}
Ok(out)
}
/// Loads every fragment in a course directory and merges them into one course.
///
/// # Arguments
///
/// * `root` - the course directory.
///
/// # Returns
///
/// The merged course, with [`CourseFile::origins`] recording which file defined
/// each id.
///
/// # Errors
///
/// Propagates load errors, and returns [`Error::Invalid`] with every merge
/// problem at once: an id defined twice, a section in the wrong kind of file, a
/// fragment written against another major schema version, or no `course:`
/// section anywhere.
///
/// Cross-references are *not* checked here. A dangling objective id is a
/// content problem, and content problems are [`CourseFile::validate`]'s, so that
/// they are reported the same way whether or not the course is split.
pub fn assemble(root: &Path) -> Result<CourseFile> {
let layout = Layout::new(root);
let mut merge = Merge::default();
for (path, role) in files(&layout)? {
let fragment = Fragment::load(&path)?;
let shown = path.strip_prefix(root).unwrap_or(&path).to_path_buf();
merge.take(&shown, role, fragment);
}
merge.resolve();
merge.finish(root)
}
/// Accumulates fragments, remembering where each id came from.
#[derive(Debug, Default)]
struct Merge {
schema_version: Option<String>,
course: Option<Course>,
policy: Option<Policy>,
units: Vec<Unit>,
lectures: BTreeMap<String, Lecture>,
objectives: BTreeMap<String, Objective>,
targets: BTreeMap<String, Target>,
references: BTreeMap<String, Reference>,
stimuli: BTreeMap<String, Stimulus>,
origins: BTreeMap<(Section, String), PathBuf>,
issues: Vec<String>,
}
impl Merge {
/// Folds one fragment in.
///
/// # Arguments
///
/// * `path` - the fragment's path relative to the course root, for messages.
/// * `role` - what this file is allowed to define.
/// * `fragment` - the parsed fragment.
fn take(&mut self, path: &Path, role: Role, fragment: Fragment) {
for section in fragment.sections() {
if !role.allows(section) {
self.issues.push(format!(
"{}: a {} file may not define `{}`; that section belongs in {}",
path.display(),
role.label(),
section.key(),
section.home()
));
}
}
if let Some(declared) = &fragment.schema_version {
if !SUPPORTED_MAJORS.contains(&major(declared)) {
self.issues.push(format!(
"{}: declares schema_version {declared}, which this build cannot read. It \
writes {SCHEMA_VERSION} and reads {}.",
path.display(),
SUPPORTED_MAJORS
.iter()
.map(|m| format!("{m}.x"))
.collect::<Vec<_>>()
.join(" and ")
));
}
if self.schema_version.is_none() {
self.schema_version = Some(declared.clone());
}
}
if role.allows(Section::Course) {
if let Some(course) = fragment.course {
if self.claim(Section::Course, path) {
self.course = Some(course);
}
}
}
if role.allows(Section::Policy) {
if let Some(policy) = fragment.policy {
if self.claim(Section::Policy, path) {
self.policy = Some(policy);
}
}
}
if role.allows(Section::Units) {
for unit in fragment.units {
let key = (Section::Units, unit.id.clone());
if let Some(first) = self.origins.get(&key) {
let message = duplicate(Section::Units, &unit.id, first.as_path(), path);
self.issues.push(message);
continue;
}
self.origins.insert(key, path.to_path_buf());
self.units.push(unit);
}
}
if role.allows(Section::Lectures) {
absorb(
&mut self.lectures,
fragment.lectures,
Section::Lectures,
path,
&mut self.origins,
&mut self.issues,
);
}
if role.allows(Section::Objectives) {
absorb(
&mut self.objectives,
fragment.learning_objectives,
Section::Objectives,
path,
&mut self.origins,
&mut self.issues,
);
}
if role.allows(Section::Targets) {
absorb(
&mut self.targets,
fragment.learning_targets,
Section::Targets,
path,
&mut self.origins,
&mut self.issues,
);
}
if role.allows(Section::References) {
absorb(
&mut self.references,
fragment.references,
Section::References,
path,
&mut self.origins,
&mut self.issues,
);
}
if role.allows(Section::Stimuli) {
absorb(
&mut self.stimuli,
fragment.stimuli,
Section::Stimuli,
path,
&mut self.origins,
&mut self.issues,
);
}
}
/// Records a section that may only be declared once.
///
/// # Arguments
///
/// * `section` - the section being claimed.
/// * `path` - the file claiming it.
///
/// # Returns
///
/// Whether the claim was the first, and so whether the caller should store
/// what it parsed.
fn claim(&mut self, section: Section, path: &Path) -> bool {
let key = (section, String::new());
if let Some(first) = self.origins.get(&key) {
let message = format!(
"`{}` is declared twice: {} and {}. It applies to the whole course, so it has \
one definition site.",
section.key(),
first.display(),
path.display()
);
self.issues.push(message);
return false;
}
self.origins.insert(key, path.to_path_buf());
true
}
/// Fills in the two fields a split layout would otherwise duplicate.
fn resolve(&mut self) {
// A lecture says what it teaches; the objective's lecture list follows.
for (lecture_id, lecture) in &self.lectures {
for objective_id in &lecture.teaches {
if let Some(objective) = self.objectives.get_mut(objective_id) {
if !objective.lectures.iter().any(|l| l == lecture_id) {
objective.lectures.push(lecture_id.clone());
}
}
}
}
// A target with no lecture of its own is taught wherever its objective
// is. Collected first: the read of `objectives` and the write to
// `targets` cannot overlap in one pass.
let inherited: Vec<(String, Vec<String>)> = self
.targets
.iter()
.filter(|(_, target)| target.lectures.is_empty())
.filter_map(|(id, target)| {
self.objectives
.get(&target.objective)
.map(|objective| (id.clone(), objective.lectures.clone()))
})
.collect();
for (id, lectures) in inherited {
if let Some(target) = self.targets.get_mut(&id) {
target.lectures = lectures;
}
}
}
/// Builds the course, or reports every merge problem at once.
fn finish(self, root: &Path) -> Result<CourseFile> {
let mut issues = self.issues;
let course = match self.course {
Some(course) => course,
None => {
issues.push(format!(
"no file in {} declares a `course:` section, so the course has no code, \
title, or term",
root.display()
));
return Err(Error::Invalid(issues));
}
};
if !issues.is_empty() {
return Err(Error::Invalid(issues));
}
Ok(CourseFile {
schema_version: self
.schema_version
.unwrap_or_else(|| SCHEMA_VERSION.to_string()),
course,
policy: self.policy.unwrap_or_default(),
units: self.units,
lectures: self.lectures,
learning_objectives: self.objectives,
learning_targets: self.targets,
references: self.references,
stimuli: self.stimuli,
origins: self.origins,
})
}
}
/// Moves one section's entries across, refusing a second definition.
fn absorb<T>(
into: &mut BTreeMap<String, T>,
from: BTreeMap<String, T>,
section: Section,
path: &Path,
origins: &mut BTreeMap<(Section, String), PathBuf>,
issues: &mut Vec<String>,
) {
for (id, value) in from {
let key = (section, id.clone());
if let Some(first) = origins.get(&key) {
issues.push(duplicate(section, &id, first.as_path(), path));
continue;
}
origins.insert(key, path.to_path_buf());
into.insert(id, value);
}
}
/// The message for an id defined in two files.
fn duplicate(section: Section, id: &str, first: &Path, second: &Path) -> String {
format!(
"{} `{id}` is defined in two places: {} and {}. An id has one definition site; delete \
one or rename it.",
section.noun(),
first.display(),
second.display()
)
}
/// The part of a schema version before the first dot.
fn major(version: &str) -> &str {
version.split('.').next().unwrap_or(version)
}
#[cfg(test)]
mod tests {
use super::*;
fn tmp(tag: &str) -> PathBuf {
let p = std::env::temp_dir().join(format!("coursebank-frag-{tag}-{}", std::process::id()));
let _ = std::fs::remove_dir_all(&p);
std::fs::create_dir_all(&p).unwrap();
p
}
fn write(root: &Path, relative: &str, body: &str) {
let path = root.join(relative);
std::fs::create_dir_all(path.parent().unwrap()).unwrap();
std::fs::write(path, body).unwrap();
}
const ROOT: &str = r#"
course:
code: BIOSC 1540
title: Computational Biology
term: 2026f
policy:
points_per_item: 1.0
units:
- id: u1
title: Search and Similarity
"#;
#[test]
fn a_split_course_merges_into_one_model() {
let root = tmp("merge");
write(&root, "course.yaml", ROOT);
write(
&root,
"references.yaml",
"references:\n ismail2023:\n title: Bioinformatics\n",
);
write(
&root,
"lectures/l-1-2.yaml",
"lectures:\n L1.2:\n title: The Digital Genome\n unit: u1\n \
teaches: [lo-read-file-formats]\n",
);
write(
&root,
"objectives/lo-read-file-formats.yaml",
"learning_objectives:\n lo-read-file-formats:\n text: Read the text formats.\n \
unit: u1\nlearning_targets:\n t-fastq-structure:\n text: Identify the four \
lines.\n objective: lo-read-file-formats\n",
);
let course = assemble(&root).unwrap();
assert_eq!(course.course.code, "BIOSC 1540");
assert_eq!(course.units.len(), 1);
assert_eq!(course.references.len(), 1);
assert!(course.validate().is_empty(), "{:?}", course.validate());
// Derived: the lecture registered the objective, and the target
// inherited the objective's lecture.
assert_eq!(
course.learning_objectives["lo-read-file-formats"].lectures,
vec!["L1.2".to_string()]
);
assert_eq!(
course.learning_targets["t-fastq-structure"].lectures,
vec!["L1.2".to_string()]
);
assert_eq!(course.lecture_targets("L1.2"), vec!["t-fastq-structure"]);
}
#[test]
fn an_unsplit_course_file_still_loads() {
let root = tmp("monolith");
write(
&root,
"course.yaml",
&format!(
"{ROOT}lectures:\n L1.2:\n title: The Digital Genome\nlearning_objectives:\n \
lo-x:\n text: Do the thing.\n lectures: [L1.2]\nlearning_targets:\n \
t-x:\n text: Do the smaller thing.\n objective: lo-x\nreferences:\n \
ismail2023:\n title: Bioinformatics\n"
),
);
let course = assemble(&root).unwrap();
assert!(course.validate().is_empty(), "{:?}", course.validate());
assert_eq!(course.lecture_objectives("L1.2"), vec!["lo-x"]);
assert_eq!(
course.origin("lo-x").map(|(_, p)| p.to_path_buf()),
Some(PathBuf::from(COURSE_FILE))
);
}
#[test]
fn an_id_defined_twice_names_both_files() {
let root = tmp("dup");
write(&root, "course.yaml", ROOT);
let body = "learning_objectives:\n lo-x:\n text: Do the thing.\n";
write(&root, "objectives/lo-x.yaml", body);
write(&root, "objectives/lo-x-old.yaml", body);
let err = assemble(&root).unwrap_err();
let message = err.to_string();
assert!(message.contains("objectives/lo-x.yaml"), "{message}");
assert!(message.contains("objectives/lo-x-old.yaml"), "{message}");
assert!(message.contains("one definition site"), "{message}");
}
#[test]
fn a_section_in_the_wrong_kind_of_file_is_rejected() {
let root = tmp("misplaced");
write(&root, "course.yaml", ROOT);
write(
&root,
"lectures/l-1-2.yaml",
"lectures:\n L1.2:\n title: The Digital Genome\nlearning_objectives:\n lo-x:\n \
text: Do the thing.\n",
);
let message = assemble(&root).unwrap_err().to_string();
assert!(
message.contains("may not define `learning_objectives`"),
"{message}"
);
assert!(message.contains("objectives/*.yaml"), "{message}");
}
#[test]
fn a_course_with_no_identity_says_so() {
let root = tmp("no-course");
write(&root, "course.yaml", "units:\n - id: u1\n title: One\n");
let message = assemble(&root).unwrap_err().to_string();
assert!(message.contains("`course:` section"), "{message}");
}
#[test]
fn a_fragment_from_another_major_version_is_refused() {
let root = tmp("version");
write(&root, "course.yaml", ROOT);
write(
&root,
"objectives/lo-x.yaml",
"schema_version: '9.0'\nlearning_objectives:\n lo-x:\n text: Do the thing.\n",
);
let message = assemble(&root).unwrap_err().to_string();
assert!(message.contains("schema_version 9.0"), "{message}");
}
#[test]
fn origins_point_at_the_fragment_that_defined_each_id() {
let root = tmp("origins");
write(&root, "course.yaml", ROOT);
write(
&root,
"lectures/l-1-2.yaml",
"lectures:\n L1.2:\n title: The Digital Genome\n",
);
write(
&root,
"objectives/lo-x.yaml",
"learning_objectives:\n lo-x:\n text: Do the thing.\n",
);
let course = assemble(&root).unwrap();
let (section, path) = course.origin("lo-x").unwrap();
assert_eq!(section, Section::Objectives);
assert_eq!(path, Path::new("objectives/lo-x.yaml"));
let (section, path) = course.origin("L1.2").unwrap();
assert_eq!(section, Section::Lectures);
assert_eq!(path, Path::new("lectures/l-1-2.yaml"));
assert!(course.origin("nothing-like-this").is_none());
}
#[test]
fn teaching_the_same_objective_from_two_lectures_unions() {
let root = tmp("union");
write(&root, "course.yaml", ROOT);
write(
&root,
"lectures/l-1-2.yaml",
"lectures:\n L1.2:\n title: One\n teaches: [lo-x]\n",
);
write(
&root,
"lectures/l-1-3.yaml",
"lectures:\n L1.3:\n title: Two\n teaches: [lo-x]\n",
);
write(
&root,
"objectives/lo-x.yaml",
"learning_objectives:\n lo-x:\n text: Do the thing.\n",
);
let course = assemble(&root).unwrap();
assert_eq!(
course.learning_objectives["lo-x"].lectures,
vec!["L1.2".to_string(), "L1.3".to_string()]
);
}
#[test]
fn a_declaration_on_both_sides_is_not_duplicated() {
let root = tmp("both-sides");
write(&root, "course.yaml", ROOT);
write(
&root,
"lectures/l-1-2.yaml",
"lectures:\n L1.2:\n title: One\n teaches: [lo-x]\n",
);
write(
&root,
"objectives/lo-x.yaml",
"learning_objectives:\n lo-x:\n text: Do the thing.\n lectures: [L1.2]\n",
);
let course = assemble(&root).unwrap();
assert_eq!(
course.learning_objectives["lo-x"].lectures,
vec!["L1.2".to_string()]
);
}
#[test]
fn teaching_an_unknown_objective_is_a_validation_problem_not_a_merge_one() {
let root = tmp("unknown-teaches");
write(&root, "course.yaml", ROOT);
write(
&root,
"lectures/l-1-2.yaml",
"lectures:\n L1.2:\n title: One\n teaches: [lo-nope]\n",
);
let course = assemble(&root).unwrap();
let issues = course.validate();
assert!(issues.iter().any(|i| i.contains("lo-nope")), "{issues:?}");
}
}