Dev (#1)
Sync README to GitHub / sync (push) Successful in 12s
CI / check (push) Successful in 8m31s
Deploy docs / deploy (push) Successful in 6m44s
Nightly / nightly (push) Successful in 10m33s

Reviewed-on: #1
This commit was merged in pull request #1.
This commit is contained in:
2026-08-07 15:48:11 -04:00
parent 228a0da47f
commit abc0bdf621
79 changed files with 31370 additions and 0 deletions
+382
View File
@@ -0,0 +1,382 @@
// SPDX-License-Identifier: Prosperity-3.0.0
// Copyright Scientific Computing Studio
// Source: https://git.scient.ing/education/coursebank
//! Converting the authoring markup into HTML, plain text, and Typst.
//!
//! Stems are written in a small markup that is a subset of Typst with a few
//! Markdown conveniences, because chemistry and biology questions need
//! subscripts, arrows, and Greek letters, and typing HTML entities into YAML by
//! hand is miserable.
//!
//! There is no regular expression engine here. Every rule is a scan, which keeps
//! the dependency list short and makes the escaping order explicit: HTML is
//! escaped *first*, then symbol substitutions run, so that a substitution
//! producing `→` is not itself escaped into `→`.
/// Symbol substitutions, applied after HTML escaping.
///
/// Ordered longest-first within each family so `#sym.arrow.r` is not consumed by
/// a shorter prefix.
const SYMBOLS: &[(&str, &str, &str)] = &[
// (source token, html, plain text)
("#sym.gt.eq", "≥", "\u{2265}"),
("#sym.lt.eq", "≤", "\u{2264}"),
("#sym.eq.not", "≠", "\u{2260}"),
("#sym.plus.minus", "±", "\u{00b1}"),
("#sym.arrow.r", "→", "\u{2192}"),
("#sym.arrow.l", "←", "\u{2190}"),
("#sym.arrow.lr", "↔", "\u{2194}"),
("#sym.rightarrow", "→", "\u{2192}"),
("#sym.leftarrow", "←", "\u{2190}"),
("#sym.times", "×", "\u{00d7}"),
("#sym.dot", "·", "\u{00b7}"),
("#sym.degree", "°", "\u{00b0}"),
("#sym.infinity", "∞", "\u{221e}"),
("#sym.approx", "≈", "\u{2248}"),
("#sym.alpha", "α", "\u{03b1}"),
("#sym.beta", "β", "\u{03b2}"),
("#sym.gamma", "γ", "\u{03b3}"),
("#sym.delta.cap", "Δ", "\u{0394}"),
("#sym.delta", "δ", "\u{03b4}"),
("#sym.epsilon", "ε", "\u{03b5}"),
("#sym.lambda", "λ", "\u{03bb}"),
("#sym.mu", "μ", "\u{03bc}"),
("#sym.pi", "π", "\u{03c0}"),
("#sym.sigma", "σ", "\u{03c3}"),
("#sym.tau", "τ", "\u{03c4}"),
("#sym.phi", "φ", "\u{03c6}"),
("#sym.omega", "ω", "\u{03c9}"),
];
/// Converts authoring markup to an HTML fragment.
///
/// Handles paragraphs, bold, italic, inline code, subscripts, superscripts, and
/// the symbol table. Anything unrecognized passes through escaped, so a stray
/// `<script>` in a stem cannot become markup in a Canvas quiz.
///
/// # Arguments
///
/// * `src` - the authoring source.
///
/// # Returns
///
/// An HTML fragment, with each paragraph wrapped in `<p>`.
pub fn to_html(src: &str) -> String {
let escaped = escape_html(src);
let symbolized = apply_symbols(&escaped, true);
let inline = apply_inline(&symbolized);
let paragraphs: Vec<String> = inline
.split("\n\n")
.map(|p| p.trim())
.filter(|p| !p.is_empty())
.map(|p| {
let joined = p
.lines()
.map(|l| l.trim())
.filter(|l| !l.is_empty())
.collect::<Vec<_>>()
.join(" ");
format!("<p>{joined}</p>")
})
.collect();
if paragraphs.is_empty() {
String::new()
} else {
paragraphs.join("\n")
}
}
/// Converts authoring markup to plain text.
///
/// Used for CSV columns, terminal output, and any place a fragment of HTML would
/// be noise.
///
/// # Arguments
///
/// * `src` - the authoring source.
///
/// # Returns
///
/// Plain text with markup removed and symbols rendered as Unicode.
pub fn to_plain(src: &str) -> String {
let symbolized = apply_symbols(src, false);
let mut out = strip_inline(&symbolized);
out = out
.lines()
.map(|l| l.trim())
.filter(|l| !l.is_empty())
.collect::<Vec<_>>()
.join(" ");
out.trim().to_string()
}
/// Passes authoring markup through for Typst.
///
/// The markup is already a Typst subset, so this only normalizes whitespace and
/// escapes the few characters Typst treats specially in content mode.
///
/// # Arguments
///
/// * `src` - the authoring source.
///
/// # Returns
///
/// Typst content-mode markup.
pub fn to_typst(src: &str) -> String {
let mut out = String::with_capacity(src.len());
for ch in src.trim().chars() {
match ch {
// A bare `@` or `<` starts a Typst reference or label.
'@' => out.push_str("\\@"),
'<' => out.push_str("\\<"),
'>' => out.push_str("\\>"),
_ => out.push(ch),
}
}
out
}
/// Escapes the five XML-significant characters.
///
/// # Arguments
///
/// * `s` - the text to escape.
///
/// # Returns
///
/// The escaped text.
pub fn escape_html(s: &str) -> String {
let mut out = String::with_capacity(s.len());
for ch in s.chars() {
match ch {
'&' => out.push_str("&amp;"),
'<' => out.push_str("&lt;"),
'>' => out.push_str("&gt;"),
'"' => out.push_str("&quot;"),
'\'' => out.push_str("&apos;"),
_ => out.push(ch),
}
}
out
}
/// Applies the symbol table.
///
/// # Arguments
///
/// * `s` - the text.
/// * `html` - whether to emit HTML entities rather than Unicode.
///
/// # Returns
///
/// The substituted text.
fn apply_symbols(s: &str, html: bool) -> String {
let mut out = s.to_string();
for (token, entity, plain) in SYMBOLS {
if out.contains(token) {
out = out.replace(token, if html { entity } else { plain });
}
}
out
}
/// Applies inline markup rules, producing HTML.
///
/// # Arguments
///
/// * `s` - escaped, symbol-substituted text.
///
/// # Returns
///
/// The text with inline markup converted.
fn apply_inline(s: &str) -> String {
let mut out = s.to_string();
// Bracketed forms first: their contents may contain other markup characters.
out = wrap_bracket(&out, "#sub[", "<sub>", "</sub>");
out = wrap_bracket(&out, "#sup[", "<sup>", "</sup>");
out = wrap_delimited(&out, "`", "<code>", "</code>");
out = wrap_delimited(&out, "**", "<strong>", "</strong>");
out = wrap_delimited(&out, "*", "<em>", "</em>");
out = wrap_delimited(&out, "_", "<em>", "</em>");
out
}
/// Removes inline markup without replacing it.
///
/// # Arguments
///
/// * `s` - the text.
///
/// # Returns
///
/// The text with markup delimiters stripped.
fn strip_inline(s: &str) -> String {
let mut out = s.to_string();
out = wrap_bracket(&out, "#sub[", "", "");
out = wrap_bracket(&out, "#sup[", "", "");
out = wrap_delimited(&out, "`", "", "");
out = wrap_delimited(&out, "**", "", "");
out = wrap_delimited(&out, "*", "", "");
out = wrap_delimited(&out, "_", "", "");
out
}
/// Replaces `open...]` spans with wrapped content.
///
/// # Arguments
///
/// * `s` - the text.
/// * `open` - the opening token, e.g. `"#sub["`.
/// * `pre` - text to emit before the content.
/// * `post` - text to emit after the content.
///
/// # Returns
///
/// The rewritten text. Unclosed spans are left alone.
fn wrap_bracket(s: &str, open: &str, pre: &str, post: &str) -> String {
let mut out = String::with_capacity(s.len());
let mut rest = s;
loop {
match rest.find(open) {
None => {
out.push_str(rest);
return out;
}
Some(i) => {
let after = &rest[i + open.len()..];
match after.find(']') {
None => {
out.push_str(rest);
return out;
}
Some(j) => {
out.push_str(&rest[..i]);
out.push_str(pre);
out.push_str(&after[..j]);
out.push_str(post);
rest = &after[j + 1..];
}
}
}
}
}
}
/// Replaces paired `delim...delim` spans with wrapped content.
///
/// A delimiter with no partner is emitted literally, so an apostrophe-heavy stem
/// or a lone asterisk does not swallow the rest of the text.
///
/// # Arguments
///
/// * `s` - the text.
/// * `delim` - the delimiter, e.g. `"**"`.
/// * `pre` - text to emit before the content.
/// * `post` - text to emit after the content.
///
/// # Returns
///
/// The rewritten text.
fn wrap_delimited(s: &str, delim: &str, pre: &str, post: &str) -> String {
let mut out = String::with_capacity(s.len());
let mut rest = s;
loop {
match rest.find(delim) {
None => {
out.push_str(rest);
return out;
}
Some(i) => {
let after = &rest[i + delim.len()..];
match after.find(delim) {
None => {
out.push_str(rest);
return out;
}
Some(0) => {
// Empty span such as `**`; emit literally and move on.
out.push_str(&rest[..i + delim.len()]);
rest = after;
}
Some(j) => {
out.push_str(&rest[..i]);
out.push_str(pre);
out.push_str(&after[..j]);
out.push_str(post);
rest = &after[j + delim.len()..];
}
}
}
}
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn escapes_before_substituting() {
// The entity produced by the symbol table must survive escaping.
assert_eq!(to_html("a #sym.arrow.r b"), "<p>a &rarr; b</p>");
// A literal ampersand is escaped.
assert_eq!(to_html("Tris & HCl"), "<p>Tris &amp; HCl</p>");
}
#[test]
fn refuses_to_pass_through_html() {
let out = to_html("<script>alert(1)</script>");
assert!(!out.contains("<script>"));
assert!(out.contains("&lt;script&gt;"));
}
#[test]
fn converts_inline_markup() {
assert_eq!(to_html("**bold**"), "<p><strong>bold</strong></p>");
assert_eq!(to_html("*em*"), "<p><em>em</em></p>");
assert_eq!(to_html("`code`"), "<p><code>code</code></p>");
assert_eq!(to_html("H#sub[2]O"), "<p>H<sub>2</sub>O</p>");
assert_eq!(to_html("x#sup[2]"), "<p>x<sup>2</sup></p>");
}
#[test]
fn bold_wins_over_italic() {
assert_eq!(
to_html("**strong** and *weak*"),
"<p><strong>strong</strong> and <em>weak</em></p>"
);
}
#[test]
fn unpaired_delimiters_are_literal() {
assert_eq!(to_html("2 * 3 = 6"), "<p>2 * 3 = 6</p>");
assert_eq!(to_html("a_b"), "<p>a_b</p>");
}
#[test]
fn splits_paragraphs_and_joins_wrapped_lines() {
let out = to_html("first line\ncontinued\n\nsecond paragraph");
assert_eq!(out, "<p>first line continued</p>\n<p>second paragraph</p>");
}
#[test]
fn empty_input_yields_empty_output() {
assert_eq!(to_html(" \n "), "");
assert_eq!(to_plain(""), "");
}
#[test]
fn plain_text_uses_unicode_and_drops_markup() {
assert_eq!(to_plain("K#sub[m] #sym.approx 5 mM"), "Km \u{2248} 5 mM");
assert_eq!(to_plain("**bold** text"), "bold text");
}
#[test]
fn typst_escapes_reference_starters() {
assert_eq!(to_typst("a @ b"), "a \\@ b");
assert_eq!(to_typst("x < y"), "x \\< y");
}
}