@@ -0,0 +1,382 @@
|
||||
// SPDX-License-Identifier: Prosperity-3.0.0
|
||||
// Copyright Scientific Computing Studio
|
||||
// Source: https://git.scient.ing/education/coursebank
|
||||
|
||||
//! Converting the authoring markup into HTML, plain text, and Typst.
|
||||
//!
|
||||
//! Stems are written in a small markup that is a subset of Typst with a few
|
||||
//! Markdown conveniences, because chemistry and biology questions need
|
||||
//! subscripts, arrows, and Greek letters, and typing HTML entities into YAML by
|
||||
//! hand is miserable.
|
||||
//!
|
||||
//! There is no regular expression engine here. Every rule is a scan, which keeps
|
||||
//! the dependency list short and makes the escaping order explicit: HTML is
|
||||
//! escaped *first*, then symbol substitutions run, so that a substitution
|
||||
//! producing `→` is not itself escaped into `→`.
|
||||
|
||||
/// Symbol substitutions, applied after HTML escaping.
|
||||
///
|
||||
/// Ordered longest-first within each family so `#sym.arrow.r` is not consumed by
|
||||
/// a shorter prefix.
|
||||
const SYMBOLS: &[(&str, &str, &str)] = &[
|
||||
// (source token, html, plain text)
|
||||
("#sym.gt.eq", "≥", "\u{2265}"),
|
||||
("#sym.lt.eq", "≤", "\u{2264}"),
|
||||
("#sym.eq.not", "≠", "\u{2260}"),
|
||||
("#sym.plus.minus", "±", "\u{00b1}"),
|
||||
("#sym.arrow.r", "→", "\u{2192}"),
|
||||
("#sym.arrow.l", "←", "\u{2190}"),
|
||||
("#sym.arrow.lr", "↔", "\u{2194}"),
|
||||
("#sym.rightarrow", "→", "\u{2192}"),
|
||||
("#sym.leftarrow", "←", "\u{2190}"),
|
||||
("#sym.times", "×", "\u{00d7}"),
|
||||
("#sym.dot", "·", "\u{00b7}"),
|
||||
("#sym.degree", "°", "\u{00b0}"),
|
||||
("#sym.infinity", "∞", "\u{221e}"),
|
||||
("#sym.approx", "≈", "\u{2248}"),
|
||||
("#sym.alpha", "α", "\u{03b1}"),
|
||||
("#sym.beta", "β", "\u{03b2}"),
|
||||
("#sym.gamma", "γ", "\u{03b3}"),
|
||||
("#sym.delta.cap", "Δ", "\u{0394}"),
|
||||
("#sym.delta", "δ", "\u{03b4}"),
|
||||
("#sym.epsilon", "ε", "\u{03b5}"),
|
||||
("#sym.lambda", "λ", "\u{03bb}"),
|
||||
("#sym.mu", "μ", "\u{03bc}"),
|
||||
("#sym.pi", "π", "\u{03c0}"),
|
||||
("#sym.sigma", "σ", "\u{03c3}"),
|
||||
("#sym.tau", "τ", "\u{03c4}"),
|
||||
("#sym.phi", "φ", "\u{03c6}"),
|
||||
("#sym.omega", "ω", "\u{03c9}"),
|
||||
];
|
||||
|
||||
/// Converts authoring markup to an HTML fragment.
|
||||
///
|
||||
/// Handles paragraphs, bold, italic, inline code, subscripts, superscripts, and
|
||||
/// the symbol table. Anything unrecognized passes through escaped, so a stray
|
||||
/// `<script>` in a stem cannot become markup in a Canvas quiz.
|
||||
///
|
||||
/// # Arguments
|
||||
///
|
||||
/// * `src` - the authoring source.
|
||||
///
|
||||
/// # Returns
|
||||
///
|
||||
/// An HTML fragment, with each paragraph wrapped in `<p>`.
|
||||
pub fn to_html(src: &str) -> String {
|
||||
let escaped = escape_html(src);
|
||||
let symbolized = apply_symbols(&escaped, true);
|
||||
let inline = apply_inline(&symbolized);
|
||||
|
||||
let paragraphs: Vec<String> = inline
|
||||
.split("\n\n")
|
||||
.map(|p| p.trim())
|
||||
.filter(|p| !p.is_empty())
|
||||
.map(|p| {
|
||||
let joined = p
|
||||
.lines()
|
||||
.map(|l| l.trim())
|
||||
.filter(|l| !l.is_empty())
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
format!("<p>{joined}</p>")
|
||||
})
|
||||
.collect();
|
||||
|
||||
if paragraphs.is_empty() {
|
||||
String::new()
|
||||
} else {
|
||||
paragraphs.join("\n")
|
||||
}
|
||||
}
|
||||
|
||||
/// Converts authoring markup to plain text.
|
||||
///
|
||||
/// Used for CSV columns, terminal output, and any place a fragment of HTML would
|
||||
/// be noise.
|
||||
///
|
||||
/// # Arguments
|
||||
///
|
||||
/// * `src` - the authoring source.
|
||||
///
|
||||
/// # Returns
|
||||
///
|
||||
/// Plain text with markup removed and symbols rendered as Unicode.
|
||||
pub fn to_plain(src: &str) -> String {
|
||||
let symbolized = apply_symbols(src, false);
|
||||
let mut out = strip_inline(&symbolized);
|
||||
out = out
|
||||
.lines()
|
||||
.map(|l| l.trim())
|
||||
.filter(|l| !l.is_empty())
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
out.trim().to_string()
|
||||
}
|
||||
|
||||
/// Passes authoring markup through for Typst.
|
||||
///
|
||||
/// The markup is already a Typst subset, so this only normalizes whitespace and
|
||||
/// escapes the few characters Typst treats specially in content mode.
|
||||
///
|
||||
/// # Arguments
|
||||
///
|
||||
/// * `src` - the authoring source.
|
||||
///
|
||||
/// # Returns
|
||||
///
|
||||
/// Typst content-mode markup.
|
||||
pub fn to_typst(src: &str) -> String {
|
||||
let mut out = String::with_capacity(src.len());
|
||||
for ch in src.trim().chars() {
|
||||
match ch {
|
||||
// A bare `@` or `<` starts a Typst reference or label.
|
||||
'@' => out.push_str("\\@"),
|
||||
'<' => out.push_str("\\<"),
|
||||
'>' => out.push_str("\\>"),
|
||||
_ => out.push(ch),
|
||||
}
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
/// Escapes the five XML-significant characters.
|
||||
///
|
||||
/// # Arguments
|
||||
///
|
||||
/// * `s` - the text to escape.
|
||||
///
|
||||
/// # Returns
|
||||
///
|
||||
/// The escaped text.
|
||||
pub fn escape_html(s: &str) -> String {
|
||||
let mut out = String::with_capacity(s.len());
|
||||
for ch in s.chars() {
|
||||
match ch {
|
||||
'&' => out.push_str("&"),
|
||||
'<' => out.push_str("<"),
|
||||
'>' => out.push_str(">"),
|
||||
'"' => out.push_str("""),
|
||||
'\'' => out.push_str("'"),
|
||||
_ => out.push(ch),
|
||||
}
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
/// Applies the symbol table.
|
||||
///
|
||||
/// # Arguments
|
||||
///
|
||||
/// * `s` - the text.
|
||||
/// * `html` - whether to emit HTML entities rather than Unicode.
|
||||
///
|
||||
/// # Returns
|
||||
///
|
||||
/// The substituted text.
|
||||
fn apply_symbols(s: &str, html: bool) -> String {
|
||||
let mut out = s.to_string();
|
||||
for (token, entity, plain) in SYMBOLS {
|
||||
if out.contains(token) {
|
||||
out = out.replace(token, if html { entity } else { plain });
|
||||
}
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
/// Applies inline markup rules, producing HTML.
|
||||
///
|
||||
/// # Arguments
|
||||
///
|
||||
/// * `s` - escaped, symbol-substituted text.
|
||||
///
|
||||
/// # Returns
|
||||
///
|
||||
/// The text with inline markup converted.
|
||||
fn apply_inline(s: &str) -> String {
|
||||
let mut out = s.to_string();
|
||||
// Bracketed forms first: their contents may contain other markup characters.
|
||||
out = wrap_bracket(&out, "#sub[", "<sub>", "</sub>");
|
||||
out = wrap_bracket(&out, "#sup[", "<sup>", "</sup>");
|
||||
out = wrap_delimited(&out, "`", "<code>", "</code>");
|
||||
out = wrap_delimited(&out, "**", "<strong>", "</strong>");
|
||||
out = wrap_delimited(&out, "*", "<em>", "</em>");
|
||||
out = wrap_delimited(&out, "_", "<em>", "</em>");
|
||||
out
|
||||
}
|
||||
|
||||
/// Removes inline markup without replacing it.
|
||||
///
|
||||
/// # Arguments
|
||||
///
|
||||
/// * `s` - the text.
|
||||
///
|
||||
/// # Returns
|
||||
///
|
||||
/// The text with markup delimiters stripped.
|
||||
fn strip_inline(s: &str) -> String {
|
||||
let mut out = s.to_string();
|
||||
out = wrap_bracket(&out, "#sub[", "", "");
|
||||
out = wrap_bracket(&out, "#sup[", "", "");
|
||||
out = wrap_delimited(&out, "`", "", "");
|
||||
out = wrap_delimited(&out, "**", "", "");
|
||||
out = wrap_delimited(&out, "*", "", "");
|
||||
out = wrap_delimited(&out, "_", "", "");
|
||||
out
|
||||
}
|
||||
|
||||
/// Replaces `open...]` spans with wrapped content.
|
||||
///
|
||||
/// # Arguments
|
||||
///
|
||||
/// * `s` - the text.
|
||||
/// * `open` - the opening token, e.g. `"#sub["`.
|
||||
/// * `pre` - text to emit before the content.
|
||||
/// * `post` - text to emit after the content.
|
||||
///
|
||||
/// # Returns
|
||||
///
|
||||
/// The rewritten text. Unclosed spans are left alone.
|
||||
fn wrap_bracket(s: &str, open: &str, pre: &str, post: &str) -> String {
|
||||
let mut out = String::with_capacity(s.len());
|
||||
let mut rest = s;
|
||||
loop {
|
||||
match rest.find(open) {
|
||||
None => {
|
||||
out.push_str(rest);
|
||||
return out;
|
||||
}
|
||||
Some(i) => {
|
||||
let after = &rest[i + open.len()..];
|
||||
match after.find(']') {
|
||||
None => {
|
||||
out.push_str(rest);
|
||||
return out;
|
||||
}
|
||||
Some(j) => {
|
||||
out.push_str(&rest[..i]);
|
||||
out.push_str(pre);
|
||||
out.push_str(&after[..j]);
|
||||
out.push_str(post);
|
||||
rest = &after[j + 1..];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Replaces paired `delim...delim` spans with wrapped content.
|
||||
///
|
||||
/// A delimiter with no partner is emitted literally, so an apostrophe-heavy stem
|
||||
/// or a lone asterisk does not swallow the rest of the text.
|
||||
///
|
||||
/// # Arguments
|
||||
///
|
||||
/// * `s` - the text.
|
||||
/// * `delim` - the delimiter, e.g. `"**"`.
|
||||
/// * `pre` - text to emit before the content.
|
||||
/// * `post` - text to emit after the content.
|
||||
///
|
||||
/// # Returns
|
||||
///
|
||||
/// The rewritten text.
|
||||
fn wrap_delimited(s: &str, delim: &str, pre: &str, post: &str) -> String {
|
||||
let mut out = String::with_capacity(s.len());
|
||||
let mut rest = s;
|
||||
loop {
|
||||
match rest.find(delim) {
|
||||
None => {
|
||||
out.push_str(rest);
|
||||
return out;
|
||||
}
|
||||
Some(i) => {
|
||||
let after = &rest[i + delim.len()..];
|
||||
match after.find(delim) {
|
||||
None => {
|
||||
out.push_str(rest);
|
||||
return out;
|
||||
}
|
||||
Some(0) => {
|
||||
// Empty span such as `**`; emit literally and move on.
|
||||
out.push_str(&rest[..i + delim.len()]);
|
||||
rest = after;
|
||||
}
|
||||
Some(j) => {
|
||||
out.push_str(&rest[..i]);
|
||||
out.push_str(pre);
|
||||
out.push_str(&after[..j]);
|
||||
out.push_str(post);
|
||||
rest = &after[j + delim.len()..];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn escapes_before_substituting() {
|
||||
// The entity produced by the symbol table must survive escaping.
|
||||
assert_eq!(to_html("a #sym.arrow.r b"), "<p>a → b</p>");
|
||||
// A literal ampersand is escaped.
|
||||
assert_eq!(to_html("Tris & HCl"), "<p>Tris & HCl</p>");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn refuses_to_pass_through_html() {
|
||||
let out = to_html("<script>alert(1)</script>");
|
||||
assert!(!out.contains("<script>"));
|
||||
assert!(out.contains("<script>"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn converts_inline_markup() {
|
||||
assert_eq!(to_html("**bold**"), "<p><strong>bold</strong></p>");
|
||||
assert_eq!(to_html("*em*"), "<p><em>em</em></p>");
|
||||
assert_eq!(to_html("`code`"), "<p><code>code</code></p>");
|
||||
assert_eq!(to_html("H#sub[2]O"), "<p>H<sub>2</sub>O</p>");
|
||||
assert_eq!(to_html("x#sup[2]"), "<p>x<sup>2</sup></p>");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn bold_wins_over_italic() {
|
||||
assert_eq!(
|
||||
to_html("**strong** and *weak*"),
|
||||
"<p><strong>strong</strong> and <em>weak</em></p>"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unpaired_delimiters_are_literal() {
|
||||
assert_eq!(to_html("2 * 3 = 6"), "<p>2 * 3 = 6</p>");
|
||||
assert_eq!(to_html("a_b"), "<p>a_b</p>");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn splits_paragraphs_and_joins_wrapped_lines() {
|
||||
let out = to_html("first line\ncontinued\n\nsecond paragraph");
|
||||
assert_eq!(out, "<p>first line continued</p>\n<p>second paragraph</p>");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn empty_input_yields_empty_output() {
|
||||
assert_eq!(to_html(" \n "), "");
|
||||
assert_eq!(to_plain(""), "");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn plain_text_uses_unicode_and_drops_markup() {
|
||||
assert_eq!(to_plain("K#sub[m] #sym.approx 5 mM"), "Km \u{2248} 5 mM");
|
||||
assert_eq!(to_plain("**bold** text"), "bold text");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn typst_escapes_reference_starters() {
|
||||
assert_eq!(to_typst("a @ b"), "a \\@ b");
|
||||
assert_eq!(to_typst("x < y"), "x \\< y");
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user