829 lines
27 KiB
Rust
829 lines
27 KiB
Rust
// SPDX-License-Identifier: Prosperity-3.0.0
|
|
// Copyright Scientific Computing Studio
|
|
// Source: https://git.scient.ing/education/coursebank
|
|
|
|
//! Converting the authoring markup into HTML, plain text, and Typst.
|
|
//!
|
|
//! Stems are written in a small markup that is a subset of Typst with a few
|
|
//! Markdown conveniences, because chemistry and biology questions need
|
|
//! subscripts, arrows, and Greek letters, and typing HTML entities into YAML by
|
|
//! hand is miserable.
|
|
//!
|
|
//! There is no regular expression engine here. Every rule is a scan, which keeps
|
|
//! the dependency list short and makes the escaping order explicit: HTML is
|
|
//! escaped *first*, then symbol substitutions run, so that a substitution
|
|
//! producing `→` is not itself escaped into `→`.
|
|
|
|
/// Symbol substitutions, applied after HTML escaping.
|
|
///
|
|
/// Ordered longest-first within each family so `#sym.arrow.r` is not consumed by
|
|
/// a shorter prefix.
|
|
const SYMBOLS: &[(&str, &str, &str)] = &[
|
|
// (source token, html, plain text)
|
|
("#sym.gt.eq", "≥", "\u{2265}"),
|
|
("#sym.lt.eq", "≤", "\u{2264}"),
|
|
("#sym.eq.not", "≠", "\u{2260}"),
|
|
("#sym.plus.minus", "±", "\u{00b1}"),
|
|
("#sym.arrow.r", "→", "\u{2192}"),
|
|
("#sym.arrow.l", "←", "\u{2190}"),
|
|
("#sym.arrow.lr", "↔", "\u{2194}"),
|
|
("#sym.rightarrow", "→", "\u{2192}"),
|
|
("#sym.leftarrow", "←", "\u{2190}"),
|
|
("#sym.times", "×", "\u{00d7}"),
|
|
("#sym.dot", "·", "\u{00b7}"),
|
|
("#sym.degree", "°", "\u{00b0}"),
|
|
("#sym.infinity", "∞", "\u{221e}"),
|
|
("#sym.approx", "≈", "\u{2248}"),
|
|
("#sym.alpha", "α", "\u{03b1}"),
|
|
("#sym.beta", "β", "\u{03b2}"),
|
|
("#sym.gamma", "γ", "\u{03b3}"),
|
|
("#sym.delta.cap", "Δ", "\u{0394}"),
|
|
("#sym.delta", "δ", "\u{03b4}"),
|
|
("#sym.epsilon", "ε", "\u{03b5}"),
|
|
("#sym.lambda", "λ", "\u{03bb}"),
|
|
("#sym.mu", "μ", "\u{03bc}"),
|
|
("#sym.pi", "π", "\u{03c0}"),
|
|
("#sym.sigma", "σ", "\u{03c3}"),
|
|
("#sym.tau", "τ", "\u{03c4}"),
|
|
("#sym.phi", "φ", "\u{03c6}"),
|
|
("#sym.omega", "ω", "\u{03c9}"),
|
|
];
|
|
|
|
/// Converts authoring markup to an HTML fragment.
|
|
///
|
|
/// Handles paragraphs, bold, italic, inline code, subscripts, superscripts, and
|
|
/// the symbol table. Anything unrecognized passes through escaped, so a stray
|
|
/// `<script>` in a stem cannot become markup in a Canvas quiz.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `src` - the authoring source.
|
|
///
|
|
/// # Returns
|
|
///
|
|
/// An HTML fragment, with each paragraph wrapped in `<p>`.
|
|
pub fn to_html(src: &str) -> String {
|
|
let escaped = escape_html(src);
|
|
let symbolized = apply_symbols(&escaped, true);
|
|
let inline = apply_inline(&symbolized);
|
|
|
|
let paragraphs: Vec<String> = inline
|
|
.split("\n\n")
|
|
.map(|p| p.trim())
|
|
.filter(|p| !p.is_empty())
|
|
.map(|p| {
|
|
let joined = p
|
|
.lines()
|
|
.map(|l| l.trim())
|
|
.filter(|l| !l.is_empty())
|
|
.collect::<Vec<_>>()
|
|
.join(" ");
|
|
format!("<p>{joined}</p>")
|
|
})
|
|
.collect();
|
|
|
|
if paragraphs.is_empty() {
|
|
String::new()
|
|
} else {
|
|
paragraphs.join("\n")
|
|
}
|
|
}
|
|
|
|
/// Converts authoring markup to plain text.
|
|
///
|
|
/// Used for CSV columns, terminal output, and any place a fragment of HTML would
|
|
/// be noise.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `src` - the authoring source.
|
|
///
|
|
/// # Returns
|
|
///
|
|
/// Plain text with markup removed and symbols rendered as Unicode.
|
|
pub fn to_plain(src: &str) -> String {
|
|
let symbolized = apply_symbols(src, false);
|
|
let mut out = strip_inline(&symbolized);
|
|
out = out
|
|
.lines()
|
|
.map(|l| l.trim())
|
|
.filter(|l| !l.is_empty())
|
|
.collect::<Vec<_>>()
|
|
.join(" ");
|
|
out.trim().to_string()
|
|
}
|
|
|
|
/// Converts authoring markup to Pandoc-flavoured Markdown, for a Quarto document.
|
|
///
|
|
/// Subscripts and superscripts become Pandoc's `~x~` and `^x^`, the symbol table
|
|
/// renders as Unicode, and the bold, italic, and inline-code spans are already
|
|
/// Markdown, so they pass through unchanged. Paragraph breaks are kept. Nothing is
|
|
/// HTML-escaped, because the consumer is a Markdown renderer rather than a page.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `src` - the authoring source.
|
|
///
|
|
/// # Returns
|
|
///
|
|
/// Pandoc Markdown, trimmed, with paragraph breaks preserved.
|
|
pub fn to_markdown(src: &str) -> String {
|
|
let symbolized = apply_symbols(src, false);
|
|
let mut out = wrap_bracket(&symbolized, "#sub[", "~", "~");
|
|
out = wrap_bracket(&out, "#sup[", "^", "^");
|
|
out.lines()
|
|
.map(|l| l.trim_end())
|
|
.collect::<Vec<_>>()
|
|
.join("\n")
|
|
.trim()
|
|
.to_string()
|
|
}
|
|
|
|
/// Passes authoring markup through for Typst, translating inline LaTeX math on
|
|
/// the way.
|
|
///
|
|
/// Outside math this only escapes the characters Typst treats specially in
|
|
/// content mode: a bare `@` or `<` starts a reference or label.
|
|
///
|
|
/// Math is different. Authors write ordinary LaTeX between `$...$`, and Typst's
|
|
/// own math grammar is not LaTeX's — a backslash escapes the next character
|
|
/// rather than naming a symbol, so `\Delta`, `\times`, `\ln` compile without
|
|
/// error and print wrong. Each `$...$` span is instead handed whole to
|
|
/// mitex's `mi`, which parses LaTeX grammar on purpose: `$\phi$` becomes
|
|
/// `#mi("\\phi")`. mhchem's `\ce{...}` is the one exception, and goes to
|
|
/// whalogen's `ce` instead, via `push_math_span`. The `@`/`<`/`>` escaping
|
|
/// above is skipped for anything
|
|
/// inside the span, since it reaches Typst as a string argument, not as
|
|
/// markup — escaping `<` there would corrupt the LaTeX rather than protect
|
|
/// anything.
|
|
///
|
|
/// `\$` is left alone, matching LaTeX's own convention for a literal dollar
|
|
/// sign. A `$` with no matching close is escaped the same way rather than left
|
|
/// to open Typst's own math mode on a stray price or a malformed source line.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `src` - the authoring source.
|
|
///
|
|
/// # Returns
|
|
///
|
|
/// Typst content-mode markup.
|
|
pub fn to_typst(src: &str) -> String {
|
|
let src = src.trim();
|
|
let chars: Vec<(usize, char)> = src.char_indices().collect();
|
|
let mut out = String::with_capacity(src.len());
|
|
let mut i = 0;
|
|
while i < chars.len() {
|
|
let (_, c) = chars[i];
|
|
|
|
// A `\ce{...}` written outside math is still chemistry and still cannot
|
|
// reach Typst as a backslash: `\c` is an escape in content mode. Checked
|
|
// before the escaped-pair rule below, which would otherwise copy `\c`
|
|
// through and leave `e{...}` as literal text. The cheap two-character
|
|
// guard keeps a stem full of `\Delta` and `\times` from rescanning the
|
|
// remainder at every backslash.
|
|
if c == '\\'
|
|
&& chars.get(i + 1).map(|c| c.1) == Some('c')
|
|
&& chars.get(i + 2).map(|c| c.1) == Some('e')
|
|
{
|
|
if let Some(span) = find_ce(&src[chars[i].0..]) {
|
|
if span.start == 0 {
|
|
let at = chars[i].0;
|
|
push_ce(&mut out, &src[at + span.arg.0..at + span.arg.1]);
|
|
i = char_index(&chars, at + span.end);
|
|
continue;
|
|
}
|
|
}
|
|
}
|
|
|
|
// An escaped pair is copied verbatim and never reconsidered, so `\$`
|
|
// can't be mistaken for the start of math and an `\@`/`\<`/`\>` an
|
|
// author already wrote is not escaped a second time.
|
|
if c == '\\' && i + 1 < chars.len() {
|
|
out.push('\\');
|
|
out.push(chars[i + 1].1);
|
|
i += 2;
|
|
continue;
|
|
}
|
|
|
|
if c == '$' {
|
|
match find_math_close(&chars, i) {
|
|
Some(close) => {
|
|
let start = chars[i + 1].0;
|
|
let end = chars[close].0;
|
|
push_math_span(&mut out, &src[start..end]);
|
|
i = close + 1;
|
|
continue;
|
|
}
|
|
None => {
|
|
out.push_str("\\$");
|
|
i += 1;
|
|
continue;
|
|
}
|
|
}
|
|
}
|
|
|
|
match c {
|
|
'@' => out.push_str("\\@"),
|
|
'<' => out.push_str("\\<"),
|
|
'>' => out.push_str("\\>"),
|
|
_ => out.push(c),
|
|
}
|
|
i += 1;
|
|
}
|
|
out
|
|
}
|
|
|
|
/// Finds the index into `chars` of the `$` matching the opener at `open`. A
|
|
/// backslash-escaped pair is skipped as a unit, so a `\$` inside the math span
|
|
/// doesn't close it early.
|
|
fn find_math_close(chars: &[(usize, char)], open: usize) -> Option<usize> {
|
|
let mut j = open + 1;
|
|
while j < chars.len() {
|
|
match chars[j].1 {
|
|
'\\' if j + 1 < chars.len() => j += 2,
|
|
'$' => return Some(j),
|
|
_ => j += 1,
|
|
}
|
|
}
|
|
None
|
|
}
|
|
|
|
/// The import line a template needs before it can receive a `#ce(...)` call.
|
|
///
|
|
/// mitex renders the LaTeX in a stem, but it ships no package support, so
|
|
/// mhchem's `\ce` is simply an unknown command to it. whalogen is a Typst port
|
|
/// of mhchem, and `ce` is the function this module emits chemistry as.
|
|
pub const CHEM_IMPORT: &str = "#import \"@preview/whalogen:0.3.0\": ce";
|
|
|
|
/// Whether `body` needs [`CHEM_IMPORT`] and `template` does not provide it.
|
|
///
|
|
/// A template is checked for the package name rather than for the exact import
|
|
/// line, so a template that pins a different whalogen version, imports the
|
|
/// package under an alias, or defines its own `ce` is left alone.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `template` - the template source the body will be injected into.
|
|
/// * `body` - the generated Typst source.
|
|
///
|
|
/// # Returns
|
|
///
|
|
/// Whether the pair would fail to compile for want of the import.
|
|
pub fn needs_chem_import(template: &str, body: &str) -> bool {
|
|
body.contains("#ce(") && !template.contains("whalogen") && !template.contains("let ce")
|
|
}
|
|
|
|
/// A `\ce{...}` command located in a source fragment, as byte offsets from the
|
|
/// start of that fragment.
|
|
struct CeSpan {
|
|
/// Offset of the backslash.
|
|
start: usize,
|
|
/// Offset just past the closing brace.
|
|
end: usize,
|
|
/// The argument, without its braces.
|
|
arg: (usize, usize),
|
|
}
|
|
|
|
/// Finds the first `\ce{...}` in `s`.
|
|
///
|
|
/// Three details matter, and all three come from the same place: the scan has to
|
|
/// agree with what LaTeX itself would read.
|
|
///
|
|
/// A backslash-escaped pair is skipped as a unit, so the line break `\\`
|
|
/// followed by the letters `ce` is not read as the command. `$a \\ ce{x}$` is a
|
|
/// break and then literal text; mhchem's own command is a single backslash.
|
|
///
|
|
/// The name must end at the `e`, so `\cellcolor` and `\century` are left for
|
|
/// mitex rather than half-consumed here.
|
|
///
|
|
/// The argument is matched on brace depth rather than on the first `}`, so
|
|
/// `\ce{Fe^{2+}}` keeps its superscript. An unbalanced argument yields `None`:
|
|
/// there is nothing safe to convert, and mitex's own error names the line.
|
|
fn find_ce(s: &str) -> Option<CeSpan> {
|
|
let chars: Vec<(usize, char)> = s.char_indices().collect();
|
|
let mut i = 0;
|
|
while i < chars.len() {
|
|
if chars[i].1 != '\\' {
|
|
i += 1;
|
|
continue;
|
|
}
|
|
if matches!(chars.get(i + 1), Some((_, '\\'))) {
|
|
i += 2;
|
|
continue;
|
|
}
|
|
let named = chars.get(i + 1).map(|c| c.1) == Some('c')
|
|
&& chars.get(i + 2).map(|c| c.1) == Some('e')
|
|
&& !matches!(chars.get(i + 3), Some((_, c)) if c.is_ascii_alphabetic());
|
|
if !named {
|
|
i += 1;
|
|
continue;
|
|
}
|
|
|
|
// LaTeX allows whitespace between a control word and its argument.
|
|
let mut open = i + 3;
|
|
while matches!(chars.get(open), Some((_, c)) if c.is_whitespace()) {
|
|
open += 1;
|
|
}
|
|
if !matches!(chars.get(open), Some((_, '{'))) {
|
|
i += 1;
|
|
continue;
|
|
}
|
|
|
|
let mut depth = 1usize;
|
|
let mut k = open + 1;
|
|
while k < chars.len() {
|
|
match chars[k].1 {
|
|
'\\' => k += 2,
|
|
'{' => {
|
|
depth += 1;
|
|
k += 1;
|
|
}
|
|
'}' => {
|
|
depth -= 1;
|
|
if depth == 0 {
|
|
return Some(CeSpan {
|
|
start: chars[i].0,
|
|
end: chars[k].0 + 1,
|
|
arg: (chars[open].0 + 1, chars[k].0),
|
|
});
|
|
}
|
|
k += 1;
|
|
}
|
|
_ => k += 1,
|
|
}
|
|
}
|
|
return None;
|
|
}
|
|
None
|
|
}
|
|
|
|
/// The index into `chars` of the entry at byte offset `byte`, or the length when
|
|
/// the offset is past the end.
|
|
fn char_index(chars: &[(usize, char)], byte: usize) -> usize {
|
|
chars
|
|
.iter()
|
|
.position(|(at, _)| *at >= byte)
|
|
.unwrap_or(chars.len())
|
|
}
|
|
|
|
/// Emits one `$...$` span as Typst content.
|
|
///
|
|
/// Most of a span goes to mitex's `mi`, which parses LaTeX on purpose. The
|
|
/// exception is mhchem: `\ce{...}` is not a LaTeX primitive but a package
|
|
/// command with its own character-level grammar, and mitex implements no
|
|
/// packages, so the plugin aborts the whole document with `unknown command:
|
|
/// \ce` rather than degrading to something printable. Those runs are handed to
|
|
/// whalogen's `ce` instead and the rest of the span still goes to `mi`, so an
|
|
/// equation that mixes chemistry with ordinary math renders both.
|
|
///
|
|
/// A span with no chemistry in it produces exactly one `mi` call, as it always
|
|
/// has.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `out` - the buffer to append to.
|
|
/// * `latex` - the LaTeX between the dollar signs.
|
|
fn push_math_span(out: &mut String, latex: &str) {
|
|
if find_ce(latex).is_none() {
|
|
push_mi(out, latex);
|
|
return;
|
|
}
|
|
|
|
let mut rest = latex;
|
|
while let Some(span) = find_ce(rest) {
|
|
push_run(out, &rest[..span.start]);
|
|
push_ce(out, &rest[span.arg.0..span.arg.1]);
|
|
rest = &rest[span.end..];
|
|
}
|
|
push_run(out, rest);
|
|
}
|
|
|
|
/// Emits a non-chemistry run of a math span, dropping an empty one and keeping a
|
|
/// whitespace-only one as the single space it separates two calls with.
|
|
fn push_run(out: &mut String, latex: &str) {
|
|
if latex.is_empty() {
|
|
return;
|
|
}
|
|
if latex.trim().is_empty() {
|
|
out.push(' ');
|
|
return;
|
|
}
|
|
push_mi(out, latex);
|
|
}
|
|
|
|
/// Emits a `#mi(...)` call wrapping LaTeX math.
|
|
fn push_mi(out: &mut String, latex: &str) {
|
|
out.push_str("#mi(");
|
|
push_typst_string(out, latex);
|
|
out.push(')');
|
|
}
|
|
|
|
/// Emits a `#ce(...)` call wrapping an mhchem argument.
|
|
///
|
|
/// The argument is passed through as written. whalogen reads the same formula,
|
|
/// charge, bond, and arrow syntax mhchem does, so `H2O`, `<=>`, `[AgCl2]-`, and
|
|
/// `->[H2O]` need no translation. Its isotope and oxidation-number spellings do
|
|
/// differ (`@Th,227,90@` against mhchem's `^{227}_{90}Th`), and so do `\pu` and
|
|
/// `\bond`, which whalogen has no equivalent for. Those print oddly rather than
|
|
/// failing the build, so a bank that uses them needs its own pass.
|
|
fn push_ce(out: &mut String, argument: &str) {
|
|
out.push_str("#ce(");
|
|
push_typst_string(out, argument.trim());
|
|
out.push(')');
|
|
}
|
|
|
|
/// Writes `s` as a quoted Typst string. Kept local, duplicating the five-case
|
|
/// match in `typst::value::write_string`, rather than reaching into the
|
|
/// Typst-specific value writer for one small helper.
|
|
fn push_typst_string(out: &mut String, s: &str) {
|
|
out.push('"');
|
|
for ch in s.chars() {
|
|
match ch {
|
|
'"' => out.push_str("\\\""),
|
|
'\\' => out.push_str("\\\\"),
|
|
'\n' => out.push_str("\\n"),
|
|
'\r' => out.push_str("\\r"),
|
|
'\t' => out.push_str("\\t"),
|
|
_ => out.push(ch),
|
|
}
|
|
}
|
|
out.push('"');
|
|
}
|
|
|
|
/// Escapes the five XML-significant characters.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `s` - the text to escape.
|
|
///
|
|
/// # Returns
|
|
///
|
|
/// The escaped text.
|
|
pub fn escape_html(s: &str) -> String {
|
|
let mut out = String::with_capacity(s.len());
|
|
for ch in s.chars() {
|
|
match ch {
|
|
'&' => out.push_str("&"),
|
|
'<' => out.push_str("<"),
|
|
'>' => out.push_str(">"),
|
|
'"' => out.push_str("""),
|
|
'\'' => out.push_str("'"),
|
|
_ => out.push(ch),
|
|
}
|
|
}
|
|
out
|
|
}
|
|
|
|
/// Applies the symbol table.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `s` - the text.
|
|
/// * `html` - whether to emit HTML entities rather than Unicode.
|
|
///
|
|
/// # Returns
|
|
///
|
|
/// The substituted text.
|
|
pub(crate) fn apply_symbols(s: &str, html: bool) -> String {
|
|
let mut out = s.to_string();
|
|
for (token, entity, plain) in SYMBOLS {
|
|
if out.contains(token) {
|
|
out = out.replace(token, if html { entity } else { plain });
|
|
}
|
|
}
|
|
out
|
|
}
|
|
|
|
/// Applies inline markup rules, producing HTML.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `s` - escaped, symbol-substituted text.
|
|
///
|
|
/// # Returns
|
|
///
|
|
/// The text with inline markup converted.
|
|
pub(crate) fn apply_inline(s: &str) -> String {
|
|
let mut out = s.to_string();
|
|
// Bracketed forms first: their contents may contain other markup characters.
|
|
out = wrap_bracket(&out, "#sub[", "<sub>", "</sub>");
|
|
out = wrap_bracket(&out, "#sup[", "<sup>", "</sup>");
|
|
out = wrap_delimited(&out, "`", "<code>", "</code>");
|
|
out = wrap_delimited(&out, "**", "<strong>", "</strong>");
|
|
out = wrap_delimited(&out, "*", "<em>", "</em>");
|
|
out = wrap_delimited(&out, "_", "<em>", "</em>");
|
|
out
|
|
}
|
|
|
|
/// Removes inline markup without replacing it.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `s` - the text.
|
|
///
|
|
/// # Returns
|
|
///
|
|
/// The text with markup delimiters stripped.
|
|
fn strip_inline(s: &str) -> String {
|
|
let mut out = s.to_string();
|
|
out = wrap_bracket(&out, "#sub[", "", "");
|
|
out = wrap_bracket(&out, "#sup[", "", "");
|
|
out = wrap_delimited(&out, "`", "", "");
|
|
out = wrap_delimited(&out, "**", "", "");
|
|
out = wrap_delimited(&out, "*", "", "");
|
|
out = wrap_delimited(&out, "_", "", "");
|
|
out
|
|
}
|
|
|
|
/// Replaces `open...]` spans with wrapped content.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `s` - the text.
|
|
/// * `open` - the opening token, e.g. `"#sub["`.
|
|
/// * `pre` - text to emit before the content.
|
|
/// * `post` - text to emit after the content.
|
|
///
|
|
/// # Returns
|
|
///
|
|
/// The rewritten text. Unclosed spans are left alone.
|
|
fn wrap_bracket(s: &str, open: &str, pre: &str, post: &str) -> String {
|
|
let mut out = String::with_capacity(s.len());
|
|
let mut rest = s;
|
|
loop {
|
|
match rest.find(open) {
|
|
None => {
|
|
out.push_str(rest);
|
|
return out;
|
|
}
|
|
Some(i) => {
|
|
let after = &rest[i + open.len()..];
|
|
match after.find(']') {
|
|
None => {
|
|
out.push_str(rest);
|
|
return out;
|
|
}
|
|
Some(j) => {
|
|
out.push_str(&rest[..i]);
|
|
out.push_str(pre);
|
|
out.push_str(&after[..j]);
|
|
out.push_str(post);
|
|
rest = &after[j + 1..];
|
|
}
|
|
}
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Replaces paired `delim...delim` spans with wrapped content.
|
|
///
|
|
/// A delimiter with no partner is emitted literally, so an apostrophe-heavy stem
|
|
/// or a lone asterisk does not swallow the rest of the text.
|
|
///
|
|
/// # Arguments
|
|
///
|
|
/// * `s` - the text.
|
|
/// * `delim` - the delimiter, e.g. `"**"`.
|
|
/// * `pre` - text to emit before the content.
|
|
/// * `post` - text to emit after the content.
|
|
///
|
|
/// # Returns
|
|
///
|
|
/// The rewritten text.
|
|
fn wrap_delimited(s: &str, delim: &str, pre: &str, post: &str) -> String {
|
|
let mut out = String::with_capacity(s.len());
|
|
let mut rest = s;
|
|
loop {
|
|
match rest.find(delim) {
|
|
None => {
|
|
out.push_str(rest);
|
|
return out;
|
|
}
|
|
Some(i) => {
|
|
let after = &rest[i + delim.len()..];
|
|
match after.find(delim) {
|
|
None => {
|
|
out.push_str(rest);
|
|
return out;
|
|
}
|
|
Some(0) => {
|
|
// Empty span such as `**`; emit literally and move on.
|
|
out.push_str(&rest[..i + delim.len()]);
|
|
rest = after;
|
|
}
|
|
Some(j) => {
|
|
out.push_str(&rest[..i]);
|
|
out.push_str(pre);
|
|
out.push_str(&after[..j]);
|
|
out.push_str(post);
|
|
rest = &after[j + delim.len()..];
|
|
}
|
|
}
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
#[cfg(test)]
|
|
mod tests {
|
|
use super::*;
|
|
|
|
#[test]
|
|
fn escapes_before_substituting() {
|
|
// The entity produced by the symbol table must survive escaping.
|
|
assert_eq!(to_html("a #sym.arrow.r b"), "<p>a → b</p>");
|
|
// A literal ampersand is escaped.
|
|
assert_eq!(to_html("Tris & HCl"), "<p>Tris & HCl</p>");
|
|
}
|
|
|
|
#[test]
|
|
fn refuses_to_pass_through_html() {
|
|
let out = to_html("<script>alert(1)</script>");
|
|
assert!(!out.contains("<script>"));
|
|
assert!(out.contains("<script>"));
|
|
}
|
|
|
|
#[test]
|
|
fn converts_inline_markup() {
|
|
assert_eq!(to_html("**bold**"), "<p><strong>bold</strong></p>");
|
|
assert_eq!(to_html("*em*"), "<p><em>em</em></p>");
|
|
assert_eq!(to_html("`code`"), "<p><code>code</code></p>");
|
|
assert_eq!(to_html("H#sub[2]O"), "<p>H<sub>2</sub>O</p>");
|
|
assert_eq!(to_html("x#sup[2]"), "<p>x<sup>2</sup></p>");
|
|
}
|
|
|
|
#[test]
|
|
fn bold_wins_over_italic() {
|
|
assert_eq!(
|
|
to_html("**strong** and *weak*"),
|
|
"<p><strong>strong</strong> and <em>weak</em></p>"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn unpaired_delimiters_are_literal() {
|
|
assert_eq!(to_html("2 * 3 = 6"), "<p>2 * 3 = 6</p>");
|
|
assert_eq!(to_html("a_b"), "<p>a_b</p>");
|
|
}
|
|
|
|
#[test]
|
|
fn splits_paragraphs_and_joins_wrapped_lines() {
|
|
let out = to_html("first line\ncontinued\n\nsecond paragraph");
|
|
assert_eq!(out, "<p>first line continued</p>\n<p>second paragraph</p>");
|
|
}
|
|
|
|
#[test]
|
|
fn empty_input_yields_empty_output() {
|
|
assert_eq!(to_html(" \n "), "");
|
|
assert_eq!(to_plain(""), "");
|
|
}
|
|
|
|
#[test]
|
|
fn plain_text_uses_unicode_and_drops_markup() {
|
|
assert_eq!(to_plain("K#sub[m] #sym.approx 5 mM"), "Km \u{2248} 5 mM");
|
|
assert_eq!(to_plain("**bold** text"), "bold text");
|
|
}
|
|
|
|
#[test]
|
|
fn typst_escapes_reference_starters() {
|
|
assert_eq!(to_typst("a @ b"), "a \\@ b");
|
|
assert_eq!(to_typst("x < y"), "x \\< y");
|
|
}
|
|
|
|
#[test]
|
|
fn markdown_uses_pandoc_scripts_and_unicode_symbols() {
|
|
assert_eq!(to_markdown("H#sub[2]O"), "H~2~O");
|
|
assert_eq!(to_markdown("x#sup[2]"), "x^2^");
|
|
assert_eq!(
|
|
to_markdown("K#sub[m] #sym.approx 5 mM"),
|
|
"K~m~ \u{2248} 5 mM"
|
|
);
|
|
// Bold, italic, and code are already Markdown.
|
|
assert_eq!(
|
|
to_markdown("**bold** and *em* and `code`"),
|
|
"**bold** and *em* and `code`"
|
|
);
|
|
// Paragraph breaks survive.
|
|
assert_eq!(to_markdown("one\n\ntwo"), "one\n\ntwo");
|
|
}
|
|
}
|
|
|
|
#[test]
|
|
fn latex_math_becomes_a_mitex_call() {
|
|
assert_eq!(
|
|
to_typst("angle $\\phi$ (phi)"),
|
|
"angle #mi(\"\\\\phi\") (phi)"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn comparison_operators_inside_math_are_not_escaped() {
|
|
assert_eq!(to_typst("$\\Delta H < 0$"), "#mi(\"\\\\Delta H < 0\")");
|
|
}
|
|
|
|
#[test]
|
|
fn reference_starters_outside_math_are_still_escaped() {
|
|
assert_eq!(to_typst("see @fig:x and x < y"), "see \\@fig:x and x \\< y");
|
|
}
|
|
|
|
#[test]
|
|
fn escaped_and_unmatched_dollar_signs_are_left_or_escaped() {
|
|
assert_eq!(to_typst("costs \\$5 total"), "costs \\$5 total");
|
|
assert_eq!(to_typst("just $5"), "just \\$5");
|
|
}
|
|
|
|
#[test]
|
|
fn mhchem_goes_to_whalogen_rather_than_mitex() {
|
|
// The regression: mitex implements no LaTeX packages, so `\ce` reached the
|
|
// plugin as an unknown command and failed the whole document rather than
|
|
// printing badly.
|
|
assert_eq!(to_typst("$\\ce{H2O}$"), "#ce(\"H2O\")");
|
|
assert_eq!(
|
|
to_typst("water ionizes, $\\ce{H2O <=> H+ + OH-}$, giving"),
|
|
"water ionizes, #ce(\"H2O <=> H+ + OH-\"), giving"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn a_span_can_mix_chemistry_with_ordinary_math() {
|
|
// The chemistry leaves the span and the rest of it still reaches mitex, so
|
|
// an equation that needs both renders both.
|
|
assert_eq!(to_typst("$K_w = \\ce{H2O}$"), "#mi(\"K_w = \")#ce(\"H2O\")");
|
|
// The space between the two runs stays inside the `mi` call rather than
|
|
// becoming content-mode whitespace, so the spacing is TeX's to decide.
|
|
assert_eq!(
|
|
to_typst("$\\ce{H2O} \\to \\Delta H$"),
|
|
"#ce(\"H2O\")#mi(\" \\\\to \\\\Delta H\")"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn mhchem_arguments_keep_their_nested_braces() {
|
|
// Matched on brace depth, not on the first `}`, or the charge is orphaned
|
|
// and the remaining `}` closes the `#ce(` call early.
|
|
assert_eq!(to_typst("$\\ce{Fe^{2+}}$"), "#ce(\"Fe^{2+}\")");
|
|
assert_eq!(
|
|
to_typst("$\\ce{SO4^{2-} + Ba^{2+}}$"),
|
|
"#ce(\"SO4^{2-} + Ba^{2+}\")"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn a_latex_line_break_is_not_read_as_the_chemistry_command() {
|
|
// `\\` is an escaped backslash followed by the letters `ce`, not `\ce`.
|
|
// Scanning that skips escaped pairs as a unit keeps the two apart; scanning
|
|
// that does not would convert a line break into a formula.
|
|
assert_eq!(to_typst("$a \\\\ce{x}$"), "#mi(\"a \\\\\\\\ce{x}\")");
|
|
}
|
|
|
|
#[test]
|
|
fn commands_that_merely_start_with_ce_are_left_to_mitex() {
|
|
// The name has to end at the `e`, or `\cellcolor` is half-consumed and the
|
|
// conversion invents a formula out of its argument.
|
|
assert_eq!(
|
|
to_typst("$\\cellcolor{red} x$"),
|
|
"#mi(\"\\\\cellcolor{red} x\")"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn unbalanced_chemistry_is_left_for_mitex_to_report() {
|
|
// Nothing safe to convert. mitex's own error names the file and line, which
|
|
// is more useful than a silently truncated formula.
|
|
assert_eq!(to_typst("$\\ce{H2O$"), "#mi(\"\\\\ce{H2O\")");
|
|
}
|
|
|
|
#[test]
|
|
fn chemistry_outside_math_is_converted_too() {
|
|
// A bare `\ce` never reaches mitex at all, and `\c` is an escape in Typst
|
|
// content mode, so leaving it alone produces a document that either fails
|
|
// or prints the letters.
|
|
assert_eq!(
|
|
to_typst("the backbone \\ce{-NH} group"),
|
|
"the backbone #ce(\"-NH\") group"
|
|
);
|
|
}
|
|
|
|
#[test]
|
|
fn a_template_without_the_chemistry_import_is_named() {
|
|
let body = "#question((stem: [#ce(\"H2O\")]))";
|
|
assert!(needs_chem_import(
|
|
"#import \"@preview/mitex:0.2.7\": mi",
|
|
body
|
|
));
|
|
// An import of any whalogen version, or a template's own `ce`, is enough.
|
|
assert!(!needs_chem_import(CHEM_IMPORT, body));
|
|
assert!(!needs_chem_import(
|
|
"#import \"@preview/whalogen:0.2.0\": ce as ce",
|
|
body
|
|
));
|
|
assert!(!needs_chem_import("#let ce(f) = f", body));
|
|
// No chemistry in the body, nothing to warn about.
|
|
assert!(!needs_chem_import(
|
|
"#import \"@preview/mitex:0.2.7\": mi",
|
|
"#mi(\"x\")"
|
|
));
|
|
}
|