tests/oracle.rs
596 lines · 22216 bytes
1//! The `emacs --batch` ground-truth oracle (spec §5, Phase 0).
2//!
3//! Every other test in this suite checks orgo against orgo: a snapshot says our
4//! output has not *changed*, never that it is *right*. Those two questions are different,
5//! and only one of them matters to someone whose site is currently published by Emacs.
6//! This file answers the second by exporting the same fixture with org's own HTML
7//! exporter — the exporter weblorg wraps to publish the target corpus — and diffing the
8//! two.
9//!
10//! **What is compared.** Byte equality is not a useful goal: org wraps every section in
11//! `outline-container` divs keyed by generated ids, and no amount of agreement on
12//! semantics would survive that. Both sides are reduced to a *semantic skeleton* — the
13//! sequence of element opens, closes, and text runs, with `<div>`s and all attributes
14//! except `href`/`src` dropped, whitespace collapsed, and entities decoded. What remains
15//! is the question worth asking: does org think this is a `<blockquote><p>`, and do we?
16//!
17//! **What the result means.** These tests do not assert agreement — they *snapshot the
18//! disagreement*. A divergence report that is checked in and reviewed is worth more than
19//! a red test nobody can act on, and it makes any new divergence show up as a diff in
20//! code review. A few invariants that must never break are asserted outright.
21//!
22//! The suite skips cleanly when Emacs is absent, so it never blocks a machine or CI
23//! runner that has no Emacs.
24
25use std::process::Command;
26
27use camino::Utf8PathBuf;
28
29use orgo::parser::parse;
30use orgo::render::{render, Html, SyntectHighlighter};
31use orgo::resolve::ResolvedDoc;
32use orgo::skeleton::{decode_entities, skeleton};
33
34fn manifest_dir() -> Utf8PathBuf {
35 Utf8PathBuf::from(env!("CARGO_MANIFEST_DIR"))
36}
37
38/// Is a usable Emacs on PATH? The oracle is a development instrument, not a build
39/// dependency, so its absence skips rather than fails.
40fn emacs_available() -> bool {
41 Command::new("emacs")
42 .arg("--version")
43 .output()
44 .map(|o| o.status.success())
45 .unwrap_or(false)
46}
47
48/// Export a fixture with org's own HTML exporter.
49fn org_export(fixture: &str) -> String {
50 let root = manifest_dir();
51 let output = Command::new("emacs")
52 .args(["-Q", "--batch", "-l"])
53 .arg(root.join("tests/oracle.el"))
54 .env("ORG_ORACLE_INPUT", root.join("fixtures").join(fixture))
55 .current_dir(&root)
56 .output()
57 .expect("run emacs");
58 assert!(
59 output.status.success(),
60 "emacs export of {fixture} failed:\n{}",
61 String::from_utf8_lossy(&output.stderr)
62 );
63 String::from_utf8(output.stdout).expect("emacs emits UTF-8")
64}
65
66/// Render a fixture with orgo.
67fn our_export(fixture: &str) -> String {
68 let path = manifest_dir().join("fixtures").join(fixture);
69 let source = std::fs::read_to_string(&path).expect("read fixture");
70 let document = parse(Utf8PathBuf::from(fixture).as_path(), &source).expect("parse");
71 let Html(html) = render(&ResolvedDoc { document }, &SyntectHighlighter::new());
72 html
73}
74
75// ---------------------------------------------------------------------------
76// Divergence report
77// ---------------------------------------------------------------------------
78
79/// One divergence orgo makes on purpose, so the report can separate "we chose this"
80/// from "we got this wrong".
81///
82/// Without this split the agreement percentage is noise: the timestamps fixture sat at
83/// 40% while being entirely correct, because org writes `<2024-01-15 Mon>` as text and
84/// orgo writes a `<time datetime>` element. A number that cannot fall when a real
85/// defect appears is not measuring anything.
86struct Deliberate {
87 name: &'static str,
88 /// Does this hunk consist only of the difference described? `ours` are the `-` lines,
89 /// `theirs` the `+` lines.
90 matches: fn(&[String], &[String]) -> bool,
91 /// Only applies once the notes section has started. Two footnote definitions differ
92 /// by `</li><li>` against `</sup><p>`, which is the same shape difference — but a
93 /// list where a paragraph was expected is a real defect anywhere else, so the rule
94 /// is not allowed to explain it anywhere else.
95 in_notes_only: bool,
96}
97
98fn is_tag(line: &str, names: &[&str]) -> bool {
99 names
100 .iter()
101 .any(|n| line == format!("<{n}>") || line == format!("</{n}>"))
102}
103
104fn text_of(line: &str) -> Option<&str> {
105 line.strip_prefix('"')?.strip_suffix('"')
106}
107
108/// Does this text run contain an org timestamp, `<2024-01-15 Mon>` or `[2024-01-15]`?
109fn has_timestamp(line: &str) -> bool {
110 let t = match text_of(line) {
111 Some(t) => t,
112 None => return false,
113 };
114 t.contains('<') && t.contains('-') || t.contains('[') && t.contains('-')
115}
116
117const DELIBERATE: &[Deliberate] = &[
118 // `<time datetime="…">` instead of org's plain text: the date is data, and a reader's
119 // browser can do something with it.
120 Deliberate {
121 name: "semantic-time",
122 matches: |ours, theirs| {
123 !ours.is_empty()
124 && ours
125 .iter()
126 .all(|l| is_tag(l, &["time"]) || text_of(l).is_some())
127 && theirs.iter().all(|l| has_timestamp(l) || text_of(l).is_some())
128 },
129 in_notes_only: false,
130 },
131 // `<figure>`/`<figcaption>` instead of two paragraphs in a div.
132 Deliberate {
133 name: "figure-element",
134 matches: |ours, theirs| {
135 ours.iter().any(|l| is_tag(l, &["figure", "figcaption"]))
136 && ours
137 .iter()
138 .all(|l| is_tag(l, &["figure", "figcaption"]) || text_of(l).is_some())
139 && theirs.iter().all(|l| is_tag(l, &["p"]) || text_of(l).is_some())
140 },
141 in_notes_only: false,
142 },
143 // `<em>`/`<strong>` instead of org's presentational `<i>`/`<b>`.
144 Deliberate {
145 name: "semantic-emphasis",
146 matches: |ours, theirs| {
147 !ours.is_empty()
148 && ours.iter().all(|l| is_tag(l, &["em", "strong", "del"]))
149 && theirs.iter().all(|l| is_tag(l, &["i", "b", "s", "del"]))
150 },
151 in_notes_only: false,
152 },
153 // `<pre><code>` instead of a bare `<pre>`: the nested element is what every syntax
154 // highlighter and every reader's stylesheet expects.
155 Deliberate {
156 name: "pre-code",
157 matches: |ours, theirs| {
158 !ours.is_empty() && ours.iter().all(|l| is_tag(l, &["code"])) && theirs.is_empty()
159 },
160 in_notes_only: false,
161 },
162 // Org emits a `<colgroup>` of empty `<col>`s to carry column alignment; orgo
163 // leaves alignment to the stylesheet.
164 Deliberate {
165 name: "no-colgroup",
166 matches: |ours, theirs| {
167 ours.is_empty()
168 && !theirs.is_empty()
169 && theirs.iter().all(|l| is_tag(l, &["colgroup", "col"]) || l == "<col>")
170 },
171 in_notes_only: false,
172 },
173 // Footnote ids: `fn-1` rather than org's `fn.1`, because a dot in an id is awkward in
174 // a CSS selector.
175 Deliberate {
176 name: "footnote-anchor-naming",
177 matches: |ours, theirs| {
178 ours.len() == 1
179 && theirs.len() == 1
180 && ours[0].replace("fn-", "fn.") == theirs[0]
181 && ours[0].contains("#fn")
182 },
183 in_notes_only: false,
184 },
185 // The notes section itself: an `<ol>` under a rule, rather than org's headed div of
186 // paragraphs, and a `↩` back-link rather than a repeated superscript number. Same
187 // notes, same order, same links, in the shape a screen reader announces as a list.
188 Deliberate {
189 name: "footnote-section-shape",
190 matches: |ours, theirs| {
191 let ours_is_notes = ours.iter().all(|l| {
192 is_tag(l, &["section", "ol", "li", "a", "sup", "p"])
193 || l == "<hr>"
194 || l.starts_with("<a href=\"#fn")
195 || text_of(l) == Some("↩")
196 || text_of(l).is_some()
197 });
198 let theirs_is_notes = theirs.iter().all(|l| {
199 is_tag(l, &["h2", "sup", "p", "a", "div"])
200 || l.starts_with("<a href=\"#fn")
201 || text_of(l).is_some()
202 });
203 let touches_notes = ours.iter().chain(theirs).any(|l| {
204 l.contains("#fn") || text_of(l) == Some("Footnotes:") || is_tag(l, &["section"])
205 });
206 touches_notes && ours_is_notes && theirs_is_notes
207 },
208 in_notes_only: false,
209 },
210 // Two note definitions abutting: `</li><li>` where org writes `</sup><p>`.
211 Deliberate {
212 name: "footnote-section-shape",
213 matches: |ours, theirs| {
214 !ours.is_empty()
215 && ours.iter().all(|l| is_tag(l, &["li", "ol", "section"]))
216 && theirs.iter().all(|l| is_tag(l, &["p", "sup", "div"]))
217 },
218 in_notes_only: true,
219 },
220 // Verse ends without a trailing `<br>`: org emits one for the final newline, which is
221 // a blank line at the end of the stanza and nothing else.
222 Deliberate {
223 name: "verse-trailing-break",
224 matches: |ours, theirs| ours.is_empty() && theirs == ["<br>"],
225 in_notes_only: false,
226 },
227 // `[[id:…]]` resolves here and does not in the oracle: a single-file `emacs --batch`
228 // export has no id database, so org drops the link and keeps its text. This is a
229 // property of the harness, not of either exporter.
230 Deliberate {
231 name: "id-link-resolution",
232 matches: |ours, theirs| {
233 ours.iter().any(|l| l.starts_with("<a href=\"#"))
234 && ours.iter().all(|l| is_tag(l, &["a"]) || l.starts_with("<a href=") || text_of(l).is_some())
235 && theirs.iter().all(|l| text_of(l).is_some())
236 },
237 in_notes_only: false,
238 },
239 // A change of bullet starts a new list. Org instead continues the list — two blank
240 // lines end one, not a switch from `1.` to `-` — so a dash item written under a
241 // numbered list is exported *numbered*. We split, which is what the author drew.
242 // Deliberate, and the one entry here that is arguably worth revisiting.
243 Deliberate {
244 name: "list-per-bullet-type",
245 matches: |ours, theirs| {
246 !ours.is_empty()
247 && ours.iter().all(|l| is_tag(l, &["ol", "ul"]))
248 && theirs.iter().all(|l| is_tag(l, &["ol", "ul"]))
249 },
250 in_notes_only: false,
251 },
252];
253
254/// One run of differing lines: what we wrote, and what Emacs wrote.
255struct Hunk {
256 ours: Vec<String>,
257 theirs: Vec<String>,
258}
259
260impl Hunk {
261 fn len(&self) -> usize {
262 self.ours.len().max(self.theirs.len())
263 }
264
265 fn deliberate(&self, in_notes: bool) -> Option<&'static str> {
266 DELIBERATE
267 .iter()
268 .filter(|d| in_notes || !d.in_notes_only)
269 .find(|d| (d.matches)(&self.ours, &self.theirs))
270 .map(|d| d.name)
271 }
272
273 /// Does this hunk start the footnote section?
274 fn starts_notes(&self) -> bool {
275 self.ours
276 .iter()
277 .chain(&self.theirs)
278 .any(|l| l == "<section>" || text_of(l) == Some("Footnotes:"))
279 }
280}
281
282enum Op {
283 Same(String),
284 Differs(Hunk),
285}
286
287/// Longest-common-subsequence walk, grouped into runs of agreement and disagreement.
288fn align(ours: &[String], theirs: &[String]) -> Vec<Op> {
289 let (n, m) = (ours.len(), theirs.len());
290 let mut lcs = vec![vec![0usize; m + 1]; n + 1];
291 for i in (0..n).rev() {
292 for j in (0..m).rev() {
293 lcs[i][j] = if ours[i] == theirs[j] {
294 lcs[i + 1][j + 1] + 1
295 } else {
296 lcs[i + 1][j].max(lcs[i][j + 1])
297 };
298 }
299 }
300
301 let mut out: Vec<Op> = Vec::new();
302 let push_diff = |out: &mut Vec<Op>, mine: Option<String>, theirs: Option<String>| {
303 if let Some(Op::Differs(h)) = out.last_mut() {
304 h.ours.extend(mine);
305 h.theirs.extend(theirs);
306 return;
307 }
308 out.push(Op::Differs(Hunk {
309 ours: mine.into_iter().collect(),
310 theirs: theirs.into_iter().collect(),
311 }));
312 };
313
314 let (mut i, mut j) = (0, 0);
315 while i < n && j < m {
316 if ours[i] == theirs[j] {
317 out.push(Op::Same(ours[i].clone()));
318 i += 1;
319 j += 1;
320 } else if lcs[i + 1][j] >= lcs[i][j + 1] {
321 push_diff(&mut out, Some(ours[i].clone()), None);
322 i += 1;
323 } else {
324 push_diff(&mut out, None, Some(theirs[j].clone()));
325 j += 1;
326 }
327 }
328 for line in &ours[i..] {
329 push_diff(&mut out, Some(line.clone()), None);
330 }
331 for line in &theirs[j..] {
332 push_diff(&mut out, None, Some(line.clone()));
333 }
334 out
335}
336
337/// A unified diff of the two skeletons, with hunks that are deliberate collapsed to a
338/// named line. `-` is orgo, `+` is Emacs.
339///
340/// The number that matters is the last one: *unexplained* lines. Agreement can be low
341/// while unexplained is zero, and that is a passing state.
342fn divergence(ours: &[String], theirs: &[String]) -> String {
343 let ops = align(ours, theirs);
344 let mut agreed = 0usize;
345 let mut deliberate = 0usize;
346 let mut unexplained = 0usize;
347 let mut by_rule: std::collections::BTreeMap<&str, usize> = std::collections::BTreeMap::new();
348 let mut body = String::new();
349 let mut in_notes = false;
350
351 for op in &ops {
352 match op {
353 Op::Same(line) => {
354 agreed += 1;
355 body.push_str(&format!(" {line}\n"));
356 }
357 Op::Differs(hunk) => {
358 in_notes |= hunk.starts_notes();
359 match hunk.deliberate(in_notes) {
360 Some(rule) => {
361 deliberate += hunk.len();
362 *by_rule.entry(rule).or_default() += 1;
363 body.push_str(&format!("~ {rule} ({} line(s))\n", hunk.len()));
364 }
365 None => {
366 unexplained += hunk.len();
367 for line in &hunk.ours {
368 body.push_str(&format!("- {line}\n"));
369 }
370 for line in &hunk.theirs {
371 body.push_str(&format!("+ {line}\n"));
372 }
373 }
374 }
375 }
376 }
377 }
378
379 let total = ours.len().max(theirs.len());
380 let pct = if total == 0 {
381 100.0
382 } else {
383 100.0 * agreed as f64 / total as f64
384 };
385 let rules: Vec<String> = by_rule
386 .iter()
387 .map(|(name, n)| format!("{name} ×{n}"))
388 .collect();
389 format!(
390 "agreement: {agreed}/{total} skeleton lines ({pct:.1}%)\n\
391 deliberate: {deliberate} line(s){}\n\
392 unexplained: {unexplained} line(s)\n\
393 (- orgo, + emacs, ~ a difference we mean to have)\n\n{body}",
394 if rules.is_empty() {
395 String::new()
396 } else {
397 format!(" — {}", rules.join(", "))
398 }
399 )
400}
401
402/// Snapshot the divergence between orgo and Emacs for one fixture.
403fn compare(fixture: &str) -> Option<String> {
404 if !emacs_available() {
405 eprintln!("skipping oracle comparison for {fixture}: no emacs on PATH");
406 return None;
407 }
408 let ours = skeleton(&our_export(fixture));
409 let theirs = skeleton(&org_export(fixture));
410 Some(divergence(&ours, &theirs))
411}
412
413macro_rules! oracle_test {
414 ($name:ident, $fixture:literal) => {
415 #[test]
416 fn $name() {
417 if let Some(report) = compare($fixture) {
418 insta::assert_snapshot!(report);
419 }
420 }
421 };
422}
423
424oracle_test!(oracle_minimal, "minimal.org");
425oracle_test!(oracle_core, "core.org");
426oracle_test!(oracle_headings, "headings.org");
427oracle_test!(oracle_lists, "lists.org");
428oracle_test!(oracle_blocks, "blocks.org");
429oracle_test!(oracle_table, "table.org");
430oracle_test!(oracle_footnote, "footnote.org");
431oracle_test!(oracle_timestamps, "timestamps.org");
432oracle_test!(oracle_images, "images.org");
433oracle_test!(oracle_elements, "elements.org");
434
435/// The gate the snapshots cannot be: *every* divergence from Emacs must be one we chose.
436///
437/// The percentages above are context, not a target — the timestamps fixture agrees on
438/// 40% of its lines and is entirely correct, because org writes a date as text where
439/// orgo writes `<time datetime>`. What must hold is that nothing diverges for a
440/// reason nobody has written down. A new unexplained line means either a defect to fix
441/// or a decision to record in `DELIBERATE`.
442#[test]
443fn every_divergence_from_emacs_is_deliberate() {
444 let fixtures = [
445 "minimal.org",
446 "core.org",
447 "headings.org",
448 "lists.org",
449 "blocks.org",
450 "table.org",
451 "footnote.org",
452 "timestamps.org",
453 "images.org",
454 "elements.org",
455 "tblfm.org",
456 "audit-entities.org",
457 ];
458 let mut offenders = Vec::new();
459 for fixture in fixtures {
460 let Some(report) = compare(fixture) else {
461 return; // no emacs on PATH; the suite skips cleanly
462 };
463 if !report.contains("unexplained: 0 line(s)") {
464 let count = report
465 .lines()
466 .find(|l| l.starts_with("unexplained:"))
467 .unwrap_or("unexplained: ?");
468 offenders.push(format!("{fixture}: {count}"));
469 }
470 }
471 assert!(
472 offenders.is_empty(),
473 "these fixtures diverge from Emacs for unrecorded reasons — fix the defect, or \
474 add a rule to DELIBERATE saying why the difference is wanted:\n {}",
475 offenders.join("\n ")
476 );
477}
478
479// ---------------------------------------------------------------------------
480// Invariants that must hold against the oracle, not merely be snapshotted
481// ---------------------------------------------------------------------------
482
483/// How many headings a document has and at what depth is the shape of the document.
484/// Getting it wrong reorganizes someone's writing, so it is asserted rather than
485/// snapshotted. Heading *decoration* (priority cookies, tag markup) is a policy
486/// difference and is left to the snapshots.
487#[test]
488fn heading_structure_matches_emacs() {
489 if !emacs_available() {
490 eprintln!("skipping: no emacs on PATH");
491 return;
492 }
493 for fixture in ["minimal.org", "core.org", "headings.org", "lists.org"] {
494 let ours = heading_levels(&skeleton(&our_export(fixture)));
495 let theirs = heading_levels(&skeleton(&org_export(fixture)));
496 assert_eq!(
497 ours, theirs,
498 "heading structure diverges from Emacs in {fixture}"
499 );
500 }
501}
502
503/// The sequence of heading open tags, e.g. `["<h1>", "<h2>", "<h1>"]`.
504fn heading_levels(skeleton: &[String]) -> Vec<String> {
505 skeleton
506 .iter()
507 .filter(|l| l.starts_with("<h") && l[2..].starts_with(|c: char| c.is_ascii_digit()))
508 .cloned()
509 .collect()
510}
511
512/// A list is the construct where nesting is easiest to get subtly wrong, and where being
513/// wrong changes the meaning of the document rather than its looks.
514#[test]
515fn list_nesting_matches_emacs() {
516 if !emacs_available() {
517 eprintln!("skipping: no emacs on PATH");
518 return;
519 }
520 let ours = list_shape(&skeleton(&our_export("lists.org")));
521 let theirs = list_shape(&skeleton(&org_export("lists.org")));
522 assert_eq!(ours, theirs, "list nesting diverges from Emacs");
523}
524
525/// The sequence of list opens/closes, ignoring content — the shape of the nesting.
526fn list_shape(skeleton: &[String]) -> Vec<String> {
527 skeleton
528 .iter()
529 .filter(|l| {
530 matches!(
531 l.as_str(),
532 "<ul>" | "</ul>" | "<ol>" | "</ol>" | "<li>" | "</li>" | "<dl>" | "</dl>"
533 | "<dt>" | "</dt>" | "<dd>" | "</dd>"
534 )
535 })
536 .cloned()
537 .collect()
538}
539
540/// Code must survive verbatim. Highlighting markup differs by construction (syntect
541/// spans vs htmlize), but if the *characters of the program* differ, we have corrupted
542/// the author's content.
543#[test]
544fn source_block_text_matches_emacs() {
545 if !emacs_available() {
546 eprintln!("skipping: no emacs on PATH");
547 return;
548 }
549 for fixture in ["blocks.org", "core.org", "elements.org"] {
550 let ours = code_text(&our_export(fixture));
551 let theirs = code_text(&org_export(fixture));
552 assert_eq!(ours, theirs, "source block text diverges from Emacs in {fixture}");
553 }
554}
555
556/// All text inside `<pre>` blocks, with tags stripped and whitespace collapsed.
557fn code_text(html: &str) -> Vec<String> {
558 let mut out = Vec::new();
559 let mut rest = html;
560 while let Some(start) = rest.find("<pre") {
561 let after = &rest[start..];
562 let Some(open_end) = after.find('>') else { break };
563 let Some(close) = after.find("</pre>") else { break };
564 let inner = &after[open_end + 1..close];
565 out.push(strip_tags(inner));
566 rest = &after[close + 6..];
567 }
568 out
569}
570
571/// All text in a fragment with tags removed and entities decoded, then whitespace
572/// collapsed once at the end.
573///
574/// [`skeleton`] cannot do this job: it trims each text run individually, which is
575/// invisible for prose (one run per paragraph) but destructive for highlighted code,
576/// where syntect splits a line into one run per token and the spaces *between* tokens
577/// live at the edges of those runs. Trimming each run turns `def greet` into `defgreet`.
578fn strip_tags(html: &str) -> String {
579 let mut text = String::new();
580 let mut rest = html;
581 while let Some(open) = rest.find('<') {
582 text.push_str(&rest[..open]);
583 match rest[open..].find('>') {
584 Some(close) => rest = &rest[open + close + 1..],
585 None => {
586 rest = "";
587 break;
588 }
589 }
590 }
591 text.push_str(rest);
592 decode_entities(&text)
593 .split_whitespace()
594 .collect::<Vec<_>>()
595 .join(" ")
596}