krz/orgo

Lightning fast org-mode static site generator. fast go org-mode static-site-generator

Commit 69c12446b8

69c12446b8b42b1e888948409b4b6e092352d823

parent: 2eac492d17

Verified · cmc

cmc <hello@cleberg.net> · 2026-08-27 16:07 UTC

skeleton: promote the HTML→semantic-skeleton reducer to a public module

Move the skeleton reduction out of tests/oracle.rs into src/skeleton.rs and
expose skeleton() and decode_entities(). The oracle now imports it, and an
examples/skeleton binary renders a fixture to HTML or skeleton — the reference
outputs the org-conformance corpus is built from. tests/conformance.rs verifies
orgo still matches the corpus goldens (opt-in via ORG_CONFORMANCE_DIR), so the
generated corpus cannot drift from orgo's renderer unnoticed.

Layout: unified · split

examples/skeleton.rs added +46
@@ -0,0 +1,46 @@
1//! Render an org file the way orgo would and print either its HTML or its semantic
2//! skeleton — the reference outputs the cross-language conformance corpus is built from.
3//!
4//! Usage:
5//! cargo run --example skeleton -- <file.org> # print the skeleton
6//! cargo run --example skeleton -- --html <file.org> # print the rendered HTML
7//!
8//! The skeleton is one line per element open, element close, or text run (see
9//! [`orgo::skeleton`]). Another implementation conforms when its own output, reduced by
10//! its own port of the same reduction, equals this.
11
12use camino::Utf8PathBuf;
13
14use orgo::parser::parse;
15use orgo::render::{render, Html, SyntectHighlighter};
16use orgo::resolve::ResolvedDoc;
17use orgo::skeleton::skeleton;
18
19fn main() {
20 let mut args = std::env::args().skip(1);
21 let mut html_mode = false;
22 let mut path: Option<String> = None;
23 for arg in args.by_ref() {
24 match arg.as_str() {
25 "--html" => html_mode = true,
26 _ => path = Some(arg),
27 }
28 }
29 let Some(path) = path else {
30 eprintln!("usage: skeleton [--html] <file.org>");
31 std::process::exit(2);
32 };
33
34 let path = Utf8PathBuf::from(path);
35 let source = std::fs::read_to_string(&path).expect("read org file");
36 let document = parse(path.as_path(), &source).expect("parse org file");
37 let Html(html) = render(&ResolvedDoc { document }, &SyntectHighlighter::new());
38
39 if html_mode {
40 print!("{html}");
41 } else {
42 for line in skeleton(&html) {
43 println!("{line}");
44 }
45 }
46}
src/lib.rs +1
@@ -19,6 +19,7 @@ pub mod render;
1919pub mod resolve;
2020pub mod serve;
2121pub mod site;
22pub mod skeleton;
2223pub mod template;
2324pub mod theme;
2425pub mod util;
src/skeleton.rs added +189
@@ -0,0 +1,189 @@
1//! HTML → semantic skeleton reduction.
2//!
3//! Two HTML exporters that agree on what a document *means* will still disagree on how
4//! they wrap it: org buries every section in `outline-container` divs keyed by generated
5//! ids, syntect emits one `<span>` per code token, and each backend picks its own class
6//! names. Byte equality therefore measures nothing. The skeleton throws all of that away
7//! and keeps the part worth comparing: the ordered sequence of element opens, element
8//! closes, and text runs, with `<div>`/`<span>` dropped, every attribute except
9//! `href`/`src` removed, whitespace collapsed, and entities decoded.
10//!
11//! This is the comparison primitive behind two things at once: orgo's differential check
12//! against Emacs (`tests/oracle.rs`), and the cross-language conformance corpus, where an
13//! implementation in another language ports [`skeleton`] and asserts its output matches
14//! orgo's checked-in golden skeleton. Keeping one implementation here, rather than a copy
15//! per consumer, is the point — the reduction *is* the contract, so it must be identical
16//! everywhere.
17
18/// Elements dropped from the skeleton entirely, because once attributes are gone they
19/// carry no meaning two exporters could agree or disagree *about*.
20///
21/// `div` is pure layout: org wraps every section in `outline-container`/`outline-text`
22/// wrappers and we emit none. `span` is the same story at the inline level, and matters
23/// far more than it looks: syntect emits one span per code token, so keeping them made a
24/// source block contribute ~60 skeleton lines of pure noise and dragged the agreement on
25/// `blocks.org` down to 36% — a number that said nothing about whether we render blocks
26/// correctly. Text still carries the signal: a `<span class="todo">` shows up as its
27/// text, `"TODO"`, which is the part worth comparing.
28const IGNORED: &[&str] = &["div", "span"];
29
30/// Attributes kept in the skeleton. Ids and classes are generated (`org6c28c1b`) or
31/// cosmetic (`org-ul`); `href` and `src` are the content.
32const KEPT_ATTRS: &[&str] = &["href", "src"];
33
34/// HTML void elements, which never emit a close event.
35const VOID: &[&str] = &[
36 "br", "hr", "img", "input", "meta", "link", "col", "area", "base", "source", "wbr",
37];
38
39/// Reduce an HTML fragment to its semantic skeleton: one line per element open, element
40/// close, or text run.
41pub fn skeleton(html: &str) -> Vec<String> {
42 let mut out = Vec::new();
43 let chars: Vec<char> = html.chars().collect();
44 let mut i = 0;
45 let mut text = String::new();
46
47 while i < chars.len() {
48 if chars[i] != '<' {
49 text.push(chars[i]);
50 i += 1;
51 continue;
52 }
53
54 // Comments and doctypes carry nothing.
55 if chars[i..].starts_with(&['<', '!']) {
56 i += match find_from(&chars, i, ">") {
57 Some(end) => end - i + 1,
58 None => break,
59 };
60 continue;
61 }
62 let Some(end) = find_from(&chars, i, ">") else {
63 break;
64 };
65 let raw: String = chars[i + 1..end].iter().collect();
66 i = end + 1;
67
68 let raw = raw.trim().trim_end_matches('/').trim().to_string();
69 // Text is flushed only when a tag is actually *emitted*. Text either side of an
70 // ignored tag therefore merges into one run, which is what makes a highlighted
71 // source block compare as the one string of code it is, rather than as a
72 // token-by-token sequence that has to line up exactly.
73 if let Some(name) = raw.strip_prefix('/') {
74 let name = name.trim().to_ascii_lowercase();
75 if !IGNORED.contains(&name.as_str()) && !VOID.contains(&name.as_str()) {
76 flush_text(&mut text, &mut out);
77 out.push(format!("</{name}>"));
78 }
79 continue;
80 }
81 let mut parts = raw.splitn(2, char::is_whitespace);
82 let name = parts.next().unwrap_or("").to_ascii_lowercase();
83 if name.is_empty() || IGNORED.contains(&name.as_str()) {
84 continue;
85 }
86 let attrs = kept_attributes(parts.next().unwrap_or(""));
87 flush_text(&mut text, &mut out);
88 out.push(format!("<{name}{attrs}>"));
89 }
90 flush_text(&mut text, &mut out);
91 out
92}
93
94fn flush_text(text: &mut String, out: &mut Vec<String>) {
95 let decoded = decode_entities(text);
96 let collapsed = decoded.split_whitespace().collect::<Vec<_>>().join(" ");
97 if !collapsed.is_empty() {
98 out.push(format!("{collapsed:?}"));
99 }
100 text.clear();
101}
102
103fn find_from(chars: &[char], from: usize, needle: &str) -> Option<usize> {
104 let n: Vec<char> = needle.chars().collect();
105 (from..chars.len()).find(|&k| chars[k..].starts_with(&n[..]))
106}
107
108/// Keep only the content-bearing attributes, in a stable order.
109fn kept_attributes(rest: &str) -> String {
110 let mut kept: Vec<(String, String)> = Vec::new();
111 for attr in KEPT_ATTRS {
112 if let Some(value) = attribute_value(rest, attr) {
113 kept.push(((*attr).to_string(), value));
114 }
115 }
116 kept.iter()
117 .map(|(k, v)| format!(" {k}=\"{}\"", decode_entities(v)))
118 .collect()
119}
120
121fn attribute_value(rest: &str, name: &str) -> Option<String> {
122 let mut search = rest;
123 while let Some(pos) = search.find(name) {
124 let before_ok = pos == 0
125 || search[..pos]
126 .chars()
127 .next_back()
128 .is_some_and(char::is_whitespace);
129 let after = &search[pos + name.len()..];
130 let after_trimmed = after.trim_start();
131 if before_ok && after_trimmed.starts_with('=') {
132 let value = after_trimmed[1..].trim_start();
133 let quote = value.chars().next()?;
134 if quote == '"' || quote == '\'' {
135 let end = value[1..].find(quote)? + 1;
136 return Some(value[1..end].to_string());
137 }
138 let end = value.find(char::is_whitespace).unwrap_or(value.len());
139 return Some(value[..end].to_string());
140 }
141 search = &search[pos + name.len()..];
142 }
143 None
144}
145
146/// Decode the entities either exporter is likely to emit, so an encoding difference is
147/// never reported as a semantic one.
148pub fn decode_entities(s: &str) -> String {
149 let mut out = String::with_capacity(s.len());
150 let mut rest = s;
151 while let Some(amp) = rest.find('&') {
152 out.push_str(&rest[..amp]);
153 let tail = &rest[amp..];
154 let Some(semi) = tail.find(';').filter(|s| *s <= 12) else {
155 out.push('&');
156 rest = &tail[1..];
157 continue;
158 };
159 let entity = &tail[1..semi];
160 let decoded = match entity {
161 "amp" => Some('&'),
162 "lt" => Some('<'),
163 "gt" => Some('>'),
164 "quot" => Some('"'),
165 "apos" => Some('\''),
166 "nbsp" => Some(' '),
167 _ => entity
168 .strip_prefix('#')
169 .and_then(|n| match n.strip_prefix(['x', 'X']) {
170 Some(hex) => u32::from_str_radix(hex, 16).ok(),
171 None => n.parse::<u32>().ok(),
172 })
173 .and_then(char::from_u32),
174 };
175 match decoded {
176 // A non-breaking space is a space for comparison purposes.
177 Some('\u{a0}') => out.push(' '),
178 Some(c) => out.push(c),
179 None => {
180 out.push('&');
181 rest = &tail[1..];
182 continue;
183 }
184 }
185 rest = &tail[semi + 1..];
186 }
187 out.push_str(rest);
188 out
189}
tests/conformance.rs added +73
@@ -0,0 +1,73 @@
1//! orgo against the shared conformance corpus (`krz/org-conformance`).
2//!
3//! orgo *generates* that corpus's goldens, so this test is not looking for orgo to be
4//! wrong — it is the tripwire that fires when orgo's renderer changes and the checked-in
5//! corpus was not regenerated to match. Without it the two drift silently: orgo's own
6//! snapshots update with `cargo insta accept`, the corpus does not, and every other
7//! language is then conforming to a stale reference.
8//!
9//! The corpus lives in a sibling repo, so the test is opt-in: point `ORG_CONFORMANCE_DIR`
10//! at a checkout and it runs; unset, it skips cleanly, exactly like the Emacs oracle.
11//!
12//! ORG_CONFORMANCE_DIR=../org-conformance cargo test --test conformance
13
14use std::path::PathBuf;
15
16use camino::Utf8PathBuf;
17
18use orgo::parser::parse;
19use orgo::render::{render, Html, SyntectHighlighter};
20use orgo::resolve::ResolvedDoc;
21use orgo::skeleton::skeleton;
22
23fn corpus_dir() -> Option<PathBuf> {
24 let dir = PathBuf::from(std::env::var_os("ORG_CONFORMANCE_DIR")?);
25 dir.join("cases").is_dir().then_some(dir)
26}
27
28fn render_skeleton(org_path: &std::path::Path) -> Vec<String> {
29 let source = std::fs::read_to_string(org_path).expect("read case .org");
30 let rel = Utf8PathBuf::from(org_path.file_name().unwrap().to_string_lossy().into_owned());
31 let document = parse(rel.as_path(), &source).expect("parse case");
32 let Html(html) = render(&ResolvedDoc { document }, &SyntectHighlighter::new());
33 skeleton(&html)
34}
35
36#[test]
37fn orgo_matches_the_conformance_corpus() {
38 let Some(dir) = corpus_dir() else {
39 eprintln!("ORG_CONFORMANCE_DIR unset or has no cases/ — skipping conformance check");
40 return;
41 };
42 let cases = dir.join("cases");
43
44 let mut checked = 0;
45 let mut mismatches = Vec::new();
46 for entry in std::fs::read_dir(&cases).expect("read cases dir") {
47 let path = entry.expect("dir entry").path();
48 if path.extension().and_then(|e| e.to_str()) != Some("org") {
49 continue;
50 }
51 let name = path.file_stem().unwrap().to_string_lossy().into_owned();
52 let golden_path = cases.join(format!("{name}.skeleton"));
53 let golden = std::fs::read_to_string(&golden_path)
54 .unwrap_or_else(|_| panic!("missing golden skeleton for case {name}"));
55 let golden_lines: Vec<&str> = golden.lines().collect();
56
57 let ours = render_skeleton(&path);
58 if ours != golden_lines {
59 mismatches.push(name);
60 }
61 checked += 1;
62 }
63
64 assert!(checked > 0, "corpus at {} has no .org cases", cases.display());
65 assert!(
66 mismatches.is_empty(),
67 "orgo's skeleton no longer matches the corpus goldens for: {}. \
68 If the renderer change is intended, regenerate the corpus \
69 (ORGO_DIR=../orgo tools/generate.sh) and commit it.",
70 mismatches.join(", ")
71 );
72 eprintln!("conformance: {checked} cases match the corpus goldens");
73}
tests/oracle.rs +1 −177
@@ -29,6 +29,7 @@ use camino::Utf8PathBuf;
2929use orgo::parser::parse;
3030use orgo::render::{render, Html, SyntectHighlighter};
3131use orgo::resolve::ResolvedDoc;
32use orgo::skeleton::{decode_entities, skeleton};
3233
3334fn manifest_dir() -> Utf8PathBuf {
3435 Utf8PathBuf::from(env!("CARGO_MANIFEST_DIR"))
@@ -71,183 +72,6 @@ fn our_export(fixture: &str) -> String {
7172 html
7273}
7374
74// ---------------------------------------------------------------------------
75// HTML → semantic skeleton
76// ---------------------------------------------------------------------------
77
78/// Elements dropped from the skeleton entirely, because once attributes are gone they
79/// carry no meaning the two exporters could agree or disagree *about*.
80///
81/// `div` is pure layout: org wraps every section in `outline-container`/`outline-text`
82/// wrappers and we emit none. `span` is the same story at the inline level, and matters
83/// far more than it looks: syntect emits one span per code token, so keeping them made a
84/// source block contribute ~60 skeleton lines of pure noise and dragged the agreement on
85/// `blocks.org` down to 36% — a number that said nothing about whether we render blocks
86/// correctly. Text still carries the signal: a `<span class="todo">` shows up as its
87/// text, `"TODO"`, which is the part worth comparing.
88const IGNORED: &[&str] = &["div", "span"];
89
90/// Attributes kept in the skeleton. Ids and classes are generated (`org6c28c1b`) or
91/// cosmetic (`org-ul`); `href` and `src` are the content.
92const KEPT_ATTRS: &[&str] = &["href", "src"];
93
94/// HTML void elements, which never emit a close event.
95const VOID: &[&str] = &[
96 "br", "hr", "img", "input", "meta", "link", "col", "area", "base", "source", "wbr",
97];
98
99/// Reduce an HTML fragment to its semantic skeleton: one line per element open, element
100/// close, or text run.
101fn skeleton(html: &str) -> Vec<String> {
102 let mut out = Vec::new();
103 let chars: Vec<char> = html.chars().collect();
104 let mut i = 0;
105 let mut text = String::new();
106
107 while i < chars.len() {
108 if chars[i] != '<' {
109 text.push(chars[i]);
110 i += 1;
111 continue;
112 }
113
114 // Comments and doctypes carry nothing.
115 if chars[i..].starts_with(&['<', '!']) {
116 i += match find_from(&chars, i, ">") {
117 Some(end) => end - i + 1,
118 None => break,
119 };
120 continue;
121 }
122 let Some(end) = find_from(&chars, i, ">") else {
123 break;
124 };
125 let raw: String = chars[i + 1..end].iter().collect();
126 i = end + 1;
127
128 let raw = raw.trim().trim_end_matches('/').trim().to_string();
129 // Text is flushed only when a tag is actually *emitted*. Text either side of an
130 // ignored tag therefore merges into one run, which is what makes a highlighted
131 // source block compare as the one string of code it is, rather than as a
132 // token-by-token sequence that has to line up exactly.
133 if let Some(name) = raw.strip_prefix('/') {
134 let name = name.trim().to_ascii_lowercase();
135 if !IGNORED.contains(&name.as_str()) && !VOID.contains(&name.as_str()) {
136 flush_text(&mut text, &mut out);
137 out.push(format!("</{name}>"));
138 }
139 continue;
140 }
141 let mut parts = raw.splitn(2, char::is_whitespace);
142 let name = parts.next().unwrap_or("").to_ascii_lowercase();
143 if name.is_empty() || IGNORED.contains(&name.as_str()) {
144 continue;
145 }
146 let attrs = kept_attributes(parts.next().unwrap_or(""));
147 flush_text(&mut text, &mut out);
148 out.push(format!("<{name}{attrs}>"));
149 }
150 flush_text(&mut text, &mut out);
151 out
152}
153
154fn flush_text(text: &mut String, out: &mut Vec<String>) {
155 let decoded = decode_entities(text);
156 let collapsed = decoded.split_whitespace().collect::<Vec<_>>().join(" ");
157 if !collapsed.is_empty() {
158 out.push(format!("{collapsed:?}"));
159 }
160 text.clear();
161}
162
163fn find_from(chars: &[char], from: usize, needle: &str) -> Option<usize> {
164 let n: Vec<char> = needle.chars().collect();
165 (from..chars.len()).find(|&k| chars[k..].starts_with(&n[..]))
166}
167
168/// Keep only the content-bearing attributes, in a stable order.
169fn kept_attributes(rest: &str) -> String {
170 let mut kept: Vec<(String, String)> = Vec::new();
171 for attr in KEPT_ATTRS {
172 if let Some(value) = attribute_value(rest, attr) {
173 kept.push(((*attr).to_string(), value));
174 }
175 }
176 kept.iter()
177 .map(|(k, v)| format!(" {k}=\"{}\"", decode_entities(v)))
178 .collect()
179}
180
181fn attribute_value(rest: &str, name: &str) -> Option<String> {
182 let mut search = rest;
183 while let Some(pos) = search.find(name) {
184 let before_ok = pos == 0
185 || search[..pos]
186 .chars()
187 .next_back()
188 .is_some_and(char::is_whitespace);
189 let after = &search[pos + name.len()..];
190 let after_trimmed = after.trim_start();
191 if before_ok && after_trimmed.starts_with('=') {
192 let value = after_trimmed[1..].trim_start();
193 let quote = value.chars().next()?;
194 if quote == '"' || quote == '\'' {
195 let end = value[1..].find(quote)? + 1;
196 return Some(value[1..end].to_string());
197 }
198 let end = value.find(char::is_whitespace).unwrap_or(value.len());
199 return Some(value[..end].to_string());
200 }
201 search = &search[pos + name.len()..];
202 }
203 None
204}
205
206/// Decode the entities either exporter is likely to emit, so an encoding difference is
207/// never reported as a semantic one.
208fn decode_entities(s: &str) -> String {
209 let mut out = String::with_capacity(s.len());
210 let mut rest = s;
211 while let Some(amp) = rest.find('&') {
212 out.push_str(&rest[..amp]);
213 let tail = &rest[amp..];
214 let Some(semi) = tail.find(';').filter(|s| *s <= 12) else {
215 out.push('&');
216 rest = &tail[1..];
217 continue;
218 };
219 let entity = &tail[1..semi];
220 let decoded = match entity {
221 "amp" => Some('&'),
222 "lt" => Some('<'),
223 "gt" => Some('>'),
224 "quot" => Some('"'),
225 "apos" => Some('\''),
226 "nbsp" => Some(' '),
227 _ => entity
228 .strip_prefix('#')
229 .and_then(|n| match n.strip_prefix(['x', 'X']) {
230 Some(hex) => u32::from_str_radix(hex, 16).ok(),
231 None => n.parse::<u32>().ok(),
232 })
233 .and_then(char::from_u32),
234 };
235 match decoded {
236 // A non-breaking space is a space for comparison purposes.
237 Some('\u{a0}') => out.push(' '),
238 Some(c) => out.push(c),
239 None => {
240 out.push('&');
241 rest = &tail[1..];
242 continue;
243 }
244 }
245 rest = &tail[semi + 1..];
246 }
247 out.push_str(rest);
248 out
249}
250
25175// ---------------------------------------------------------------------------
25276// Divergence report
25377// ---------------------------------------------------------------------------