src/skeleton.rs
189 lines · 7625 bytes
1//! HTML → semantic skeleton reduction.
2//!
3//! Two HTML exporters that agree on what a document *means* will still disagree on how
4//! they wrap it: org buries every section in `outline-container` divs keyed by generated
5//! ids, syntect emits one `<span>` per code token, and each backend picks its own class
6//! names. Byte equality therefore measures nothing. The skeleton throws all of that away
7//! and keeps the part worth comparing: the ordered sequence of element opens, element
8//! closes, and text runs, with `<div>`/`<span>` dropped, every attribute except
9//! `href`/`src` removed, whitespace collapsed, and entities decoded.
10//!
11//! This is the comparison primitive behind two things at once: orgo's differential check
12//! against Emacs (`tests/oracle.rs`), and the cross-language conformance corpus, where an
13//! implementation in another language ports [`skeleton`] and asserts its output matches
14//! orgo's checked-in golden skeleton. Keeping one implementation here, rather than a copy
15//! per consumer, is the point — the reduction *is* the contract, so it must be identical
16//! everywhere.
17
18/// Elements dropped from the skeleton entirely, because once attributes are gone they
19/// carry no meaning two exporters could agree or disagree *about*.
20///
21/// `div` is pure layout: org wraps every section in `outline-container`/`outline-text`
22/// wrappers and we emit none. `span` is the same story at the inline level, and matters
23/// far more than it looks: syntect emits one span per code token, so keeping them made a
24/// source block contribute ~60 skeleton lines of pure noise and dragged the agreement on
25/// `blocks.org` down to 36% — a number that said nothing about whether we render blocks
26/// correctly. Text still carries the signal: a `<span class="todo">` shows up as its
27/// text, `"TODO"`, which is the part worth comparing.
28const IGNORED: &[&str] = &["div", "span"];
29
30/// Attributes kept in the skeleton. Ids and classes are generated (`org6c28c1b`) or
31/// cosmetic (`org-ul`); `href` and `src` are the content.
32const KEPT_ATTRS: &[&str] = &["href", "src"];
33
34/// HTML void elements, which never emit a close event.
35const VOID: &[&str] = &[
36 "br", "hr", "img", "input", "meta", "link", "col", "area", "base", "source", "wbr",
37];
38
39/// Reduce an HTML fragment to its semantic skeleton: one line per element open, element
40/// close, or text run.
41pub fn skeleton(html: &str) -> Vec<String> {
42 let mut out = Vec::new();
43 let chars: Vec<char> = html.chars().collect();
44 let mut i = 0;
45 let mut text = String::new();
46
47 while i < chars.len() {
48 if chars[i] != '<' {
49 text.push(chars[i]);
50 i += 1;
51 continue;
52 }
53
54 // Comments and doctypes carry nothing.
55 if chars[i..].starts_with(&['<', '!']) {
56 i += match find_from(&chars, i, ">") {
57 Some(end) => end - i + 1,
58 None => break,
59 };
60 continue;
61 }
62 let Some(end) = find_from(&chars, i, ">") else {
63 break;
64 };
65 let raw: String = chars[i + 1..end].iter().collect();
66 i = end + 1;
67
68 let raw = raw.trim().trim_end_matches('/').trim().to_string();
69 // Text is flushed only when a tag is actually *emitted*. Text either side of an
70 // ignored tag therefore merges into one run, which is what makes a highlighted
71 // source block compare as the one string of code it is, rather than as a
72 // token-by-token sequence that has to line up exactly.
73 if let Some(name) = raw.strip_prefix('/') {
74 let name = name.trim().to_ascii_lowercase();
75 if !IGNORED.contains(&name.as_str()) && !VOID.contains(&name.as_str()) {
76 flush_text(&mut text, &mut out);
77 out.push(format!("</{name}>"));
78 }
79 continue;
80 }
81 let mut parts = raw.splitn(2, char::is_whitespace);
82 let name = parts.next().unwrap_or("").to_ascii_lowercase();
83 if name.is_empty() || IGNORED.contains(&name.as_str()) {
84 continue;
85 }
86 let attrs = kept_attributes(parts.next().unwrap_or(""));
87 flush_text(&mut text, &mut out);
88 out.push(format!("<{name}{attrs}>"));
89 }
90 flush_text(&mut text, &mut out);
91 out
92}
93
94fn flush_text(text: &mut String, out: &mut Vec<String>) {
95 let decoded = decode_entities(text);
96 let collapsed = decoded.split_whitespace().collect::<Vec<_>>().join(" ");
97 if !collapsed.is_empty() {
98 out.push(format!("{collapsed:?}"));
99 }
100 text.clear();
101}
102
103fn find_from(chars: &[char], from: usize, needle: &str) -> Option<usize> {
104 let n: Vec<char> = needle.chars().collect();
105 (from..chars.len()).find(|&k| chars[k..].starts_with(&n[..]))
106}
107
108/// Keep only the content-bearing attributes, in a stable order.
109fn kept_attributes(rest: &str) -> String {
110 let mut kept: Vec<(String, String)> = Vec::new();
111 for attr in KEPT_ATTRS {
112 if let Some(value) = attribute_value(rest, attr) {
113 kept.push(((*attr).to_string(), value));
114 }
115 }
116 kept.iter()
117 .map(|(k, v)| format!(" {k}=\"{}\"", decode_entities(v)))
118 .collect()
119}
120
121fn attribute_value(rest: &str, name: &str) -> Option<String> {
122 let mut search = rest;
123 while let Some(pos) = search.find(name) {
124 let before_ok = pos == 0
125 || search[..pos]
126 .chars()
127 .next_back()
128 .is_some_and(char::is_whitespace);
129 let after = &search[pos + name.len()..];
130 let after_trimmed = after.trim_start();
131 if before_ok && after_trimmed.starts_with('=') {
132 let value = after_trimmed[1..].trim_start();
133 let quote = value.chars().next()?;
134 if quote == '"' || quote == '\'' {
135 let end = value[1..].find(quote)? + 1;
136 return Some(value[1..end].to_string());
137 }
138 let end = value.find(char::is_whitespace).unwrap_or(value.len());
139 return Some(value[..end].to_string());
140 }
141 search = &search[pos + name.len()..];
142 }
143 None
144}
145
146/// Decode the entities either exporter is likely to emit, so an encoding difference is
147/// never reported as a semantic one.
148pub fn decode_entities(s: &str) -> String {
149 let mut out = String::with_capacity(s.len());
150 let mut rest = s;
151 while let Some(amp) = rest.find('&') {
152 out.push_str(&rest[..amp]);
153 let tail = &rest[amp..];
154 let Some(semi) = tail.find(';').filter(|s| *s <= 12) else {
155 out.push('&');
156 rest = &tail[1..];
157 continue;
158 };
159 let entity = &tail[1..semi];
160 let decoded = match entity {
161 "amp" => Some('&'),
162 "lt" => Some('<'),
163 "gt" => Some('>'),
164 "quot" => Some('"'),
165 "apos" => Some('\''),
166 "nbsp" => Some(' '),
167 _ => entity
168 .strip_prefix('#')
169 .and_then(|n| match n.strip_prefix(['x', 'X']) {
170 Some(hex) => u32::from_str_radix(hex, 16).ok(),
171 None => n.parse::<u32>().ok(),
172 })
173 .and_then(char::from_u32),
174 };
175 match decoded {
176 // A non-breaking space is a space for comparison purposes.
177 Some('\u{a0}') => out.push(' '),
178 Some(c) => out.push(c),
179 None => {
180 out.push('&');
181 rest = &tail[1..];
182 continue;
183 }
184 }
185 rest = &tail[semi + 1..];
186 }
187 out.push_str(rest);
188 out
189}