krz/orgo

Lightning fast org-mode static site generator. fast go org-mode static-site-generator

src/skeleton.rs

189 lines · 7625 bytes

  1//! HTML → semantic skeleton reduction.
  2//!
  3//! Two HTML exporters that agree on what a document *means* will still disagree on how
  4//! they wrap it: org buries every section in `outline-container` divs keyed by generated
  5//! ids, syntect emits one `<span>` per code token, and each backend picks its own class
  6//! names. Byte equality therefore measures nothing. The skeleton throws all of that away
  7//! and keeps the part worth comparing: the ordered sequence of element opens, element
  8//! closes, and text runs, with `<div>`/`<span>` dropped, every attribute except
  9//! `href`/`src` removed, whitespace collapsed, and entities decoded.
 10//!
 11//! This is the comparison primitive behind two things at once: orgo's differential check
 12//! against Emacs (`tests/oracle.rs`), and the cross-language conformance corpus, where an
 13//! implementation in another language ports [`skeleton`] and asserts its output matches
 14//! orgo's checked-in golden skeleton. Keeping one implementation here, rather than a copy
 15//! per consumer, is the point — the reduction *is* the contract, so it must be identical
 16//! everywhere.
 17
 18/// Elements dropped from the skeleton entirely, because once attributes are gone they
 19/// carry no meaning two exporters could agree or disagree *about*.
 20///
 21/// `div` is pure layout: org wraps every section in `outline-container`/`outline-text`
 22/// wrappers and we emit none. `span` is the same story at the inline level, and matters
 23/// far more than it looks: syntect emits one span per code token, so keeping them made a
 24/// source block contribute ~60 skeleton lines of pure noise and dragged the agreement on
 25/// `blocks.org` down to 36% — a number that said nothing about whether we render blocks
 26/// correctly. Text still carries the signal: a `<span class="todo">` shows up as its
 27/// text, `"TODO"`, which is the part worth comparing.
 28const IGNORED: &[&str] = &["div", "span"];
 29
 30/// Attributes kept in the skeleton. Ids and classes are generated (`org6c28c1b`) or
 31/// cosmetic (`org-ul`); `href` and `src` are the content.
 32const KEPT_ATTRS: &[&str] = &["href", "src"];
 33
 34/// HTML void elements, which never emit a close event.
 35const VOID: &[&str] = &[
 36    "br", "hr", "img", "input", "meta", "link", "col", "area", "base", "source", "wbr",
 37];
 38
 39/// Reduce an HTML fragment to its semantic skeleton: one line per element open, element
 40/// close, or text run.
 41pub fn skeleton(html: &str) -> Vec<String> {
 42    let mut out = Vec::new();
 43    let chars: Vec<char> = html.chars().collect();
 44    let mut i = 0;
 45    let mut text = String::new();
 46
 47    while i < chars.len() {
 48        if chars[i] != '<' {
 49            text.push(chars[i]);
 50            i += 1;
 51            continue;
 52        }
 53
 54        // Comments and doctypes carry nothing.
 55        if chars[i..].starts_with(&['<', '!']) {
 56            i += match find_from(&chars, i, ">") {
 57                Some(end) => end - i + 1,
 58                None => break,
 59            };
 60            continue;
 61        }
 62        let Some(end) = find_from(&chars, i, ">") else {
 63            break;
 64        };
 65        let raw: String = chars[i + 1..end].iter().collect();
 66        i = end + 1;
 67
 68        let raw = raw.trim().trim_end_matches('/').trim().to_string();
 69        // Text is flushed only when a tag is actually *emitted*. Text either side of an
 70        // ignored tag therefore merges into one run, which is what makes a highlighted
 71        // source block compare as the one string of code it is, rather than as a
 72        // token-by-token sequence that has to line up exactly.
 73        if let Some(name) = raw.strip_prefix('/') {
 74            let name = name.trim().to_ascii_lowercase();
 75            if !IGNORED.contains(&name.as_str()) && !VOID.contains(&name.as_str()) {
 76                flush_text(&mut text, &mut out);
 77                out.push(format!("</{name}>"));
 78            }
 79            continue;
 80        }
 81        let mut parts = raw.splitn(2, char::is_whitespace);
 82        let name = parts.next().unwrap_or("").to_ascii_lowercase();
 83        if name.is_empty() || IGNORED.contains(&name.as_str()) {
 84            continue;
 85        }
 86        let attrs = kept_attributes(parts.next().unwrap_or(""));
 87        flush_text(&mut text, &mut out);
 88        out.push(format!("<{name}{attrs}>"));
 89    }
 90    flush_text(&mut text, &mut out);
 91    out
 92}
 93
 94fn flush_text(text: &mut String, out: &mut Vec<String>) {
 95    let decoded = decode_entities(text);
 96    let collapsed = decoded.split_whitespace().collect::<Vec<_>>().join(" ");
 97    if !collapsed.is_empty() {
 98        out.push(format!("{collapsed:?}"));
 99    }
100    text.clear();
101}
102
103fn find_from(chars: &[char], from: usize, needle: &str) -> Option<usize> {
104    let n: Vec<char> = needle.chars().collect();
105    (from..chars.len()).find(|&k| chars[k..].starts_with(&n[..]))
106}
107
108/// Keep only the content-bearing attributes, in a stable order.
109fn kept_attributes(rest: &str) -> String {
110    let mut kept: Vec<(String, String)> = Vec::new();
111    for attr in KEPT_ATTRS {
112        if let Some(value) = attribute_value(rest, attr) {
113            kept.push(((*attr).to_string(), value));
114        }
115    }
116    kept.iter()
117        .map(|(k, v)| format!(" {k}=\"{}\"", decode_entities(v)))
118        .collect()
119}
120
121fn attribute_value(rest: &str, name: &str) -> Option<String> {
122    let mut search = rest;
123    while let Some(pos) = search.find(name) {
124        let before_ok = pos == 0
125            || search[..pos]
126                .chars()
127                .next_back()
128                .is_some_and(char::is_whitespace);
129        let after = &search[pos + name.len()..];
130        let after_trimmed = after.trim_start();
131        if before_ok && after_trimmed.starts_with('=') {
132            let value = after_trimmed[1..].trim_start();
133            let quote = value.chars().next()?;
134            if quote == '"' || quote == '\'' {
135                let end = value[1..].find(quote)? + 1;
136                return Some(value[1..end].to_string());
137            }
138            let end = value.find(char::is_whitespace).unwrap_or(value.len());
139            return Some(value[..end].to_string());
140        }
141        search = &search[pos + name.len()..];
142    }
143    None
144}
145
146/// Decode the entities either exporter is likely to emit, so an encoding difference is
147/// never reported as a semantic one.
148pub fn decode_entities(s: &str) -> String {
149    let mut out = String::with_capacity(s.len());
150    let mut rest = s;
151    while let Some(amp) = rest.find('&') {
152        out.push_str(&rest[..amp]);
153        let tail = &rest[amp..];
154        let Some(semi) = tail.find(';').filter(|s| *s <= 12) else {
155            out.push('&');
156            rest = &tail[1..];
157            continue;
158        };
159        let entity = &tail[1..semi];
160        let decoded = match entity {
161            "amp" => Some('&'),
162            "lt" => Some('<'),
163            "gt" => Some('>'),
164            "quot" => Some('"'),
165            "apos" => Some('\''),
166            "nbsp" => Some(' '),
167            _ => entity
168                .strip_prefix('#')
169                .and_then(|n| match n.strip_prefix(['x', 'X']) {
170                    Some(hex) => u32::from_str_radix(hex, 16).ok(),
171                    None => n.parse::<u32>().ok(),
172                })
173                .and_then(char::from_u32),
174        };
175        match decoded {
176            // A non-breaking space is a space for comparison purposes.
177            Some('\u{a0}') => out.push(' '),
178            Some(c) => out.push(c),
179            None => {
180                out.push('&');
181                rest = &tail[1..];
182                continue;
183            }
184        }
185        rest = &tail[semi + 1..];
186    }
187    out.push_str(rest);
188    out
189}