Commit 69c12446b8
Verified · cmc
Layout: unified · split
examples/skeleton.rs added +46
| @@ -0,0 +1,46 @@ | ||
| 1 | //! Render an org file the way orgo would and print either its HTML or its semantic | |
| 2 | //! skeleton — the reference outputs the cross-language conformance corpus is built from. | |
| 3 | //! | |
| 4 | //! Usage: | |
| 5 | //! cargo run --example skeleton -- <file.org> # print the skeleton | |
| 6 | //! cargo run --example skeleton -- --html <file.org> # print the rendered HTML | |
| 7 | //! | |
| 8 | //! The skeleton is one line per element open, element close, or text run (see | |
| 9 | //! [`orgo::skeleton`]). Another implementation conforms when its own output, reduced by | |
| 10 | //! its own port of the same reduction, equals this. | |
| 11 | ||
| 12 | use camino::Utf8PathBuf; | |
| 13 | ||
| 14 | use orgo::parser::parse; | |
| 15 | use orgo::render::{render, Html, SyntectHighlighter}; | |
| 16 | use orgo::resolve::ResolvedDoc; | |
| 17 | use orgo::skeleton::skeleton; | |
| 18 | ||
| 19 | fn main() { | |
| 20 | let mut args = std::env::args().skip(1); | |
| 21 | let mut html_mode = false; | |
| 22 | let mut path: Option<String> = None; | |
| 23 | for arg in args.by_ref() { | |
| 24 | match arg.as_str() { | |
| 25 | "--html" => html_mode = true, | |
| 26 | _ => path = Some(arg), | |
| 27 | } | |
| 28 | } | |
| 29 | let Some(path) = path else { | |
| 30 | eprintln!("usage: skeleton [--html] <file.org>"); | |
| 31 | std::process::exit(2); | |
| 32 | }; | |
| 33 | ||
| 34 | let path = Utf8PathBuf::from(path); | |
| 35 | let source = std::fs::read_to_string(&path).expect("read org file"); | |
| 36 | let document = parse(path.as_path(), &source).expect("parse org file"); | |
| 37 | let Html(html) = render(&ResolvedDoc { document }, &SyntectHighlighter::new()); | |
| 38 | ||
| 39 | if html_mode { | |
| 40 | print!("{html}"); | |
| 41 | } else { | |
| 42 | for line in skeleton(&html) { | |
| 43 | println!("{line}"); | |
| 44 | } | |
| 45 | } | |
| 46 | } | |
src/lib.rs +1
| @@ -19,6 +19,7 @@ pub mod render; | ||
| 19 | 19 | pub mod resolve; |
| 20 | 20 | pub mod serve; |
| 21 | 21 | pub mod site; |
| 22 | pub mod skeleton; | |
| 22 | 23 | pub mod template; |
| 23 | 24 | pub mod theme; |
| 24 | 25 | pub mod util; |
src/skeleton.rs added +189
| @@ -0,0 +1,189 @@ | ||
| 1 | //! HTML → semantic skeleton reduction. | |
| 2 | //! | |
| 3 | //! Two HTML exporters that agree on what a document *means* will still disagree on how | |
| 4 | //! they wrap it: org buries every section in `outline-container` divs keyed by generated | |
| 5 | //! ids, syntect emits one `<span>` per code token, and each backend picks its own class | |
| 6 | //! names. Byte equality therefore measures nothing. The skeleton throws all of that away | |
| 7 | //! and keeps the part worth comparing: the ordered sequence of element opens, element | |
| 8 | //! closes, and text runs, with `<div>`/`<span>` dropped, every attribute except | |
| 9 | //! `href`/`src` removed, whitespace collapsed, and entities decoded. | |
| 10 | //! | |
| 11 | //! This is the comparison primitive behind two things at once: orgo's differential check | |
| 12 | //! against Emacs (`tests/oracle.rs`), and the cross-language conformance corpus, where an | |
| 13 | //! implementation in another language ports [`skeleton`] and asserts its output matches | |
| 14 | //! orgo's checked-in golden skeleton. Keeping one implementation here, rather than a copy | |
| 15 | //! per consumer, is the point — the reduction *is* the contract, so it must be identical | |
| 16 | //! everywhere. | |
| 17 | ||
| 18 | /// Elements dropped from the skeleton entirely, because once attributes are gone they | |
| 19 | /// carry no meaning two exporters could agree or disagree *about*. | |
| 20 | /// | |
| 21 | /// `div` is pure layout: org wraps every section in `outline-container`/`outline-text` | |
| 22 | /// wrappers and we emit none. `span` is the same story at the inline level, and matters | |
| 23 | /// far more than it looks: syntect emits one span per code token, so keeping them made a | |
| 24 | /// source block contribute ~60 skeleton lines of pure noise and dragged the agreement on | |
| 25 | /// `blocks.org` down to 36% — a number that said nothing about whether we render blocks | |
| 26 | /// correctly. Text still carries the signal: a `<span class="todo">` shows up as its | |
| 27 | /// text, `"TODO"`, which is the part worth comparing. | |
| 28 | const IGNORED: &[&str] = &["div", "span"]; | |
| 29 | ||
| 30 | /// Attributes kept in the skeleton. Ids and classes are generated (`org6c28c1b`) or | |
| 31 | /// cosmetic (`org-ul`); `href` and `src` are the content. | |
| 32 | const KEPT_ATTRS: &[&str] = &["href", "src"]; | |
| 33 | ||
| 34 | /// HTML void elements, which never emit a close event. | |
| 35 | const VOID: &[&str] = &[ | |
| 36 | "br", "hr", "img", "input", "meta", "link", "col", "area", "base", "source", "wbr", | |
| 37 | ]; | |
| 38 | ||
| 39 | /// Reduce an HTML fragment to its semantic skeleton: one line per element open, element | |
| 40 | /// close, or text run. | |
| 41 | pub fn skeleton(html: &str) -> Vec<String> { | |
| 42 | let mut out = Vec::new(); | |
| 43 | let chars: Vec<char> = html.chars().collect(); | |
| 44 | let mut i = 0; | |
| 45 | let mut text = String::new(); | |
| 46 | ||
| 47 | while i < chars.len() { | |
| 48 | if chars[i] != '<' { | |
| 49 | text.push(chars[i]); | |
| 50 | i += 1; | |
| 51 | continue; | |
| 52 | } | |
| 53 | ||
| 54 | // Comments and doctypes carry nothing. | |
| 55 | if chars[i..].starts_with(&['<', '!']) { | |
| 56 | i += match find_from(&chars, i, ">") { | |
| 57 | Some(end) => end - i + 1, | |
| 58 | None => break, | |
| 59 | }; | |
| 60 | continue; | |
| 61 | } | |
| 62 | let Some(end) = find_from(&chars, i, ">") else { | |
| 63 | break; | |
| 64 | }; | |
| 65 | let raw: String = chars[i + 1..end].iter().collect(); | |
| 66 | i = end + 1; | |
| 67 | ||
| 68 | let raw = raw.trim().trim_end_matches('/').trim().to_string(); | |
| 69 | // Text is flushed only when a tag is actually *emitted*. Text either side of an | |
| 70 | // ignored tag therefore merges into one run, which is what makes a highlighted | |
| 71 | // source block compare as the one string of code it is, rather than as a | |
| 72 | // token-by-token sequence that has to line up exactly. | |
| 73 | if let Some(name) = raw.strip_prefix('/') { | |
| 74 | let name = name.trim().to_ascii_lowercase(); | |
| 75 | if !IGNORED.contains(&name.as_str()) && !VOID.contains(&name.as_str()) { | |
| 76 | flush_text(&mut text, &mut out); | |
| 77 | out.push(format!("</{name}>")); | |
| 78 | } | |
| 79 | continue; | |
| 80 | } | |
| 81 | let mut parts = raw.splitn(2, char::is_whitespace); | |
| 82 | let name = parts.next().unwrap_or("").to_ascii_lowercase(); | |
| 83 | if name.is_empty() || IGNORED.contains(&name.as_str()) { | |
| 84 | continue; | |
| 85 | } | |
| 86 | let attrs = kept_attributes(parts.next().unwrap_or("")); | |
| 87 | flush_text(&mut text, &mut out); | |
| 88 | out.push(format!("<{name}{attrs}>")); | |
| 89 | } | |
| 90 | flush_text(&mut text, &mut out); | |
| 91 | out | |
| 92 | } | |
| 93 | ||
| 94 | fn flush_text(text: &mut String, out: &mut Vec<String>) { | |
| 95 | let decoded = decode_entities(text); | |
| 96 | let collapsed = decoded.split_whitespace().collect::<Vec<_>>().join(" "); | |
| 97 | if !collapsed.is_empty() { | |
| 98 | out.push(format!("{collapsed:?}")); | |
| 99 | } | |
| 100 | text.clear(); | |
| 101 | } | |
| 102 | ||
| 103 | fn find_from(chars: &[char], from: usize, needle: &str) -> Option<usize> { | |
| 104 | let n: Vec<char> = needle.chars().collect(); | |
| 105 | (from..chars.len()).find(|&k| chars[k..].starts_with(&n[..])) | |
| 106 | } | |
| 107 | ||
| 108 | /// Keep only the content-bearing attributes, in a stable order. | |
| 109 | fn kept_attributes(rest: &str) -> String { | |
| 110 | let mut kept: Vec<(String, String)> = Vec::new(); | |
| 111 | for attr in KEPT_ATTRS { | |
| 112 | if let Some(value) = attribute_value(rest, attr) { | |
| 113 | kept.push(((*attr).to_string(), value)); | |
| 114 | } | |
| 115 | } | |
| 116 | kept.iter() | |
| 117 | .map(|(k, v)| format!(" {k}=\"{}\"", decode_entities(v))) | |
| 118 | .collect() | |
| 119 | } | |
| 120 | ||
| 121 | fn attribute_value(rest: &str, name: &str) -> Option<String> { | |
| 122 | let mut search = rest; | |
| 123 | while let Some(pos) = search.find(name) { | |
| 124 | let before_ok = pos == 0 | |
| 125 | || search[..pos] | |
| 126 | .chars() | |
| 127 | .next_back() | |
| 128 | .is_some_and(char::is_whitespace); | |
| 129 | let after = &search[pos + name.len()..]; | |
| 130 | let after_trimmed = after.trim_start(); | |
| 131 | if before_ok && after_trimmed.starts_with('=') { | |
| 132 | let value = after_trimmed[1..].trim_start(); | |
| 133 | let quote = value.chars().next()?; | |
| 134 | if quote == '"' || quote == '\'' { | |
| 135 | let end = value[1..].find(quote)? + 1; | |
| 136 | return Some(value[1..end].to_string()); | |
| 137 | } | |
| 138 | let end = value.find(char::is_whitespace).unwrap_or(value.len()); | |
| 139 | return Some(value[..end].to_string()); | |
| 140 | } | |
| 141 | search = &search[pos + name.len()..]; | |
| 142 | } | |
| 143 | None | |
| 144 | } | |
| 145 | ||
| 146 | /// Decode the entities either exporter is likely to emit, so an encoding difference is | |
| 147 | /// never reported as a semantic one. | |
| 148 | pub fn decode_entities(s: &str) -> String { | |
| 149 | let mut out = String::with_capacity(s.len()); | |
| 150 | let mut rest = s; | |
| 151 | while let Some(amp) = rest.find('&') { | |
| 152 | out.push_str(&rest[..amp]); | |
| 153 | let tail = &rest[amp..]; | |
| 154 | let Some(semi) = tail.find(';').filter(|s| *s <= 12) else { | |
| 155 | out.push('&'); | |
| 156 | rest = &tail[1..]; | |
| 157 | continue; | |
| 158 | }; | |
| 159 | let entity = &tail[1..semi]; | |
| 160 | let decoded = match entity { | |
| 161 | "amp" => Some('&'), | |
| 162 | "lt" => Some('<'), | |
| 163 | "gt" => Some('>'), | |
| 164 | "quot" => Some('"'), | |
| 165 | "apos" => Some('\''), | |
| 166 | "nbsp" => Some(' '), | |
| 167 | _ => entity | |
| 168 | .strip_prefix('#') | |
| 169 | .and_then(|n| match n.strip_prefix(['x', 'X']) { | |
| 170 | Some(hex) => u32::from_str_radix(hex, 16).ok(), | |
| 171 | None => n.parse::<u32>().ok(), | |
| 172 | }) | |
| 173 | .and_then(char::from_u32), | |
| 174 | }; | |
| 175 | match decoded { | |
| 176 | // A non-breaking space is a space for comparison purposes. | |
| 177 | Some('\u{a0}') => out.push(' '), | |
| 178 | Some(c) => out.push(c), | |
| 179 | None => { | |
| 180 | out.push('&'); | |
| 181 | rest = &tail[1..]; | |
| 182 | continue; | |
| 183 | } | |
| 184 | } | |
| 185 | rest = &tail[semi + 1..]; | |
| 186 | } | |
| 187 | out.push_str(rest); | |
| 188 | out | |
| 189 | } | |
tests/conformance.rs added +73
| @@ -0,0 +1,73 @@ | ||
| 1 | //! orgo against the shared conformance corpus (`krz/org-conformance`). | |
| 2 | //! | |
| 3 | //! orgo *generates* that corpus's goldens, so this test is not looking for orgo to be | |
| 4 | //! wrong — it is the tripwire that fires when orgo's renderer changes and the checked-in | |
| 5 | //! corpus was not regenerated to match. Without it the two drift silently: orgo's own | |
| 6 | //! snapshots update with `cargo insta accept`, the corpus does not, and every other | |
| 7 | //! language is then conforming to a stale reference. | |
| 8 | //! | |
| 9 | //! The corpus lives in a sibling repo, so the test is opt-in: point `ORG_CONFORMANCE_DIR` | |
| 10 | //! at a checkout and it runs; unset, it skips cleanly, exactly like the Emacs oracle. | |
| 11 | //! | |
| 12 | //! ORG_CONFORMANCE_DIR=../org-conformance cargo test --test conformance | |
| 13 | ||
| 14 | use std::path::PathBuf; | |
| 15 | ||
| 16 | use camino::Utf8PathBuf; | |
| 17 | ||
| 18 | use orgo::parser::parse; | |
| 19 | use orgo::render::{render, Html, SyntectHighlighter}; | |
| 20 | use orgo::resolve::ResolvedDoc; | |
| 21 | use orgo::skeleton::skeleton; | |
| 22 | ||
| 23 | fn corpus_dir() -> Option<PathBuf> { | |
| 24 | let dir = PathBuf::from(std::env::var_os("ORG_CONFORMANCE_DIR")?); | |
| 25 | dir.join("cases").is_dir().then_some(dir) | |
| 26 | } | |
| 27 | ||
| 28 | fn render_skeleton(org_path: &std::path::Path) -> Vec<String> { | |
| 29 | let source = std::fs::read_to_string(org_path).expect("read case .org"); | |
| 30 | let rel = Utf8PathBuf::from(org_path.file_name().unwrap().to_string_lossy().into_owned()); | |
| 31 | let document = parse(rel.as_path(), &source).expect("parse case"); | |
| 32 | let Html(html) = render(&ResolvedDoc { document }, &SyntectHighlighter::new()); | |
| 33 | skeleton(&html) | |
| 34 | } | |
| 35 | ||
| 36 | #[test] | |
| 37 | fn orgo_matches_the_conformance_corpus() { | |
| 38 | let Some(dir) = corpus_dir() else { | |
| 39 | eprintln!("ORG_CONFORMANCE_DIR unset or has no cases/ — skipping conformance check"); | |
| 40 | return; | |
| 41 | }; | |
| 42 | let cases = dir.join("cases"); | |
| 43 | ||
| 44 | let mut checked = 0; | |
| 45 | let mut mismatches = Vec::new(); | |
| 46 | for entry in std::fs::read_dir(&cases).expect("read cases dir") { | |
| 47 | let path = entry.expect("dir entry").path(); | |
| 48 | if path.extension().and_then(|e| e.to_str()) != Some("org") { | |
| 49 | continue; | |
| 50 | } | |
| 51 | let name = path.file_stem().unwrap().to_string_lossy().into_owned(); | |
| 52 | let golden_path = cases.join(format!("{name}.skeleton")); | |
| 53 | let golden = std::fs::read_to_string(&golden_path) | |
| 54 | .unwrap_or_else(|_| panic!("missing golden skeleton for case {name}")); | |
| 55 | let golden_lines: Vec<&str> = golden.lines().collect(); | |
| 56 | ||
| 57 | let ours = render_skeleton(&path); | |
| 58 | if ours != golden_lines { | |
| 59 | mismatches.push(name); | |
| 60 | } | |
| 61 | checked += 1; | |
| 62 | } | |
| 63 | ||
| 64 | assert!(checked > 0, "corpus at {} has no .org cases", cases.display()); | |
| 65 | assert!( | |
| 66 | mismatches.is_empty(), | |
| 67 | "orgo's skeleton no longer matches the corpus goldens for: {}. \ | |
| 68 | If the renderer change is intended, regenerate the corpus \ | |
| 69 | (ORGO_DIR=../orgo tools/generate.sh) and commit it.", | |
| 70 | mismatches.join(", ") | |
| 71 | ); | |
| 72 | eprintln!("conformance: {checked} cases match the corpus goldens"); | |
| 73 | } | |
tests/oracle.rs +1 −177
| @@ -29,6 +29,7 @@ use camino::Utf8PathBuf; | ||
| 29 | 29 | use orgo::parser::parse; |
| 30 | 30 | use orgo::render::{render, Html, SyntectHighlighter}; |
| 31 | 31 | use orgo::resolve::ResolvedDoc; |
| 32 | use orgo::skeleton::{decode_entities, skeleton}; | |
| 32 | 33 | |
| 33 | 34 | fn manifest_dir() -> Utf8PathBuf { |
| 34 | 35 | Utf8PathBuf::from(env!("CARGO_MANIFEST_DIR")) |
| @@ -71,183 +72,6 @@ fn our_export(fixture: &str) -> String { | ||
| 71 | 72 | html |
| 72 | 73 | } |
| 73 | 74 | |
| 74 | // --------------------------------------------------------------------------- | |
| 75 | // HTML → semantic skeleton | |
| 76 | // --------------------------------------------------------------------------- | |
| 77 | ||
| 78 | /// Elements dropped from the skeleton entirely, because once attributes are gone they | |
| 79 | /// carry no meaning the two exporters could agree or disagree *about*. | |
| 80 | /// | |
| 81 | /// `div` is pure layout: org wraps every section in `outline-container`/`outline-text` | |
| 82 | /// wrappers and we emit none. `span` is the same story at the inline level, and matters | |
| 83 | /// far more than it looks: syntect emits one span per code token, so keeping them made a | |
| 84 | /// source block contribute ~60 skeleton lines of pure noise and dragged the agreement on | |
| 85 | /// `blocks.org` down to 36% — a number that said nothing about whether we render blocks | |
| 86 | /// correctly. Text still carries the signal: a `<span class="todo">` shows up as its | |
| 87 | /// text, `"TODO"`, which is the part worth comparing. | |
| 88 | const IGNORED: &[&str] = &["div", "span"]; | |
| 89 | ||
| 90 | /// Attributes kept in the skeleton. Ids and classes are generated (`org6c28c1b`) or | |
| 91 | /// cosmetic (`org-ul`); `href` and `src` are the content. | |
| 92 | const KEPT_ATTRS: &[&str] = &["href", "src"]; | |
| 93 | ||
| 94 | /// HTML void elements, which never emit a close event. | |
| 95 | const VOID: &[&str] = &[ | |
| 96 | "br", "hr", "img", "input", "meta", "link", "col", "area", "base", "source", "wbr", | |
| 97 | ]; | |
| 98 | ||
| 99 | /// Reduce an HTML fragment to its semantic skeleton: one line per element open, element | |
| 100 | /// close, or text run. | |
| 101 | fn skeleton(html: &str) -> Vec<String> { | |
| 102 | let mut out = Vec::new(); | |
| 103 | let chars: Vec<char> = html.chars().collect(); | |
| 104 | let mut i = 0; | |
| 105 | let mut text = String::new(); | |
| 106 | ||
| 107 | while i < chars.len() { | |
| 108 | if chars[i] != '<' { | |
| 109 | text.push(chars[i]); | |
| 110 | i += 1; | |
| 111 | continue; | |
| 112 | } | |
| 113 | ||
| 114 | // Comments and doctypes carry nothing. | |
| 115 | if chars[i..].starts_with(&['<', '!']) { | |
| 116 | i += match find_from(&chars, i, ">") { | |
| 117 | Some(end) => end - i + 1, | |
| 118 | None => break, | |
| 119 | }; | |
| 120 | continue; | |
| 121 | } | |
| 122 | let Some(end) = find_from(&chars, i, ">") else { | |
| 123 | break; | |
| 124 | }; | |
| 125 | let raw: String = chars[i + 1..end].iter().collect(); | |
| 126 | i = end + 1; | |
| 127 | ||
| 128 | let raw = raw.trim().trim_end_matches('/').trim().to_string(); | |
| 129 | // Text is flushed only when a tag is actually *emitted*. Text either side of an | |
| 130 | // ignored tag therefore merges into one run, which is what makes a highlighted | |
| 131 | // source block compare as the one string of code it is, rather than as a | |
| 132 | // token-by-token sequence that has to line up exactly. | |
| 133 | if let Some(name) = raw.strip_prefix('/') { | |
| 134 | let name = name.trim().to_ascii_lowercase(); | |
| 135 | if !IGNORED.contains(&name.as_str()) && !VOID.contains(&name.as_str()) { | |
| 136 | flush_text(&mut text, &mut out); | |
| 137 | out.push(format!("</{name}>")); | |
| 138 | } | |
| 139 | continue; | |
| 140 | } | |
| 141 | let mut parts = raw.splitn(2, char::is_whitespace); | |
| 142 | let name = parts.next().unwrap_or("").to_ascii_lowercase(); | |
| 143 | if name.is_empty() || IGNORED.contains(&name.as_str()) { | |
| 144 | continue; | |
| 145 | } | |
| 146 | let attrs = kept_attributes(parts.next().unwrap_or("")); | |
| 147 | flush_text(&mut text, &mut out); | |
| 148 | out.push(format!("<{name}{attrs}>")); | |
| 149 | } | |
| 150 | flush_text(&mut text, &mut out); | |
| 151 | out | |
| 152 | } | |
| 153 | ||
| 154 | fn flush_text(text: &mut String, out: &mut Vec<String>) { | |
| 155 | let decoded = decode_entities(text); | |
| 156 | let collapsed = decoded.split_whitespace().collect::<Vec<_>>().join(" "); | |
| 157 | if !collapsed.is_empty() { | |
| 158 | out.push(format!("{collapsed:?}")); | |
| 159 | } | |
| 160 | text.clear(); | |
| 161 | } | |
| 162 | ||
| 163 | fn find_from(chars: &[char], from: usize, needle: &str) -> Option<usize> { | |
| 164 | let n: Vec<char> = needle.chars().collect(); | |
| 165 | (from..chars.len()).find(|&k| chars[k..].starts_with(&n[..])) | |
| 166 | } | |
| 167 | ||
| 168 | /// Keep only the content-bearing attributes, in a stable order. | |
| 169 | fn kept_attributes(rest: &str) -> String { | |
| 170 | let mut kept: Vec<(String, String)> = Vec::new(); | |
| 171 | for attr in KEPT_ATTRS { | |
| 172 | if let Some(value) = attribute_value(rest, attr) { | |
| 173 | kept.push(((*attr).to_string(), value)); | |
| 174 | } | |
| 175 | } | |
| 176 | kept.iter() | |
| 177 | .map(|(k, v)| format!(" {k}=\"{}\"", decode_entities(v))) | |
| 178 | .collect() | |
| 179 | } | |
| 180 | ||
| 181 | fn attribute_value(rest: &str, name: &str) -> Option<String> { | |
| 182 | let mut search = rest; | |
| 183 | while let Some(pos) = search.find(name) { | |
| 184 | let before_ok = pos == 0 | |
| 185 | || search[..pos] | |
| 186 | .chars() | |
| 187 | .next_back() | |
| 188 | .is_some_and(char::is_whitespace); | |
| 189 | let after = &search[pos + name.len()..]; | |
| 190 | let after_trimmed = after.trim_start(); | |
| 191 | if before_ok && after_trimmed.starts_with('=') { | |
| 192 | let value = after_trimmed[1..].trim_start(); | |
| 193 | let quote = value.chars().next()?; | |
| 194 | if quote == '"' || quote == '\'' { | |
| 195 | let end = value[1..].find(quote)? + 1; | |
| 196 | return Some(value[1..end].to_string()); | |
| 197 | } | |
| 198 | let end = value.find(char::is_whitespace).unwrap_or(value.len()); | |
| 199 | return Some(value[..end].to_string()); | |
| 200 | } | |
| 201 | search = &search[pos + name.len()..]; | |
| 202 | } | |
| 203 | None | |
| 204 | } | |
| 205 | ||
| 206 | /// Decode the entities either exporter is likely to emit, so an encoding difference is | |
| 207 | /// never reported as a semantic one. | |
| 208 | fn decode_entities(s: &str) -> String { | |
| 209 | let mut out = String::with_capacity(s.len()); | |
| 210 | let mut rest = s; | |
| 211 | while let Some(amp) = rest.find('&') { | |
| 212 | out.push_str(&rest[..amp]); | |
| 213 | let tail = &rest[amp..]; | |
| 214 | let Some(semi) = tail.find(';').filter(|s| *s <= 12) else { | |
| 215 | out.push('&'); | |
| 216 | rest = &tail[1..]; | |
| 217 | continue; | |
| 218 | }; | |
| 219 | let entity = &tail[1..semi]; | |
| 220 | let decoded = match entity { | |
| 221 | "amp" => Some('&'), | |
| 222 | "lt" => Some('<'), | |
| 223 | "gt" => Some('>'), | |
| 224 | "quot" => Some('"'), | |
| 225 | "apos" => Some('\''), | |
| 226 | "nbsp" => Some(' '), | |
| 227 | _ => entity | |
| 228 | .strip_prefix('#') | |
| 229 | .and_then(|n| match n.strip_prefix(['x', 'X']) { | |
| 230 | Some(hex) => u32::from_str_radix(hex, 16).ok(), | |
| 231 | None => n.parse::<u32>().ok(), | |
| 232 | }) | |
| 233 | .and_then(char::from_u32), | |
| 234 | }; | |
| 235 | match decoded { | |
| 236 | // A non-breaking space is a space for comparison purposes. | |
| 237 | Some('\u{a0}') => out.push(' '), | |
| 238 | Some(c) => out.push(c), | |
| 239 | None => { | |
| 240 | out.push('&'); | |
| 241 | rest = &tail[1..]; | |
| 242 | continue; | |
| 243 | } | |
| 244 | } | |
| 245 | rest = &tail[semi + 1..]; | |
| 246 | } | |
| 247 | out.push_str(rest); | |
| 248 | out | |
| 249 | } | |
| 250 | ||
| 251 | 75 | // --------------------------------------------------------------------------- |
| 252 | 76 | // Divergence report |
| 253 | 77 | // --------------------------------------------------------------------------- |