Commit 69c12446b8
Verified · cmc
Layout: unified · split
examples/skeleton.rs added +46
| @@ -0,0 +1,46 @@ | |||
| 1 | //! Render an org file the way orgo would and print either its HTML or its semantic | ||
| 2 | //! skeleton — the reference outputs the cross-language conformance corpus is built from. | ||
| 3 | //! | ||
| 4 | //! Usage: | ||
| 5 | //! cargo run --example skeleton -- <file.org> # print the skeleton | ||
| 6 | //! cargo run --example skeleton -- --html <file.org> # print the rendered HTML | ||
| 7 | //! | ||
| 8 | //! The skeleton is one line per element open, element close, or text run (see | ||
| 9 | //! [`orgo::skeleton`]). Another implementation conforms when its own output, reduced by | ||
| 10 | //! its own port of the same reduction, equals this. | ||
| 11 | |||
| 12 | use camino::Utf8PathBuf; | ||
| 13 | |||
| 14 | use orgo::parser::parse; | ||
| 15 | use orgo::render::{render, Html, SyntectHighlighter}; | ||
| 16 | use orgo::resolve::ResolvedDoc; | ||
| 17 | use orgo::skeleton::skeleton; | ||
| 18 | |||
| 19 | fn main() { | ||
| 20 | let mut args = std::env::args().skip(1); | ||
| 21 | let mut html_mode = false; | ||
| 22 | let mut path: Option<String> = None; | ||
| 23 | for arg in args.by_ref() { | ||
| 24 | match arg.as_str() { | ||
| 25 | "--html" => html_mode = true, | ||
| 26 | _ => path = Some(arg), | ||
| 27 | } | ||
| 28 | } | ||
| 29 | let Some(path) = path else { | ||
| 30 | eprintln!("usage: skeleton [--html] <file.org>"); | ||
| 31 | std::process::exit(2); | ||
| 32 | }; | ||
| 33 | |||
| 34 | let path = Utf8PathBuf::from(path); | ||
| 35 | let source = std::fs::read_to_string(&path).expect("read org file"); | ||
| 36 | let document = parse(path.as_path(), &source).expect("parse org file"); | ||
| 37 | let Html(html) = render(&ResolvedDoc { document }, &SyntectHighlighter::new()); | ||
| 38 | |||
| 39 | if html_mode { | ||
| 40 | print!("{html}"); | ||
| 41 | } else { | ||
| 42 | for line in skeleton(&html) { | ||
| 43 | println!("{line}"); | ||
| 44 | } | ||
| 45 | } | ||
| 46 | } | ||
src/lib.rs +1
| @@ -19,6 +19,7 @@ pub mod render; | |||
| 19 | pub mod resolve; | 19 | pub mod resolve; |
| 20 | pub mod serve; | 20 | pub mod serve; |
| 21 | pub mod site; | 21 | pub mod site; |
| 22 | pub mod skeleton; | ||
| 22 | pub mod template; | 23 | pub mod template; |
| 23 | pub mod theme; | 24 | pub mod theme; |
| 24 | pub mod util; | 25 | pub mod util; |
src/skeleton.rs added +189
| @@ -0,0 +1,189 @@ | |||
| 1 | //! HTML → semantic skeleton reduction. | ||
| 2 | //! | ||
| 3 | //! Two HTML exporters that agree on what a document *means* will still disagree on how | ||
| 4 | //! they wrap it: org buries every section in `outline-container` divs keyed by generated | ||
| 5 | //! ids, syntect emits one `<span>` per code token, and each backend picks its own class | ||
| 6 | //! names. Byte equality therefore measures nothing. The skeleton throws all of that away | ||
| 7 | //! and keeps the part worth comparing: the ordered sequence of element opens, element | ||
| 8 | //! closes, and text runs, with `<div>`/`<span>` dropped, every attribute except | ||
| 9 | //! `href`/`src` removed, whitespace collapsed, and entities decoded. | ||
| 10 | //! | ||
| 11 | //! This is the comparison primitive behind two things at once: orgo's differential check | ||
| 12 | //! against Emacs (`tests/oracle.rs`), and the cross-language conformance corpus, where an | ||
| 13 | //! implementation in another language ports [`skeleton`] and asserts its output matches | ||
| 14 | //! orgo's checked-in golden skeleton. Keeping one implementation here, rather than a copy | ||
| 15 | //! per consumer, is the point — the reduction *is* the contract, so it must be identical | ||
| 16 | //! everywhere. | ||
| 17 | |||
| 18 | /// Elements dropped from the skeleton entirely, because once attributes are gone they | ||
| 19 | /// carry no meaning two exporters could agree or disagree *about*. | ||
| 20 | /// | ||
| 21 | /// `div` is pure layout: org wraps every section in `outline-container`/`outline-text` | ||
| 22 | /// wrappers and we emit none. `span` is the same story at the inline level, and matters | ||
| 23 | /// far more than it looks: syntect emits one span per code token, so keeping them made a | ||
| 24 | /// source block contribute ~60 skeleton lines of pure noise and dragged the agreement on | ||
| 25 | /// `blocks.org` down to 36% — a number that said nothing about whether we render blocks | ||
| 26 | /// correctly. Text still carries the signal: a `<span class="todo">` shows up as its | ||
| 27 | /// text, `"TODO"`, which is the part worth comparing. | ||
| 28 | const IGNORED: &[&str] = &["div", "span"]; | ||
| 29 | |||
| 30 | /// Attributes kept in the skeleton. Ids and classes are generated (`org6c28c1b`) or | ||
| 31 | /// cosmetic (`org-ul`); `href` and `src` are the content. | ||
| 32 | const KEPT_ATTRS: &[&str] = &["href", "src"]; | ||
| 33 | |||
| 34 | /// HTML void elements, which never emit a close event. | ||
| 35 | const VOID: &[&str] = &[ | ||
| 36 | "br", "hr", "img", "input", "meta", "link", "col", "area", "base", "source", "wbr", | ||
| 37 | ]; | ||
| 38 | |||
| 39 | /// Reduce an HTML fragment to its semantic skeleton: one line per element open, element | ||
| 40 | /// close, or text run. | ||
| 41 | pub fn skeleton(html: &str) -> Vec<String> { | ||
| 42 | let mut out = Vec::new(); | ||
| 43 | let chars: Vec<char> = html.chars().collect(); | ||
| 44 | let mut i = 0; | ||
| 45 | let mut text = String::new(); | ||
| 46 | |||
| 47 | while i < chars.len() { | ||
| 48 | if chars[i] != '<' { | ||
| 49 | text.push(chars[i]); | ||
| 50 | i += 1; | ||
| 51 | continue; | ||
| 52 | } | ||
| 53 | |||
| 54 | // Comments and doctypes carry nothing. | ||
| 55 | if chars[i..].starts_with(&['<', '!']) { | ||
| 56 | i += match find_from(&chars, i, ">") { | ||
| 57 | Some(end) => end - i + 1, | ||
| 58 | None => break, | ||
| 59 | }; | ||
| 60 | continue; | ||
| 61 | } | ||
| 62 | let Some(end) = find_from(&chars, i, ">") else { | ||
| 63 | break; | ||
| 64 | }; | ||
| 65 | let raw: String = chars[i + 1..end].iter().collect(); | ||
| 66 | i = end + 1; | ||
| 67 | |||
| 68 | let raw = raw.trim().trim_end_matches('/').trim().to_string(); | ||
| 69 | // Text is flushed only when a tag is actually *emitted*. Text either side of an | ||
| 70 | // ignored tag therefore merges into one run, which is what makes a highlighted | ||
| 71 | // source block compare as the one string of code it is, rather than as a | ||
| 72 | // token-by-token sequence that has to line up exactly. | ||
| 73 | if let Some(name) = raw.strip_prefix('/') { | ||
| 74 | let name = name.trim().to_ascii_lowercase(); | ||
| 75 | if !IGNORED.contains(&name.as_str()) && !VOID.contains(&name.as_str()) { | ||
| 76 | flush_text(&mut text, &mut out); | ||
| 77 | out.push(format!("</{name}>")); | ||
| 78 | } | ||
| 79 | continue; | ||
| 80 | } | ||
| 81 | let mut parts = raw.splitn(2, char::is_whitespace); | ||
| 82 | let name = parts.next().unwrap_or("").to_ascii_lowercase(); | ||
| 83 | if name.is_empty() || IGNORED.contains(&name.as_str()) { | ||
| 84 | continue; | ||
| 85 | } | ||
| 86 | let attrs = kept_attributes(parts.next().unwrap_or("")); | ||
| 87 | flush_text(&mut text, &mut out); | ||
| 88 | out.push(format!("<{name}{attrs}>")); | ||
| 89 | } | ||
| 90 | flush_text(&mut text, &mut out); | ||
| 91 | out | ||
| 92 | } | ||
| 93 | |||
| 94 | fn flush_text(text: &mut String, out: &mut Vec<String>) { | ||
| 95 | let decoded = decode_entities(text); | ||
| 96 | let collapsed = decoded.split_whitespace().collect::<Vec<_>>().join(" "); | ||
| 97 | if !collapsed.is_empty() { | ||
| 98 | out.push(format!("{collapsed:?}")); | ||
| 99 | } | ||
| 100 | text.clear(); | ||
| 101 | } | ||
| 102 | |||
| 103 | fn find_from(chars: &[char], from: usize, needle: &str) -> Option<usize> { | ||
| 104 | let n: Vec<char> = needle.chars().collect(); | ||
| 105 | (from..chars.len()).find(|&k| chars[k..].starts_with(&n[..])) | ||
| 106 | } | ||
| 107 | |||
| 108 | /// Keep only the content-bearing attributes, in a stable order. | ||
| 109 | fn kept_attributes(rest: &str) -> String { | ||
| 110 | let mut kept: Vec<(String, String)> = Vec::new(); | ||
| 111 | for attr in KEPT_ATTRS { | ||
| 112 | if let Some(value) = attribute_value(rest, attr) { | ||
| 113 | kept.push(((*attr).to_string(), value)); | ||
| 114 | } | ||
| 115 | } | ||
| 116 | kept.iter() | ||
| 117 | .map(|(k, v)| format!(" {k}=\"{}\"", decode_entities(v))) | ||
| 118 | .collect() | ||
| 119 | } | ||
| 120 | |||
| 121 | fn attribute_value(rest: &str, name: &str) -> Option<String> { | ||
| 122 | let mut search = rest; | ||
| 123 | while let Some(pos) = search.find(name) { | ||
| 124 | let before_ok = pos == 0 | ||
| 125 | || search[..pos] | ||
| 126 | .chars() | ||
| 127 | .next_back() | ||
| 128 | .is_some_and(char::is_whitespace); | ||
| 129 | let after = &search[pos + name.len()..]; | ||
| 130 | let after_trimmed = after.trim_start(); | ||
| 131 | if before_ok && after_trimmed.starts_with('=') { | ||
| 132 | let value = after_trimmed[1..].trim_start(); | ||
| 133 | let quote = value.chars().next()?; | ||
| 134 | if quote == '"' || quote == '\'' { | ||
| 135 | let end = value[1..].find(quote)? + 1; | ||
| 136 | return Some(value[1..end].to_string()); | ||
| 137 | } | ||
| 138 | let end = value.find(char::is_whitespace).unwrap_or(value.len()); | ||
| 139 | return Some(value[..end].to_string()); | ||
| 140 | } | ||
| 141 | search = &search[pos + name.len()..]; | ||
| 142 | } | ||
| 143 | None | ||
| 144 | } | ||
| 145 | |||
| 146 | /// Decode the entities either exporter is likely to emit, so an encoding difference is | ||
| 147 | /// never reported as a semantic one. | ||
| 148 | pub fn decode_entities(s: &str) -> String { | ||
| 149 | let mut out = String::with_capacity(s.len()); | ||
| 150 | let mut rest = s; | ||
| 151 | while let Some(amp) = rest.find('&') { | ||
| 152 | out.push_str(&rest[..amp]); | ||
| 153 | let tail = &rest[amp..]; | ||
| 154 | let Some(semi) = tail.find(';').filter(|s| *s <= 12) else { | ||
| 155 | out.push('&'); | ||
| 156 | rest = &tail[1..]; | ||
| 157 | continue; | ||
| 158 | }; | ||
| 159 | let entity = &tail[1..semi]; | ||
| 160 | let decoded = match entity { | ||
| 161 | "amp" => Some('&'), | ||
| 162 | "lt" => Some('<'), | ||
| 163 | "gt" => Some('>'), | ||
| 164 | "quot" => Some('"'), | ||
| 165 | "apos" => Some('\''), | ||
| 166 | "nbsp" => Some(' '), | ||
| 167 | _ => entity | ||
| 168 | .strip_prefix('#') | ||
| 169 | .and_then(|n| match n.strip_prefix(['x', 'X']) { | ||
| 170 | Some(hex) => u32::from_str_radix(hex, 16).ok(), | ||
| 171 | None => n.parse::<u32>().ok(), | ||
| 172 | }) | ||
| 173 | .and_then(char::from_u32), | ||
| 174 | }; | ||
| 175 | match decoded { | ||
| 176 | // A non-breaking space is a space for comparison purposes. | ||
| 177 | Some('\u{a0}') => out.push(' '), | ||
| 178 | Some(c) => out.push(c), | ||
| 179 | None => { | ||
| 180 | out.push('&'); | ||
| 181 | rest = &tail[1..]; | ||
| 182 | continue; | ||
| 183 | } | ||
| 184 | } | ||
| 185 | rest = &tail[semi + 1..]; | ||
| 186 | } | ||
| 187 | out.push_str(rest); | ||
| 188 | out | ||
| 189 | } | ||
tests/conformance.rs added +73
| @@ -0,0 +1,73 @@ | |||
| 1 | //! orgo against the shared conformance corpus (`krz/org-conformance`). | ||
| 2 | //! | ||
| 3 | //! orgo *generates* that corpus's goldens, so this test is not looking for orgo to be | ||
| 4 | //! wrong — it is the tripwire that fires when orgo's renderer changes and the checked-in | ||
| 5 | //! corpus was not regenerated to match. Without it the two drift silently: orgo's own | ||
| 6 | //! snapshots update with `cargo insta accept`, the corpus does not, and every other | ||
| 7 | //! language is then conforming to a stale reference. | ||
| 8 | //! | ||
| 9 | //! The corpus lives in a sibling repo, so the test is opt-in: point `ORG_CONFORMANCE_DIR` | ||
| 10 | //! at a checkout and it runs; unset, it skips cleanly, exactly like the Emacs oracle. | ||
| 11 | //! | ||
| 12 | //! ORG_CONFORMANCE_DIR=../org-conformance cargo test --test conformance | ||
| 13 | |||
| 14 | use std::path::PathBuf; | ||
| 15 | |||
| 16 | use camino::Utf8PathBuf; | ||
| 17 | |||
| 18 | use orgo::parser::parse; | ||
| 19 | use orgo::render::{render, Html, SyntectHighlighter}; | ||
| 20 | use orgo::resolve::ResolvedDoc; | ||
| 21 | use orgo::skeleton::skeleton; | ||
| 22 | |||
| 23 | fn corpus_dir() -> Option<PathBuf> { | ||
| 24 | let dir = PathBuf::from(std::env::var_os("ORG_CONFORMANCE_DIR")?); | ||
| 25 | dir.join("cases").is_dir().then_some(dir) | ||
| 26 | } | ||
| 27 | |||
| 28 | fn render_skeleton(org_path: &std::path::Path) -> Vec<String> { | ||
| 29 | let source = std::fs::read_to_string(org_path).expect("read case .org"); | ||
| 30 | let rel = Utf8PathBuf::from(org_path.file_name().unwrap().to_string_lossy().into_owned()); | ||
| 31 | let document = parse(rel.as_path(), &source).expect("parse case"); | ||
| 32 | let Html(html) = render(&ResolvedDoc { document }, &SyntectHighlighter::new()); | ||
| 33 | skeleton(&html) | ||
| 34 | } | ||
| 35 | |||
| 36 | #[test] | ||
| 37 | fn orgo_matches_the_conformance_corpus() { | ||
| 38 | let Some(dir) = corpus_dir() else { | ||
| 39 | eprintln!("ORG_CONFORMANCE_DIR unset or has no cases/ — skipping conformance check"); | ||
| 40 | return; | ||
| 41 | }; | ||
| 42 | let cases = dir.join("cases"); | ||
| 43 | |||
| 44 | let mut checked = 0; | ||
| 45 | let mut mismatches = Vec::new(); | ||
| 46 | for entry in std::fs::read_dir(&cases).expect("read cases dir") { | ||
| 47 | let path = entry.expect("dir entry").path(); | ||
| 48 | if path.extension().and_then(|e| e.to_str()) != Some("org") { | ||
| 49 | continue; | ||
| 50 | } | ||
| 51 | let name = path.file_stem().unwrap().to_string_lossy().into_owned(); | ||
| 52 | let golden_path = cases.join(format!("{name}.skeleton")); | ||
| 53 | let golden = std::fs::read_to_string(&golden_path) | ||
| 54 | .unwrap_or_else(|_| panic!("missing golden skeleton for case {name}")); | ||
| 55 | let golden_lines: Vec<&str> = golden.lines().collect(); | ||
| 56 | |||
| 57 | let ours = render_skeleton(&path); | ||
| 58 | if ours != golden_lines { | ||
| 59 | mismatches.push(name); | ||
| 60 | } | ||
| 61 | checked += 1; | ||
| 62 | } | ||
| 63 | |||
| 64 | assert!(checked > 0, "corpus at {} has no .org cases", cases.display()); | ||
| 65 | assert!( | ||
| 66 | mismatches.is_empty(), | ||
| 67 | "orgo's skeleton no longer matches the corpus goldens for: {}. \ | ||
| 68 | If the renderer change is intended, regenerate the corpus \ | ||
| 69 | (ORGO_DIR=../orgo tools/generate.sh) and commit it.", | ||
| 70 | mismatches.join(", ") | ||
| 71 | ); | ||
| 72 | eprintln!("conformance: {checked} cases match the corpus goldens"); | ||
| 73 | } | ||
tests/oracle.rs +1 −177
| @@ -29,6 +29,7 @@ use camino::Utf8PathBuf; | |||
| 29 | use orgo::parser::parse; | 29 | use orgo::parser::parse; |
| 30 | use orgo::render::{render, Html, SyntectHighlighter}; | 30 | use orgo::render::{render, Html, SyntectHighlighter}; |
| 31 | use orgo::resolve::ResolvedDoc; | 31 | use orgo::resolve::ResolvedDoc; |
| 32 | use orgo::skeleton::{decode_entities, skeleton}; | ||
| 32 | 33 | ||
| 33 | fn manifest_dir() -> Utf8PathBuf { | 34 | fn manifest_dir() -> Utf8PathBuf { |
| 34 | Utf8PathBuf::from(env!("CARGO_MANIFEST_DIR")) | 35 | Utf8PathBuf::from(env!("CARGO_MANIFEST_DIR")) |
| @@ -71,183 +72,6 @@ fn our_export(fixture: &str) -> String { | |||
| 71 | html | 72 | html |
| 72 | } | 73 | } |
| 73 | 74 | ||
| 74 | // --------------------------------------------------------------------------- | ||
| 75 | // HTML → semantic skeleton | ||
| 76 | // --------------------------------------------------------------------------- | ||
| 77 | |||
| 78 | /// Elements dropped from the skeleton entirely, because once attributes are gone they | ||
| 79 | /// carry no meaning the two exporters could agree or disagree *about*. | ||
| 80 | /// | ||
| 81 | /// `div` is pure layout: org wraps every section in `outline-container`/`outline-text` | ||
| 82 | /// wrappers and we emit none. `span` is the same story at the inline level, and matters | ||
| 83 | /// far more than it looks: syntect emits one span per code token, so keeping them made a | ||
| 84 | /// source block contribute ~60 skeleton lines of pure noise and dragged the agreement on | ||
| 85 | /// `blocks.org` down to 36% — a number that said nothing about whether we render blocks | ||
| 86 | /// correctly. Text still carries the signal: a `<span class="todo">` shows up as its | ||
| 87 | /// text, `"TODO"`, which is the part worth comparing. | ||
| 88 | const IGNORED: &[&str] = &["div", "span"]; | ||
| 89 | |||
| 90 | /// Attributes kept in the skeleton. Ids and classes are generated (`org6c28c1b`) or | ||
| 91 | /// cosmetic (`org-ul`); `href` and `src` are the content. | ||
| 92 | const KEPT_ATTRS: &[&str] = &["href", "src"]; | ||
| 93 | |||
| 94 | /// HTML void elements, which never emit a close event. | ||
| 95 | const VOID: &[&str] = &[ | ||
| 96 | "br", "hr", "img", "input", "meta", "link", "col", "area", "base", "source", "wbr", | ||
| 97 | ]; | ||
| 98 | |||
| 99 | /// Reduce an HTML fragment to its semantic skeleton: one line per element open, element | ||
| 100 | /// close, or text run. | ||
| 101 | fn skeleton(html: &str) -> Vec<String> { | ||
| 102 | let mut out = Vec::new(); | ||
| 103 | let chars: Vec<char> = html.chars().collect(); | ||
| 104 | let mut i = 0; | ||
| 105 | let mut text = String::new(); | ||
| 106 | |||
| 107 | while i < chars.len() { | ||
| 108 | if chars[i] != '<' { | ||
| 109 | text.push(chars[i]); | ||
| 110 | i += 1; | ||
| 111 | continue; | ||
| 112 | } | ||
| 113 | |||
| 114 | // Comments and doctypes carry nothing. | ||
| 115 | if chars[i..].starts_with(&['<', '!']) { | ||
| 116 | i += match find_from(&chars, i, ">") { | ||
| 117 | Some(end) => end - i + 1, | ||
| 118 | None => break, | ||
| 119 | }; | ||
| 120 | continue; | ||
| 121 | } | ||
| 122 | let Some(end) = find_from(&chars, i, ">") else { | ||
| 123 | break; | ||
| 124 | }; | ||
| 125 | let raw: String = chars[i + 1..end].iter().collect(); | ||
| 126 | i = end + 1; | ||
| 127 | |||
| 128 | let raw = raw.trim().trim_end_matches('/').trim().to_string(); | ||
| 129 | // Text is flushed only when a tag is actually *emitted*. Text either side of an | ||
| 130 | // ignored tag therefore merges into one run, which is what makes a highlighted | ||
| 131 | // source block compare as the one string of code it is, rather than as a | ||
| 132 | // token-by-token sequence that has to line up exactly. | ||
| 133 | if let Some(name) = raw.strip_prefix('/') { | ||
| 134 | let name = name.trim().to_ascii_lowercase(); | ||
| 135 | if !IGNORED.contains(&name.as_str()) && !VOID.contains(&name.as_str()) { | ||
| 136 | flush_text(&mut text, &mut out); | ||
| 137 | out.push(format!("</{name}>")); | ||
| 138 | } | ||
| 139 | continue; | ||
| 140 | } | ||
| 141 | let mut parts = raw.splitn(2, char::is_whitespace); | ||
| 142 | let name = parts.next().unwrap_or("").to_ascii_lowercase(); | ||
| 143 | if name.is_empty() || IGNORED.contains(&name.as_str()) { | ||
| 144 | continue; | ||
| 145 | } | ||
| 146 | let attrs = kept_attributes(parts.next().unwrap_or("")); | ||
| 147 | flush_text(&mut text, &mut out); | ||
| 148 | out.push(format!("<{name}{attrs}>")); | ||
| 149 | } | ||
| 150 | flush_text(&mut text, &mut out); | ||
| 151 | out | ||
| 152 | } | ||
| 153 | |||
| 154 | fn flush_text(text: &mut String, out: &mut Vec<String>) { | ||
| 155 | let decoded = decode_entities(text); | ||
| 156 | let collapsed = decoded.split_whitespace().collect::<Vec<_>>().join(" "); | ||
| 157 | if !collapsed.is_empty() { | ||
| 158 | out.push(format!("{collapsed:?}")); | ||
| 159 | } | ||
| 160 | text.clear(); | ||
| 161 | } | ||
| 162 | |||
| 163 | fn find_from(chars: &[char], from: usize, needle: &str) -> Option<usize> { | ||
| 164 | let n: Vec<char> = needle.chars().collect(); | ||
| 165 | (from..chars.len()).find(|&k| chars[k..].starts_with(&n[..])) | ||
| 166 | } | ||
| 167 | |||
| 168 | /// Keep only the content-bearing attributes, in a stable order. | ||
| 169 | fn kept_attributes(rest: &str) -> String { | ||
| 170 | let mut kept: Vec<(String, String)> = Vec::new(); | ||
| 171 | for attr in KEPT_ATTRS { | ||
| 172 | if let Some(value) = attribute_value(rest, attr) { | ||
| 173 | kept.push(((*attr).to_string(), value)); | ||
| 174 | } | ||
| 175 | } | ||
| 176 | kept.iter() | ||
| 177 | .map(|(k, v)| format!(" {k}=\"{}\"", decode_entities(v))) | ||
| 178 | .collect() | ||
| 179 | } | ||
| 180 | |||
| 181 | fn attribute_value(rest: &str, name: &str) -> Option<String> { | ||
| 182 | let mut search = rest; | ||
| 183 | while let Some(pos) = search.find(name) { | ||
| 184 | let before_ok = pos == 0 | ||
| 185 | || search[..pos] | ||
| 186 | .chars() | ||
| 187 | .next_back() | ||
| 188 | .is_some_and(char::is_whitespace); | ||
| 189 | let after = &search[pos + name.len()..]; | ||
| 190 | let after_trimmed = after.trim_start(); | ||
| 191 | if before_ok && after_trimmed.starts_with('=') { | ||
| 192 | let value = after_trimmed[1..].trim_start(); | ||
| 193 | let quote = value.chars().next()?; | ||
| 194 | if quote == '"' || quote == '\'' { | ||
| 195 | let end = value[1..].find(quote)? + 1; | ||
| 196 | return Some(value[1..end].to_string()); | ||
| 197 | } | ||
| 198 | let end = value.find(char::is_whitespace).unwrap_or(value.len()); | ||
| 199 | return Some(value[..end].to_string()); | ||
| 200 | } | ||
| 201 | search = &search[pos + name.len()..]; | ||
| 202 | } | ||
| 203 | None | ||
| 204 | } | ||
| 205 | |||
| 206 | /// Decode the entities either exporter is likely to emit, so an encoding difference is | ||
| 207 | /// never reported as a semantic one. | ||
| 208 | fn decode_entities(s: &str) -> String { | ||
| 209 | let mut out = String::with_capacity(s.len()); | ||
| 210 | let mut rest = s; | ||
| 211 | while let Some(amp) = rest.find('&') { | ||
| 212 | out.push_str(&rest[..amp]); | ||
| 213 | let tail = &rest[amp..]; | ||
| 214 | let Some(semi) = tail.find(';').filter(|s| *s <= 12) else { | ||
| 215 | out.push('&'); | ||
| 216 | rest = &tail[1..]; | ||
| 217 | continue; | ||
| 218 | }; | ||
| 219 | let entity = &tail[1..semi]; | ||
| 220 | let decoded = match entity { | ||
| 221 | "amp" => Some('&'), | ||
| 222 | "lt" => Some('<'), | ||
| 223 | "gt" => Some('>'), | ||
| 224 | "quot" => Some('"'), | ||
| 225 | "apos" => Some('\''), | ||
| 226 | "nbsp" => Some(' '), | ||
| 227 | _ => entity | ||
| 228 | .strip_prefix('#') | ||
| 229 | .and_then(|n| match n.strip_prefix(['x', 'X']) { | ||
| 230 | Some(hex) => u32::from_str_radix(hex, 16).ok(), | ||
| 231 | None => n.parse::<u32>().ok(), | ||
| 232 | }) | ||
| 233 | .and_then(char::from_u32), | ||
| 234 | }; | ||
| 235 | match decoded { | ||
| 236 | // A non-breaking space is a space for comparison purposes. | ||
| 237 | Some('\u{a0}') => out.push(' '), | ||
| 238 | Some(c) => out.push(c), | ||
| 239 | None => { | ||
| 240 | out.push('&'); | ||
| 241 | rest = &tail[1..]; | ||
| 242 | continue; | ||
| 243 | } | ||
| 244 | } | ||
| 245 | rest = &tail[semi + 1..]; | ||
| 246 | } | ||
| 247 | out.push_str(rest); | ||
| 248 | out | ||
| 249 | } | ||
| 250 | |||
| 251 | // --------------------------------------------------------------------------- | 75 | // --------------------------------------------------------------------------- |
| 252 | // Divergence report | 76 | // Divergence report |
| 253 | // --------------------------------------------------------------------------- | 77 | // --------------------------------------------------------------------------- |