src/audit.rs
679 lines · 25854 bytes
1//! Corpus audit (spec §5, Phase 0): measure which org constructs a real corpus actually
2//! uses, and classify each against the supported/unsupported line.
3//!
4//! This exists because the scope was recommended rather than measured — a guess about
5//! which slice of org matters. A guess about a corpus is a hypothesis, and this is the
6//! experiment. It answers two questions:
7//!
8//! 1. **Coverage** — of the constructs this corpus uses, which do we handle? A construct
9//! that is common here and out of scope is a scope bug, not a corpus quirk.
10//! 2. **Blind spots** — which constructs are here that the implementation has no opinion
11//! about at all? These are the dangerous ones: not "known unsupported" but unknown.
12//!
13//! The audit is deliberately a *separate, line-oriented scanner* rather than a reuse of
14//! [`crate::parser`]. Auditing with the parser could only ever find constructs the parser
15//! already knows about, which is precisely the wrong instrument for question 2 — it would
16//! report a blind spot as clean.
17//!
18//! Nothing here reports document *text*. Counts, construct names, and `file:line`
19//! locations only, so an audit of private notes stays publishable.
20
21use std::collections::BTreeMap;
22
23use anyhow::{Context, Result};
24use camino::{Utf8Path, Utf8PathBuf};
25use walkdir::WalkDir;
26
27/// Where a construct sits relative to the supported set, which
28/// `docs/guide/05-org-support.org` defines.
29#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)]
30pub enum Scope {
31 /// orgo handles this.
32 In,
33 /// orgo deliberately excludes this; it degrades predictably.
34 Out,
35}
36
37impl Scope {
38 fn label(self) -> &'static str {
39 match self {
40 Scope::In => "IN ",
41 Scope::Out => "OUT",
42 }
43 }
44}
45
46/// One construct's tally across the corpus.
47#[derive(Debug, Default, Clone)]
48pub struct Tally {
49 pub occurrences: usize,
50 pub files: usize,
51 /// First `file:line` the construct was seen at, to make a finding actionable.
52 pub first_seen: Option<String>,
53 /// Set while scanning one file, to count each file once.
54 seen_in_current_file: bool,
55}
56
57/// The audit result: the fixed construct catalog plus the dynamic name censuses.
58#[derive(Debug, Default)]
59pub struct Audit {
60 pub files: usize,
61 pub lines: usize,
62 /// Catalogued constructs → tally.
63 pub constructs: BTreeMap<(Scope, &'static str), Tally>,
64 /// Every distinct `#+KEYWORD:` seen, by name.
65 pub keywords: BTreeMap<String, Tally>,
66 /// Every distinct `#+BEGIN_<TYPE>` seen, by type.
67 pub blocks: BTreeMap<String, Tally>,
68 /// Every distinct `:DRAWER:` seen, by name.
69 pub drawers: BTreeMap<String, Tally>,
70 /// Every distinct link scheme seen (`https`, `file`, `id`, `denote`, ...).
71 pub link_schemes: BTreeMap<String, Tally>,
72}
73
74/// Names the implementation reads by name, so the census can mark everything else. These
75/// are the *recognized* sets, not the supported ones: `INCLUDE` is recognized and
76/// deliberately inert.
77const KNOWN_KEYWORDS: &[&str] = &[
78 "TITLE", "AUTHOR", "DATE", "EMAIL", "LANGUAGE", "OPTIONS", "FILETAGS", "DESCRIPTION",
79 "KEYWORDS", "CAPTION", "NAME", "ATTR_HTML", "RESULTS", "TBLFM", "INCLUDE", "TODO",
80 "STARTUP", "SUBTITLE", "SETUPFILE", "MACRO", "PROPERTY", "HTML_HEAD", "EXCLUDE_TAGS",
81 // orgo's own keywords, each read by name: `SLUG` names the output file
82 // (`util::output_path`), `DRAFT` decides whether the page publishes at all
83 // (`util::is_draft`), and `TEMPLATE` picks the template (`config::page_template`).
84 // Leaving them out reported the corpus's most-used keyword as unrecognized.
85 "SLUG", "DRAFT", "TEMPLATE",
86];
87const KNOWN_DRAWERS: &[&str] = &["PROPERTIES", "LOGBOOK", "END"];
88/// Keyword names conventional enough to be worth flagging when they lead a heading.
89/// A custom sequence is only *real* if some `#+TODO:` declares it, which the census
90/// reports separately — this list keeps the heading-level signal honest.
91const CONVENTIONAL_TODO_KEYWORDS: &[&str] = &[
92 "NEXT", "WAITING", "HOLD", "CANCELLED", "CANCELED", "STARTED", "SOMEDAY", "PROJ",
93 "IN-PROGRESS", "BLOCKED", "REVIEW",
94];
95const KNOWN_SCHEMES: &[&str] = &[
96 "http", "https", "mailto", "ftp", "news", "tel", "file", "id", "custom-id", "heading",
97 "relative",
98];
99
100/// The census markers. A name with dedicated handling carries none; `PASS_THROUGH` says
101/// nothing reads the name but it works anyway; `UNKNOWN` is the blind-spot signal.
102const HANDLED: &str = "";
103const PASS_THROUGH: &str = "\u{2014}";
104const UNKNOWN: &str = "???";
105
106impl Audit {
107 /// How the report marks a census name: the recognized set to look the name up in,
108 /// and what an unlisted name means for that census.
109 pub fn marker(kind: Census, name: &str) -> &'static str {
110 let (known, unlisted) = match kind {
111 // An unlisted keyword is not a blind spot. Reaching templates as
112 // `page.keywords.<name>` is the designed behaviour, so every keyword name
113 // works; what the census reports is which ones have dedicated handling.
114 // `LEDE`, read by the shipped docs theme's template, was the case that made
115 // this concrete: `???` on it was a claim about orgo's source, not a gap.
116 Census::Keyword => (KNOWN_KEYWORDS, PASS_THROUGH),
117 // Every block name renders, and renders as org renders it: the names in
118 // `block_construct` through dedicated handling, every other name as a special
119 // block — a div carrying the name, holding parsed org, which is exactly what
120 // org's exporter emits. No block name is a blind spot.
121 Census::Block => return HANDLED,
122 Census::Drawer => (KNOWN_DRAWERS, UNKNOWN),
123 Census::Scheme => (KNOWN_SCHEMES, UNKNOWN),
124 };
125 if known.iter().any(|k| k.eq_ignore_ascii_case(name)) {
126 HANDLED
127 } else {
128 unlisted
129 }
130 }
131}
132
133/// Which dynamic census a name belongs to.
134#[derive(Debug, Clone, Copy)]
135pub enum Census {
136 Keyword,
137 Block,
138 Drawer,
139 Scheme,
140}
141
142/// Walk `root`, auditing every `.org` file.
143pub fn audit(root: &Utf8Path) -> Result<Audit> {
144 let mut audit = Audit::default();
145 let mut paths: Vec<Utf8PathBuf> = Vec::new();
146
147 if root.is_file() {
148 paths.push(root.to_owned());
149 } else {
150 for entry in WalkDir::new(root).sort_by_file_name() {
151 let entry = entry.with_context(|| format!("walking {root}"))?;
152 if !entry.file_type().is_file() {
153 continue;
154 }
155 let path = Utf8PathBuf::from_path_buf(entry.into_path())
156 .map_err(|p| anyhow::anyhow!("non-UTF-8 path: {}", p.display()))?;
157 if path.extension() == Some("org") {
158 paths.push(path);
159 }
160 }
161 }
162
163 for path in &paths {
164 // A file that cannot be read is reported and skipped: an audit of 179 files
165 // should not be lost to one unreadable one.
166 let source = match std::fs::read_to_string(path) {
167 Ok(s) => s,
168 Err(e) => {
169 eprintln!("warning: skipping {path}: {e}");
170 continue;
171 }
172 };
173 let rel = path.strip_prefix(root).unwrap_or(path).to_owned();
174 audit.scan_file(&rel, &source);
175 audit.files += 1;
176 }
177 Ok(audit)
178}
179
180impl Audit {
181 fn scan_file(&mut self, path: &Utf8Path, source: &str) {
182 // Reset the per-file flags so each construct counts this file at most once.
183 for tally in self.constructs.values_mut() {
184 tally.seen_in_current_file = false;
185 }
186 for map in [
187 &mut self.keywords,
188 &mut self.blocks,
189 &mut self.drawers,
190 &mut self.link_schemes,
191 ] {
192 for tally in map.values_mut() {
193 tally.seen_in_current_file = false;
194 }
195 }
196
197 let mut in_block: Option<String> = None;
198 for (idx, line) in source.lines().enumerate() {
199 self.lines += 1;
200 let at = format!("{path}:{}", idx + 1);
201 let trimmed = line.trim_start();
202
203 // Inside a verbatim block only the terminator matters — a `*` in a source
204 // block is not a heading, and counting it as one would corrupt the audit.
205 if let Some(kind) = &in_block {
206 if trimmed.to_ascii_uppercase().starts_with("#+END_") {
207 in_block = None;
208 } else if kind.eq_ignore_ascii_case("SRC") || kind.eq_ignore_ascii_case("EXAMPLE") {
209 continue;
210 }
211 continue;
212 }
213 if let Some(rest) = trimmed.to_ascii_uppercase().strip_prefix("#+BEGIN_") {
214 let kind = rest.split_whitespace().next().unwrap_or("").to_string();
215 self.count_census(Census::Block, &kind, &at);
216 self.count(Scope::In, block_construct(&kind), &at);
217 if trimmed.to_ascii_uppercase().contains(":RESULTS") {
218 self.count(Scope::Out, "babel header args (:results)", &at);
219 }
220 in_block = Some(kind);
221 continue;
222 }
223
224 self.scan_line(line, trimmed, &at);
225 }
226 }
227
228 fn scan_line(&mut self, line: &str, trimmed: &str, at: &str) {
229 // --- headings and their metadata ---
230 if let Some(stars) = heading_stars(line) {
231 self.count(Scope::In, "heading", at);
232 let rest = line[stars..].trim();
233 let word = rest.split_whitespace().next().unwrap_or("");
234 if word == "TODO" || word == "DONE" {
235 self.count(Scope::In, "TODO keyword (default set)", at);
236 } else if CONVENTIONAL_TODO_KEYWORDS.contains(&word) {
237 // Only conventional keyword names count. "Any all-caps first word" is
238 // the tempting rule and it is wrong: it reads `* CSS Variables` as the
239 // keyword `CSS`, which on this corpus produced 23 false positives and
240 // zero true ones. An audit that overstates a gap is worse than no audit,
241 // because it argues for work nobody needs.
242 self.count(Scope::Out, "TODO keyword (custom sequence)", at);
243 }
244 if rest.contains("[#") {
245 self.count(Scope::In, "priority cookie", at);
246 }
247 if rest.trim_end().ends_with(':') && rest.trim_end().matches(':').count() >= 2 {
248 self.count(Scope::In, "heading tags", at);
249 }
250 if rest.contains("[/") || rest.contains("[%") {
251 self.count(Scope::Out, "statistics cookie", at);
252 }
253 return;
254 }
255
256 // --- planning and clocking ---
257 for marker in ["SCHEDULED:", "DEADLINE:", "CLOSED:"] {
258 if trimmed.starts_with(marker) {
259 self.count(Scope::Out, "planning line", at);
260 }
261 }
262 if trimmed.starts_with("CLOCK:") {
263 self.count(Scope::Out, "clock entry", at);
264 }
265
266 // --- keywords and drawers ---
267 if let Some(rest) = trimmed.strip_prefix("#+") {
268 if let Some(colon) = rest.find(':') {
269 let key = rest[..colon].trim().to_ascii_uppercase();
270 if !key.is_empty() && !key.contains(char::is_whitespace) {
271 self.count_census(Census::Keyword, &key, at);
272 match key.as_str() {
273 "CAPTION" | "NAME" | "ATTR_HTML" => {
274 self.count(Scope::In, "affiliated keyword", at)
275 }
276 // Not a gap: org's HTML exporter does not recalculate `#+TBLFM:`
277 // either, so the cells as written are what both exporters emit.
278 // `fixtures/tblfm.org` holds the oracle to that (`tests/oracle.rs`).
279 // Unlike `#+INCLUDE:`, nothing is lost by leaving it inert.
280 "TBLFM" => self.count(Scope::In, "table formula (#+TBLFM:)", at),
281 "INCLUDE" => self.count(Scope::Out, "#+INCLUDE:", at),
282 "RESULTS" => self.count(Scope::Out, "babel results block", at),
283 "TODO" => self.count(Scope::Out, "#+TODO: keyword sequence", at),
284 "MACRO" => self.count(Scope::Out, "macro definition", at),
285 _ => self.count(Scope::In, "#+ keyword", at),
286 }
287 }
288 }
289 } else if is_drawer(trimmed) {
290 let name = trimmed[1..trimmed.len() - 1].to_ascii_uppercase();
291 if name != "END" {
292 self.count_census(Census::Drawer, &name, at);
293 match name.as_str() {
294 "PROPERTIES" => self.count(Scope::In, "property drawer", at),
295 _ => self.count(Scope::Out, "non-PROPERTIES drawer", at),
296 }
297 }
298 }
299
300 // --- lists, tables, rules ---
301 if let Some(bullet) = list_bullet(trimmed) {
302 self.count(Scope::In, "list item", at);
303 if bullet == Bullet::Ordered {
304 self.count(Scope::In, "ordered list", at);
305 }
306 let indent = line.len() - trimmed.len();
307 if indent > 0 {
308 self.count(Scope::In, "nested list item", at);
309 }
310 if trimmed.contains(" :: ") {
311 self.count(Scope::In, "description list", at);
312 }
313 let after = trimmed.trim_start_matches(['-', '+', '*', ' ']);
314 if after.starts_with("[ ]") || after.starts_with("[X]") || after.starts_with("[-]") {
315 self.count(Scope::In, "checkbox", at);
316 }
317 }
318 if trimmed.starts_with('|') {
319 self.count(Scope::In, "table row", at);
320 }
321 if trimmed.starts_with(':') && !is_drawer(trimmed) && trimmed.starts_with(": ") {
322 self.count(Scope::Out, "fixed-width line", at);
323 }
324
325 // --- footnotes ---
326 if trimmed.starts_with("[fn:") {
327 self.count(Scope::In, "footnote definition", at);
328 } else if line.contains("[fn:") {
329 self.count(Scope::In, "footnote reference", at);
330 }
331
332 // --- inline objects ---
333 self.scan_inline(line, at);
334 }
335
336 fn scan_inline(&mut self, line: &str, at: &str) {
337 // Links: count each `[[target]]`, censusing its scheme.
338 let mut rest = line;
339 while let Some(start) = rest.find("[[") {
340 let after = &rest[start + 2..];
341 let Some(end) = after.find("]]") else { break };
342 let inner = &after[..end];
343 let target = inner.split("][").next().unwrap_or(inner);
344 self.count(Scope::In, "link", at);
345 self.count_census(Census::Scheme, &link_scheme(target), at);
346 rest = &after[end..];
347 }
348
349 // The rest read a line that has had its `=verbatim=` and `~code~` spans blanked:
350 // a construct shown inside verbatim is displayed rather than rendered, so a page
351 // documenting org syntax is not a use of the syntax it names. Emphasis pairs
352 // below deliberately still see the raw line — `=` is one of the markers scanned.
353 let bare = without_literal_spans(line);
354
355 if has_timestamp(&bare) {
356 self.count(Scope::In, "timestamp", at);
357 }
358 if bare.contains("{{{") {
359 self.count(Scope::Out, "macro call", at);
360 }
361 if bare.contains("<<<") {
362 self.count(Scope::Out, "radio target", at);
363 } else if bare.contains("<<") && bare.contains(">>") {
364 self.count(Scope::Out, "internal target", at);
365 }
366 if bare.contains("\\begin{") || latex_inline(&bare) {
367 self.count(Scope::Out, "LaTeX fragment", at);
368 }
369 if entity_ref(&bare) {
370 // Rendered, and rendered as org renders it: `\alpha` becomes `α` in
371 // both exporters. `fixtures/audit-entities.org` holds the oracle to that.
372 self.count(Scope::In, "entity (\\name)", at);
373 }
374 for (marker, name) in [
375 ('*', "bold"),
376 ('/', "italic"),
377 ('_', "underline"),
378 ('+', "strike-through"),
379 ('=', "verbatim"),
380 ('~', "code"),
381 ] {
382 if emphasis_pair(line, marker) {
383 self.count(Scope::In, name, at);
384 }
385 }
386 }
387
388 fn count(&mut self, scope: Scope, name: &'static str, at: &str) {
389 let tally = self.constructs.entry((scope, name)).or_default();
390 bump(tally, at);
391 }
392
393 fn count_census(&mut self, kind: Census, name: &str, at: &str) {
394 let map = match kind {
395 Census::Keyword => &mut self.keywords,
396 Census::Block => &mut self.blocks,
397 Census::Drawer => &mut self.drawers,
398 Census::Scheme => &mut self.link_schemes,
399 };
400 let tally = map.entry(name.to_string()).or_default();
401 bump(tally, at);
402 }
403}
404
405fn bump(tally: &mut Tally, at: &str) {
406 tally.occurrences += 1;
407 if !tally.seen_in_current_file {
408 tally.seen_in_current_file = true;
409 tally.files += 1;
410 }
411 if tally.first_seen.is_none() {
412 tally.first_seen = Some(at.to_string());
413 }
414}
415
416// ---------------------------------------------------------------------------
417// Line-level detectors. Deliberately independent of the parser (see module docs).
418// ---------------------------------------------------------------------------
419
420fn heading_stars(line: &str) -> Option<usize> {
421 if !line.starts_with('*') {
422 return None;
423 }
424 let stars = line.chars().take_while(|c| *c == '*').count();
425 let after = &line[stars..];
426 (after.starts_with(' ') || after.is_empty()).then_some(stars)
427}
428
429#[derive(PartialEq)]
430enum Bullet {
431 Unordered,
432 Ordered,
433}
434
435fn list_bullet(trimmed: &str) -> Option<Bullet> {
436 let bytes = trimmed.as_bytes();
437 if bytes.is_empty() {
438 return None;
439 }
440 if (bytes[0] == b'-' || bytes[0] == b'+') && (bytes.len() == 1 || bytes[1] == b' ') {
441 return Some(Bullet::Unordered);
442 }
443 let digits = trimmed.chars().take_while(|c| c.is_ascii_digit()).count();
444 if digits > 0 {
445 let after = &trimmed[digits..];
446 if (after.starts_with('.') || after.starts_with(')'))
447 && (after.len() == 1 || after.as_bytes()[1] == b' ')
448 {
449 return Some(Bullet::Ordered);
450 }
451 }
452 None
453}
454
455fn is_drawer(trimmed: &str) -> bool {
456 let t = trimmed.trim_end();
457 t.len() >= 3
458 && t.starts_with(':')
459 && t.ends_with(':')
460 && t[1..t.len() - 1]
461 .chars()
462 .all(|c| c.is_ascii_alphanumeric() || c == '_' || c == '-')
463 && t.len() > 2
464}
465
466fn block_construct(kind: &str) -> &'static str {
467 match kind.to_ascii_uppercase().as_str() {
468 "SRC" => "source block",
469 "QUOTE" => "quote block",
470 "EXAMPLE" => "example block",
471 "CENTER" => "center block",
472 "EXPORT" => "export block",
473 "VERSE" => "verse block",
474 "COMMENT" => "comment block",
475 _ => "special block",
476 }
477}
478
479/// The scheme of a link target, normalized into the census's vocabulary.
480fn link_scheme(target: &str) -> String {
481 if let Some(rest) = target.split_once(':') {
482 let scheme = rest.0;
483 if !scheme.is_empty()
484 && scheme
485 .chars()
486 .all(|c| c.is_ascii_alphanumeric() || c == '-' || c == '+')
487 {
488 return scheme.to_ascii_lowercase();
489 }
490 }
491 if target.starts_with('#') {
492 return "custom-id".to_string();
493 }
494 if target.starts_with('*') {
495 return "heading".to_string();
496 }
497 "relative".to_string()
498}
499
500/// A `<...>`/`[...]` span opening with an ISO date is a timestamp.
501fn has_timestamp(line: &str) -> bool {
502 let bytes = line.as_bytes();
503 for (i, c) in line.char_indices() {
504 if c != '<' && c != '[' {
505 continue;
506 }
507 let rest = &bytes[i + 1..];
508 if rest.len() >= 10
509 && rest[..4].iter().all(u8::is_ascii_digit)
510 && rest[4] == b'-'
511 && rest[5..7].iter().all(u8::is_ascii_digit)
512 && rest[7] == b'-'
513 && rest[8..10].iter().all(u8::is_ascii_digit)
514 {
515 return true;
516 }
517 }
518 false
519}
520
521/// `$x$` or `\(x\)` inline math. `$` alone (a price, a shell prompt) is not math.
522fn latex_inline(line: &str) -> bool {
523 if line.contains("\\(") && line.contains("\\)") {
524 return true;
525 }
526 let dollars = line.matches('$').count();
527 dollars >= 2 && line.contains("$\\")
528}
529
530/// A `\name` entity reference such as `\alpha`.
531///
532/// Only names org knows count, matching [`crate::parser`]'s rule for rendering one: a
533/// Windows path (`C:\Users\me`) and a namespaced identifier (`Tumblr\API\Client`) are not
534/// entity references.
535fn entity_ref(line: &str) -> bool {
536 let chars: Vec<char> = line.chars().collect();
537 for (i, c) in chars.iter().enumerate() {
538 if *c != '\\' {
539 continue;
540 }
541 let name: String = chars[i + 1..].iter().take_while(|c| c.is_ascii_alphabetic()).collect();
542 if crate::entities::lookup(&name).is_some() {
543 return true;
544 }
545 }
546 false
547}
548
549/// Blank out `=verbatim=` and `~code~` spans. Deliberately looser than the parser's
550/// border rules — the audit measures prevalence, and erring toward blanking keeps it
551/// from overstating a gap.
552fn without_literal_spans(line: &str) -> String {
553 let mut out = String::with_capacity(line.len());
554 let mut open: Option<char> = None;
555 for c in line.chars() {
556 match open {
557 Some(marker) => {
558 out.push(' ');
559 if c == marker {
560 open = None;
561 }
562 }
563 None if c == '=' || c == '~' => {
564 open = Some(c);
565 out.push(' ');
566 }
567 None => out.push(c),
568 }
569 }
570 out
571}
572
573/// A plausible `*bold*`-style emphasis pair: two markers on one line with non-space
574/// content between them. Approximate by design — the audit measures prevalence, and the
575/// parser owns the exact pre/post-character rules.
576fn emphasis_pair(line: &str, marker: char) -> bool {
577 let positions: Vec<usize> = line
578 .char_indices()
579 .filter(|(_, c)| *c == marker)
580 .map(|(i, _)| i)
581 .collect();
582 if positions.len() < 2 {
583 return false;
584 }
585 // A leading `*` is a heading, and `-`/`+` at line start is a bullet.
586 let trimmed = line.trim_start();
587 if trimmed.starts_with(marker) {
588 return false;
589 }
590 positions.windows(2).any(|w| w[1] > w[0] + 1)
591}
592
593// ---------------------------------------------------------------------------
594// Report
595// ---------------------------------------------------------------------------
596
597/// Render the audit as a readable report. Names, counts and locations only — never
598/// document text, so an audit of private notes is safe to paste into an issue.
599pub fn report(audit: &Audit) -> String {
600 let mut out = String::new();
601 out.push_str(&format!(
602 "corpus: {} file(s), {} line(s)\n",
603 audit.files, audit.lines
604 ));
605
606 let mut rows: Vec<(&(Scope, &str), &Tally)> = audit.constructs.iter().collect();
607 rows.sort_by(|a, b| {
608 b.1.occurrences
609 .cmp(&a.1.occurrences)
610 .then_with(|| a.0 .1.cmp(b.0 .1))
611 });
612
613 out.push_str("\nCONSTRUCTS (by frequency)\n");
614 out.push_str(&format!(
615 "{:<4} {:<32} {:>8} {:>7} {}\n",
616 "", "construct", "uses", "files", "first seen"
617 ));
618 for ((scope, name), tally) in &rows {
619 out.push_str(&format!(
620 "{:<4} {:<32} {:>8} {:>7} {}\n",
621 scope.label(),
622 name,
623 tally.occurrences,
624 tally.files,
625 tally.first_seen.as_deref().unwrap_or("")
626 ));
627 }
628
629 let in_uses: usize = rows
630 .iter()
631 .filter(|((s, _), _)| *s == Scope::In)
632 .map(|(_, t)| t.occurrences)
633 .sum();
634 let out_uses: usize = rows
635 .iter()
636 .filter(|((s, _), _)| *s == Scope::Out)
637 .map(|(_, t)| t.occurrences)
638 .sum();
639 let total = in_uses + out_uses;
640 let pct = |n: usize| {
641 if total == 0 {
642 0.0
643 } else {
644 100.0 * n as f64 / total as f64
645 }
646 };
647 out.push_str(&format!(
648 "\ncoverage: {in_uses} in-scope use(s) ({:.1}%), {out_uses} out-of-scope ({:.1}%)\n",
649 pct(in_uses),
650 pct(out_uses)
651 ));
652
653 for (title, kind, map) in [
654 ("KEYWORDS", Census::Keyword, &audit.keywords),
655 ("BLOCK TYPES", Census::Block, &audit.blocks),
656 ("DRAWERS", Census::Drawer, &audit.drawers),
657 ("LINK SCHEMES", Census::Scheme, &audit.link_schemes),
658 ] {
659 let mut names: Vec<(&String, &Tally)> = map.iter().collect();
660 names.sort_by(|a, b| b.1.occurrences.cmp(&a.1.occurrences).then(a.0.cmp(b.0)));
661 out.push_str(&format!("\n{title}\n"));
662 for (name, tally) in names {
663 out.push_str(&format!(
664 "{:<4}{:<32} {:>8} {:>7} {}\n",
665 Audit::marker(kind, name),
666 name,
667 tally.occurrences,
668 tally.files,
669 tally.first_seen.as_deref().unwrap_or("")
670 ));
671 }
672 }
673 out.push_str(
674 "\n`???` marks a name the implementation does not recognize at all.\n\
675 `\u{2014}` marks a keyword with no dedicated handling; it reaches templates \
676 as `page.keywords.<name>`.\n",
677 );
678 out
679}