krz/orgo

Lightning fast org-mode static site generator. fast go org-mode static-site-generator

src/audit.rs

679 lines · 25854 bytes

  1//! Corpus audit (spec §5, Phase 0): measure which org constructs a real corpus actually
  2//! uses, and classify each against the supported/unsupported line.
  3//!
  4//! This exists because the scope was recommended rather than measured — a guess about
  5//! which slice of org matters. A guess about a corpus is a hypothesis, and this is the
  6//! experiment. It answers two questions:
  7//!
  8//! 1. **Coverage** — of the constructs this corpus uses, which do we handle? A construct
  9//!    that is common here and out of scope is a scope bug, not a corpus quirk.
 10//! 2. **Blind spots** — which constructs are here that the implementation has no opinion
 11//!    about at all? These are the dangerous ones: not "known unsupported" but unknown.
 12//!
 13//! The audit is deliberately a *separate, line-oriented scanner* rather than a reuse of
 14//! [`crate::parser`]. Auditing with the parser could only ever find constructs the parser
 15//! already knows about, which is precisely the wrong instrument for question 2 — it would
 16//! report a blind spot as clean.
 17//!
 18//! Nothing here reports document *text*. Counts, construct names, and `file:line`
 19//! locations only, so an audit of private notes stays publishable.
 20
 21use std::collections::BTreeMap;
 22
 23use anyhow::{Context, Result};
 24use camino::{Utf8Path, Utf8PathBuf};
 25use walkdir::WalkDir;
 26
 27/// Where a construct sits relative to the supported set, which
 28/// `docs/guide/05-org-support.org` defines.
 29#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)]
 30pub enum Scope {
 31    /// orgo handles this.
 32    In,
 33    /// orgo deliberately excludes this; it degrades predictably.
 34    Out,
 35}
 36
 37impl Scope {
 38    fn label(self) -> &'static str {
 39        match self {
 40            Scope::In => "IN ",
 41            Scope::Out => "OUT",
 42        }
 43    }
 44}
 45
 46/// One construct's tally across the corpus.
 47#[derive(Debug, Default, Clone)]
 48pub struct Tally {
 49    pub occurrences: usize,
 50    pub files: usize,
 51    /// First `file:line` the construct was seen at, to make a finding actionable.
 52    pub first_seen: Option<String>,
 53    /// Set while scanning one file, to count each file once.
 54    seen_in_current_file: bool,
 55}
 56
 57/// The audit result: the fixed construct catalog plus the dynamic name censuses.
 58#[derive(Debug, Default)]
 59pub struct Audit {
 60    pub files: usize,
 61    pub lines: usize,
 62    /// Catalogued constructs → tally.
 63    pub constructs: BTreeMap<(Scope, &'static str), Tally>,
 64    /// Every distinct `#+KEYWORD:` seen, by name.
 65    pub keywords: BTreeMap<String, Tally>,
 66    /// Every distinct `#+BEGIN_<TYPE>` seen, by type.
 67    pub blocks: BTreeMap<String, Tally>,
 68    /// Every distinct `:DRAWER:` seen, by name.
 69    pub drawers: BTreeMap<String, Tally>,
 70    /// Every distinct link scheme seen (`https`, `file`, `id`, `denote`, ...).
 71    pub link_schemes: BTreeMap<String, Tally>,
 72}
 73
 74/// Names the implementation reads by name, so the census can mark everything else. These
 75/// are the *recognized* sets, not the supported ones: `INCLUDE` is recognized and
 76/// deliberately inert.
 77const KNOWN_KEYWORDS: &[&str] = &[
 78    "TITLE", "AUTHOR", "DATE", "EMAIL", "LANGUAGE", "OPTIONS", "FILETAGS", "DESCRIPTION",
 79    "KEYWORDS", "CAPTION", "NAME", "ATTR_HTML", "RESULTS", "TBLFM", "INCLUDE", "TODO",
 80    "STARTUP", "SUBTITLE", "SETUPFILE", "MACRO", "PROPERTY", "HTML_HEAD", "EXCLUDE_TAGS",
 81    // orgo's own keywords, each read by name: `SLUG` names the output file
 82    // (`util::output_path`), `DRAFT` decides whether the page publishes at all
 83    // (`util::is_draft`), and `TEMPLATE` picks the template (`config::page_template`).
 84    // Leaving them out reported the corpus's most-used keyword as unrecognized.
 85    "SLUG", "DRAFT", "TEMPLATE",
 86];
 87const KNOWN_DRAWERS: &[&str] = &["PROPERTIES", "LOGBOOK", "END"];
 88/// Keyword names conventional enough to be worth flagging when they lead a heading.
 89/// A custom sequence is only *real* if some `#+TODO:` declares it, which the census
 90/// reports separately — this list keeps the heading-level signal honest.
 91const CONVENTIONAL_TODO_KEYWORDS: &[&str] = &[
 92    "NEXT", "WAITING", "HOLD", "CANCELLED", "CANCELED", "STARTED", "SOMEDAY", "PROJ",
 93    "IN-PROGRESS", "BLOCKED", "REVIEW",
 94];
 95const KNOWN_SCHEMES: &[&str] = &[
 96    "http", "https", "mailto", "ftp", "news", "tel", "file", "id", "custom-id", "heading",
 97    "relative",
 98];
 99
100/// The census markers. A name with dedicated handling carries none; `PASS_THROUGH` says
101/// nothing reads the name but it works anyway; `UNKNOWN` is the blind-spot signal.
102const HANDLED: &str = "";
103const PASS_THROUGH: &str = "\u{2014}";
104const UNKNOWN: &str = "???";
105
106impl Audit {
107    /// How the report marks a census name: the recognized set to look the name up in,
108    /// and what an unlisted name means for that census.
109    pub fn marker(kind: Census, name: &str) -> &'static str {
110        let (known, unlisted) = match kind {
111            // An unlisted keyword is not a blind spot. Reaching templates as
112            // `page.keywords.<name>` is the designed behaviour, so every keyword name
113            // works; what the census reports is which ones have dedicated handling.
114            // `LEDE`, read by the shipped docs theme's template, was the case that made
115            // this concrete: `???` on it was a claim about orgo's source, not a gap.
116            Census::Keyword => (KNOWN_KEYWORDS, PASS_THROUGH),
117            // Every block name renders, and renders as org renders it: the names in
118            // `block_construct` through dedicated handling, every other name as a special
119            // block — a div carrying the name, holding parsed org, which is exactly what
120            // org's exporter emits. No block name is a blind spot.
121            Census::Block => return HANDLED,
122            Census::Drawer => (KNOWN_DRAWERS, UNKNOWN),
123            Census::Scheme => (KNOWN_SCHEMES, UNKNOWN),
124        };
125        if known.iter().any(|k| k.eq_ignore_ascii_case(name)) {
126            HANDLED
127        } else {
128            unlisted
129        }
130    }
131}
132
133/// Which dynamic census a name belongs to.
134#[derive(Debug, Clone, Copy)]
135pub enum Census {
136    Keyword,
137    Block,
138    Drawer,
139    Scheme,
140}
141
142/// Walk `root`, auditing every `.org` file.
143pub fn audit(root: &Utf8Path) -> Result<Audit> {
144    let mut audit = Audit::default();
145    let mut paths: Vec<Utf8PathBuf> = Vec::new();
146
147    if root.is_file() {
148        paths.push(root.to_owned());
149    } else {
150        for entry in WalkDir::new(root).sort_by_file_name() {
151            let entry = entry.with_context(|| format!("walking {root}"))?;
152            if !entry.file_type().is_file() {
153                continue;
154            }
155            let path = Utf8PathBuf::from_path_buf(entry.into_path())
156                .map_err(|p| anyhow::anyhow!("non-UTF-8 path: {}", p.display()))?;
157            if path.extension() == Some("org") {
158                paths.push(path);
159            }
160        }
161    }
162
163    for path in &paths {
164        // A file that cannot be read is reported and skipped: an audit of 179 files
165        // should not be lost to one unreadable one.
166        let source = match std::fs::read_to_string(path) {
167            Ok(s) => s,
168            Err(e) => {
169                eprintln!("warning: skipping {path}: {e}");
170                continue;
171            }
172        };
173        let rel = path.strip_prefix(root).unwrap_or(path).to_owned();
174        audit.scan_file(&rel, &source);
175        audit.files += 1;
176    }
177    Ok(audit)
178}
179
180impl Audit {
181    fn scan_file(&mut self, path: &Utf8Path, source: &str) {
182        // Reset the per-file flags so each construct counts this file at most once.
183        for tally in self.constructs.values_mut() {
184            tally.seen_in_current_file = false;
185        }
186        for map in [
187            &mut self.keywords,
188            &mut self.blocks,
189            &mut self.drawers,
190            &mut self.link_schemes,
191        ] {
192            for tally in map.values_mut() {
193                tally.seen_in_current_file = false;
194            }
195        }
196
197        let mut in_block: Option<String> = None;
198        for (idx, line) in source.lines().enumerate() {
199            self.lines += 1;
200            let at = format!("{path}:{}", idx + 1);
201            let trimmed = line.trim_start();
202
203            // Inside a verbatim block only the terminator matters — a `*` in a source
204            // block is not a heading, and counting it as one would corrupt the audit.
205            if let Some(kind) = &in_block {
206                if trimmed.to_ascii_uppercase().starts_with("#+END_") {
207                    in_block = None;
208                } else if kind.eq_ignore_ascii_case("SRC") || kind.eq_ignore_ascii_case("EXAMPLE") {
209                    continue;
210                }
211                continue;
212            }
213            if let Some(rest) = trimmed.to_ascii_uppercase().strip_prefix("#+BEGIN_") {
214                let kind = rest.split_whitespace().next().unwrap_or("").to_string();
215                self.count_census(Census::Block, &kind, &at);
216                self.count(Scope::In, block_construct(&kind), &at);
217                if trimmed.to_ascii_uppercase().contains(":RESULTS") {
218                    self.count(Scope::Out, "babel header args (:results)", &at);
219                }
220                in_block = Some(kind);
221                continue;
222            }
223
224            self.scan_line(line, trimmed, &at);
225        }
226    }
227
228    fn scan_line(&mut self, line: &str, trimmed: &str, at: &str) {
229        // --- headings and their metadata ---
230        if let Some(stars) = heading_stars(line) {
231            self.count(Scope::In, "heading", at);
232            let rest = line[stars..].trim();
233            let word = rest.split_whitespace().next().unwrap_or("");
234            if word == "TODO" || word == "DONE" {
235                self.count(Scope::In, "TODO keyword (default set)", at);
236            } else if CONVENTIONAL_TODO_KEYWORDS.contains(&word) {
237                // Only conventional keyword names count. "Any all-caps first word" is
238                // the tempting rule and it is wrong: it reads `* CSS Variables` as the
239                // keyword `CSS`, which on this corpus produced 23 false positives and
240                // zero true ones. An audit that overstates a gap is worse than no audit,
241                // because it argues for work nobody needs.
242                self.count(Scope::Out, "TODO keyword (custom sequence)", at);
243            }
244            if rest.contains("[#") {
245                self.count(Scope::In, "priority cookie", at);
246            }
247            if rest.trim_end().ends_with(':') && rest.trim_end().matches(':').count() >= 2 {
248                self.count(Scope::In, "heading tags", at);
249            }
250            if rest.contains("[/") || rest.contains("[%") {
251                self.count(Scope::Out, "statistics cookie", at);
252            }
253            return;
254        }
255
256        // --- planning and clocking ---
257        for marker in ["SCHEDULED:", "DEADLINE:", "CLOSED:"] {
258            if trimmed.starts_with(marker) {
259                self.count(Scope::Out, "planning line", at);
260            }
261        }
262        if trimmed.starts_with("CLOCK:") {
263            self.count(Scope::Out, "clock entry", at);
264        }
265
266        // --- keywords and drawers ---
267        if let Some(rest) = trimmed.strip_prefix("#+") {
268            if let Some(colon) = rest.find(':') {
269                let key = rest[..colon].trim().to_ascii_uppercase();
270                if !key.is_empty() && !key.contains(char::is_whitespace) {
271                    self.count_census(Census::Keyword, &key, at);
272                    match key.as_str() {
273                        "CAPTION" | "NAME" | "ATTR_HTML" => {
274                            self.count(Scope::In, "affiliated keyword", at)
275                        }
276                        // Not a gap: org's HTML exporter does not recalculate `#+TBLFM:`
277                        // either, so the cells as written are what both exporters emit.
278                        // `fixtures/tblfm.org` holds the oracle to that (`tests/oracle.rs`).
279                        // Unlike `#+INCLUDE:`, nothing is lost by leaving it inert.
280                        "TBLFM" => self.count(Scope::In, "table formula (#+TBLFM:)", at),
281                        "INCLUDE" => self.count(Scope::Out, "#+INCLUDE:", at),
282                        "RESULTS" => self.count(Scope::Out, "babel results block", at),
283                        "TODO" => self.count(Scope::Out, "#+TODO: keyword sequence", at),
284                        "MACRO" => self.count(Scope::Out, "macro definition", at),
285                        _ => self.count(Scope::In, "#+ keyword", at),
286                    }
287                }
288            }
289        } else if is_drawer(trimmed) {
290            let name = trimmed[1..trimmed.len() - 1].to_ascii_uppercase();
291            if name != "END" {
292                self.count_census(Census::Drawer, &name, at);
293                match name.as_str() {
294                    "PROPERTIES" => self.count(Scope::In, "property drawer", at),
295                    _ => self.count(Scope::Out, "non-PROPERTIES drawer", at),
296                }
297            }
298        }
299
300        // --- lists, tables, rules ---
301        if let Some(bullet) = list_bullet(trimmed) {
302            self.count(Scope::In, "list item", at);
303            if bullet == Bullet::Ordered {
304                self.count(Scope::In, "ordered list", at);
305            }
306            let indent = line.len() - trimmed.len();
307            if indent > 0 {
308                self.count(Scope::In, "nested list item", at);
309            }
310            if trimmed.contains(" :: ") {
311                self.count(Scope::In, "description list", at);
312            }
313            let after = trimmed.trim_start_matches(['-', '+', '*', ' ']);
314            if after.starts_with("[ ]") || after.starts_with("[X]") || after.starts_with("[-]") {
315                self.count(Scope::In, "checkbox", at);
316            }
317        }
318        if trimmed.starts_with('|') {
319            self.count(Scope::In, "table row", at);
320        }
321        if trimmed.starts_with(':') && !is_drawer(trimmed) && trimmed.starts_with(": ") {
322            self.count(Scope::Out, "fixed-width line", at);
323        }
324
325        // --- footnotes ---
326        if trimmed.starts_with("[fn:") {
327            self.count(Scope::In, "footnote definition", at);
328        } else if line.contains("[fn:") {
329            self.count(Scope::In, "footnote reference", at);
330        }
331
332        // --- inline objects ---
333        self.scan_inline(line, at);
334    }
335
336    fn scan_inline(&mut self, line: &str, at: &str) {
337        // Links: count each `[[target]]`, censusing its scheme.
338        let mut rest = line;
339        while let Some(start) = rest.find("[[") {
340            let after = &rest[start + 2..];
341            let Some(end) = after.find("]]") else { break };
342            let inner = &after[..end];
343            let target = inner.split("][").next().unwrap_or(inner);
344            self.count(Scope::In, "link", at);
345            self.count_census(Census::Scheme, &link_scheme(target), at);
346            rest = &after[end..];
347        }
348
349        // The rest read a line that has had its `=verbatim=` and `~code~` spans blanked:
350        // a construct shown inside verbatim is displayed rather than rendered, so a page
351        // documenting org syntax is not a use of the syntax it names. Emphasis pairs
352        // below deliberately still see the raw line — `=` is one of the markers scanned.
353        let bare = without_literal_spans(line);
354
355        if has_timestamp(&bare) {
356            self.count(Scope::In, "timestamp", at);
357        }
358        if bare.contains("{{{") {
359            self.count(Scope::Out, "macro call", at);
360        }
361        if bare.contains("<<<") {
362            self.count(Scope::Out, "radio target", at);
363        } else if bare.contains("<<") && bare.contains(">>") {
364            self.count(Scope::Out, "internal target", at);
365        }
366        if bare.contains("\\begin{") || latex_inline(&bare) {
367            self.count(Scope::Out, "LaTeX fragment", at);
368        }
369        if entity_ref(&bare) {
370            // Rendered, and rendered as org renders it: `\alpha` becomes `&alpha;` in
371            // both exporters. `fixtures/audit-entities.org` holds the oracle to that.
372            self.count(Scope::In, "entity (\\name)", at);
373        }
374        for (marker, name) in [
375            ('*', "bold"),
376            ('/', "italic"),
377            ('_', "underline"),
378            ('+', "strike-through"),
379            ('=', "verbatim"),
380            ('~', "code"),
381        ] {
382            if emphasis_pair(line, marker) {
383                self.count(Scope::In, name, at);
384            }
385        }
386    }
387
388    fn count(&mut self, scope: Scope, name: &'static str, at: &str) {
389        let tally = self.constructs.entry((scope, name)).or_default();
390        bump(tally, at);
391    }
392
393    fn count_census(&mut self, kind: Census, name: &str, at: &str) {
394        let map = match kind {
395            Census::Keyword => &mut self.keywords,
396            Census::Block => &mut self.blocks,
397            Census::Drawer => &mut self.drawers,
398            Census::Scheme => &mut self.link_schemes,
399        };
400        let tally = map.entry(name.to_string()).or_default();
401        bump(tally, at);
402    }
403}
404
405fn bump(tally: &mut Tally, at: &str) {
406    tally.occurrences += 1;
407    if !tally.seen_in_current_file {
408        tally.seen_in_current_file = true;
409        tally.files += 1;
410    }
411    if tally.first_seen.is_none() {
412        tally.first_seen = Some(at.to_string());
413    }
414}
415
416// ---------------------------------------------------------------------------
417// Line-level detectors. Deliberately independent of the parser (see module docs).
418// ---------------------------------------------------------------------------
419
420fn heading_stars(line: &str) -> Option<usize> {
421    if !line.starts_with('*') {
422        return None;
423    }
424    let stars = line.chars().take_while(|c| *c == '*').count();
425    let after = &line[stars..];
426    (after.starts_with(' ') || after.is_empty()).then_some(stars)
427}
428
429#[derive(PartialEq)]
430enum Bullet {
431    Unordered,
432    Ordered,
433}
434
435fn list_bullet(trimmed: &str) -> Option<Bullet> {
436    let bytes = trimmed.as_bytes();
437    if bytes.is_empty() {
438        return None;
439    }
440    if (bytes[0] == b'-' || bytes[0] == b'+') && (bytes.len() == 1 || bytes[1] == b' ') {
441        return Some(Bullet::Unordered);
442    }
443    let digits = trimmed.chars().take_while(|c| c.is_ascii_digit()).count();
444    if digits > 0 {
445        let after = &trimmed[digits..];
446        if (after.starts_with('.') || after.starts_with(')'))
447            && (after.len() == 1 || after.as_bytes()[1] == b' ')
448        {
449            return Some(Bullet::Ordered);
450        }
451    }
452    None
453}
454
455fn is_drawer(trimmed: &str) -> bool {
456    let t = trimmed.trim_end();
457    t.len() >= 3
458        && t.starts_with(':')
459        && t.ends_with(':')
460        && t[1..t.len() - 1]
461            .chars()
462            .all(|c| c.is_ascii_alphanumeric() || c == '_' || c == '-')
463        && t.len() > 2
464}
465
466fn block_construct(kind: &str) -> &'static str {
467    match kind.to_ascii_uppercase().as_str() {
468        "SRC" => "source block",
469        "QUOTE" => "quote block",
470        "EXAMPLE" => "example block",
471        "CENTER" => "center block",
472        "EXPORT" => "export block",
473        "VERSE" => "verse block",
474        "COMMENT" => "comment block",
475        _ => "special block",
476    }
477}
478
479/// The scheme of a link target, normalized into the census's vocabulary.
480fn link_scheme(target: &str) -> String {
481    if let Some(rest) = target.split_once(':') {
482        let scheme = rest.0;
483        if !scheme.is_empty()
484            && scheme
485                .chars()
486                .all(|c| c.is_ascii_alphanumeric() || c == '-' || c == '+')
487        {
488            return scheme.to_ascii_lowercase();
489        }
490    }
491    if target.starts_with('#') {
492        return "custom-id".to_string();
493    }
494    if target.starts_with('*') {
495        return "heading".to_string();
496    }
497    "relative".to_string()
498}
499
500/// A `<...>`/`[...]` span opening with an ISO date is a timestamp.
501fn has_timestamp(line: &str) -> bool {
502    let bytes = line.as_bytes();
503    for (i, c) in line.char_indices() {
504        if c != '<' && c != '[' {
505            continue;
506        }
507        let rest = &bytes[i + 1..];
508        if rest.len() >= 10
509            && rest[..4].iter().all(u8::is_ascii_digit)
510            && rest[4] == b'-'
511            && rest[5..7].iter().all(u8::is_ascii_digit)
512            && rest[7] == b'-'
513            && rest[8..10].iter().all(u8::is_ascii_digit)
514        {
515            return true;
516        }
517    }
518    false
519}
520
521/// `$x$` or `\(x\)` inline math. `$` alone (a price, a shell prompt) is not math.
522fn latex_inline(line: &str) -> bool {
523    if line.contains("\\(") && line.contains("\\)") {
524        return true;
525    }
526    let dollars = line.matches('$').count();
527    dollars >= 2 && line.contains("$\\")
528}
529
530/// A `\name` entity reference such as `\alpha`.
531///
532/// Only names org knows count, matching [`crate::parser`]'s rule for rendering one: a
533/// Windows path (`C:\Users\me`) and a namespaced identifier (`Tumblr\API\Client`) are not
534/// entity references.
535fn entity_ref(line: &str) -> bool {
536    let chars: Vec<char> = line.chars().collect();
537    for (i, c) in chars.iter().enumerate() {
538        if *c != '\\' {
539            continue;
540        }
541        let name: String = chars[i + 1..].iter().take_while(|c| c.is_ascii_alphabetic()).collect();
542        if crate::entities::lookup(&name).is_some() {
543            return true;
544        }
545    }
546    false
547}
548
549/// Blank out `=verbatim=` and `~code~` spans. Deliberately looser than the parser's
550/// border rules — the audit measures prevalence, and erring toward blanking keeps it
551/// from overstating a gap.
552fn without_literal_spans(line: &str) -> String {
553    let mut out = String::with_capacity(line.len());
554    let mut open: Option<char> = None;
555    for c in line.chars() {
556        match open {
557            Some(marker) => {
558                out.push(' ');
559                if c == marker {
560                    open = None;
561                }
562            }
563            None if c == '=' || c == '~' => {
564                open = Some(c);
565                out.push(' ');
566            }
567            None => out.push(c),
568        }
569    }
570    out
571}
572
573/// A plausible `*bold*`-style emphasis pair: two markers on one line with non-space
574/// content between them. Approximate by design — the audit measures prevalence, and the
575/// parser owns the exact pre/post-character rules.
576fn emphasis_pair(line: &str, marker: char) -> bool {
577    let positions: Vec<usize> = line
578        .char_indices()
579        .filter(|(_, c)| *c == marker)
580        .map(|(i, _)| i)
581        .collect();
582    if positions.len() < 2 {
583        return false;
584    }
585    // A leading `*` is a heading, and `-`/`+` at line start is a bullet.
586    let trimmed = line.trim_start();
587    if trimmed.starts_with(marker) {
588        return false;
589    }
590    positions.windows(2).any(|w| w[1] > w[0] + 1)
591}
592
593// ---------------------------------------------------------------------------
594// Report
595// ---------------------------------------------------------------------------
596
597/// Render the audit as a readable report. Names, counts and locations only — never
598/// document text, so an audit of private notes is safe to paste into an issue.
599pub fn report(audit: &Audit) -> String {
600    let mut out = String::new();
601    out.push_str(&format!(
602        "corpus: {} file(s), {} line(s)\n",
603        audit.files, audit.lines
604    ));
605
606    let mut rows: Vec<(&(Scope, &str), &Tally)> = audit.constructs.iter().collect();
607    rows.sort_by(|a, b| {
608        b.1.occurrences
609            .cmp(&a.1.occurrences)
610            .then_with(|| a.0 .1.cmp(b.0 .1))
611    });
612
613    out.push_str("\nCONSTRUCTS (by frequency)\n");
614    out.push_str(&format!(
615        "{:<4} {:<32} {:>8} {:>7}  {}\n",
616        "", "construct", "uses", "files", "first seen"
617    ));
618    for ((scope, name), tally) in &rows {
619        out.push_str(&format!(
620            "{:<4} {:<32} {:>8} {:>7}  {}\n",
621            scope.label(),
622            name,
623            tally.occurrences,
624            tally.files,
625            tally.first_seen.as_deref().unwrap_or("")
626        ));
627    }
628
629    let in_uses: usize = rows
630        .iter()
631        .filter(|((s, _), _)| *s == Scope::In)
632        .map(|(_, t)| t.occurrences)
633        .sum();
634    let out_uses: usize = rows
635        .iter()
636        .filter(|((s, _), _)| *s == Scope::Out)
637        .map(|(_, t)| t.occurrences)
638        .sum();
639    let total = in_uses + out_uses;
640    let pct = |n: usize| {
641        if total == 0 {
642            0.0
643        } else {
644            100.0 * n as f64 / total as f64
645        }
646    };
647    out.push_str(&format!(
648        "\ncoverage: {in_uses} in-scope use(s) ({:.1}%), {out_uses} out-of-scope ({:.1}%)\n",
649        pct(in_uses),
650        pct(out_uses)
651    ));
652
653    for (title, kind, map) in [
654        ("KEYWORDS", Census::Keyword, &audit.keywords),
655        ("BLOCK TYPES", Census::Block, &audit.blocks),
656        ("DRAWERS", Census::Drawer, &audit.drawers),
657        ("LINK SCHEMES", Census::Scheme, &audit.link_schemes),
658    ] {
659        let mut names: Vec<(&String, &Tally)> = map.iter().collect();
660        names.sort_by(|a, b| b.1.occurrences.cmp(&a.1.occurrences).then(a.0.cmp(b.0)));
661        out.push_str(&format!("\n{title}\n"));
662        for (name, tally) in names {
663            out.push_str(&format!(
664                "{:<4}{:<32} {:>8} {:>7}  {}\n",
665                Audit::marker(kind, name),
666                name,
667                tally.occurrences,
668                tally.files,
669                tally.first_seen.as_deref().unwrap_or("")
670            ));
671        }
672    }
673    out.push_str(
674        "\n`???` marks a name the implementation does not recognize at all.\n\
675         `\u{2014}` marks a keyword with no dedicated handling; it reaches templates \
676         as `page.keywords.<name>`.\n",
677    );
678    out
679}