Commit f828151d5f
Verified · cmc
Layout: unified · split
Cargo.lock +53 −1
| @@ -258,12 +258,43 @@ dependencies = [ | |||
| 258 | "cfg-if", | 258 | "cfg-if", |
| 259 | ] | 259 | ] |
| 260 | 260 | ||
| 261 | [[package]] | ||
| 262 | name = "crossbeam-deque" | ||
| 263 | version = "0.8.7" | ||
| 264 | source = "registry+https://github.com/rust-lang/crates.io-index" | ||
| 265 | checksum = "5181e0de7b61eb03a81e347d6dd8797bae9da5146707b51077e2d71a54ec0ceb" | ||
| 266 | dependencies = [ | ||
| 267 | "crossbeam-epoch", | ||
| 268 | "crossbeam-utils", | ||
| 269 | ] | ||
| 270 | |||
| 271 | [[package]] | ||
| 272 | name = "crossbeam-epoch" | ||
| 273 | version = "0.9.20" | ||
| 274 | source = "registry+https://github.com/rust-lang/crates.io-index" | ||
| 275 | checksum = "2d6914041f254d6e9176c01941b21115dcfb7089e55135a35411081bd106ef3f" | ||
| 276 | dependencies = [ | ||
| 277 | "crossbeam-utils", | ||
| 278 | ] | ||
| 279 | |||
| 280 | [[package]] | ||
| 281 | name = "crossbeam-utils" | ||
| 282 | version = "0.8.22" | ||
| 283 | source = "registry+https://github.com/rust-lang/crates.io-index" | ||
| 284 | checksum = "61803da095bee82a81bb1a452ecc25d3b2f1416d1897eb86430c6159ef717c17" | ||
| 285 | |||
| 261 | [[package]] | 286 | [[package]] |
| 262 | name = "deranged" | 287 | name = "deranged" |
| 263 | version = "0.5.8" | 288 | version = "0.5.8" |
| 264 | source = "registry+https://github.com/rust-lang/crates.io-index" | 289 | source = "registry+https://github.com/rust-lang/crates.io-index" |
| 265 | checksum = "7cd812cc2bc1d69d4764bd80df88b4317eaef9e773c75226407d9bc0876b211c" | 290 | checksum = "7cd812cc2bc1d69d4764bd80df88b4317eaef9e773c75226407d9bc0876b211c" |
| 266 | 291 | ||
| 292 | [[package]] | ||
| 293 | name = "either" | ||
| 294 | version = "1.17.0" | ||
| 295 | source = "registry+https://github.com/rust-lang/crates.io-index" | ||
| 296 | checksum = "9e5e8f6c15a24b9a3ee5efec809ccd006d3b30e8b3bb63c39af737c7f87daa1d" | ||
| 297 | |||
| 267 | [[package]] | 298 | [[package]] |
| 268 | name = "encode_unicode" | 299 | name = "encode_unicode" |
| 269 | version = "1.0.0" | 300 | version = "1.0.0" |
| @@ -538,7 +569,7 @@ dependencies = [ | |||
| 538 | 569 | ||
| 539 | [[package]] | 570 | [[package]] |
| 540 | name = "org-ssg" | 571 | name = "org-ssg" |
| 541 | version = "0.4.0" | 572 | version = "0.5.0" |
| 542 | dependencies = [ | 573 | dependencies = [ |
| 543 | "anyhow", | 574 | "anyhow", |
| 544 | "blake3", | 575 | "blake3", |
| @@ -547,6 +578,7 @@ dependencies = [ | |||
| 547 | "clap", | 578 | "clap", |
| 548 | "insta", | 579 | "insta", |
| 549 | "minijinja", | 580 | "minijinja", |
| 581 | "rayon", | ||
| 550 | "serde", | 582 | "serde", |
| 551 | "serde_json", | 583 | "serde_json", |
| 552 | "syntect", | 584 | "syntect", |
| @@ -618,6 +650,26 @@ version = "6.0.0" | |||
| 618 | source = "registry+https://github.com/rust-lang/crates.io-index" | 650 | source = "registry+https://github.com/rust-lang/crates.io-index" |
| 619 | checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf" | 651 | checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf" |
| 620 | 652 | ||
| 653 | [[package]] | ||
| 654 | name = "rayon" | ||
| 655 | version = "1.12.0" | ||
| 656 | source = "registry+https://github.com/rust-lang/crates.io-index" | ||
| 657 | checksum = "fb39b166781f92d482534ef4b4b1b2568f42613b53e5b6c160e24cfbfa30926d" | ||
| 658 | dependencies = [ | ||
| 659 | "either", | ||
| 660 | "rayon-core", | ||
| 661 | ] | ||
| 662 | |||
| 663 | [[package]] | ||
| 664 | name = "rayon-core" | ||
| 665 | version = "1.13.0" | ||
| 666 | source = "registry+https://github.com/rust-lang/crates.io-index" | ||
| 667 | checksum = "22e18b0f0062d30d4230b2e85ff77fdfe4326feb054b9783a3460d8435c8ab91" | ||
| 668 | dependencies = [ | ||
| 669 | "crossbeam-deque", | ||
| 670 | "crossbeam-utils", | ||
| 671 | ] | ||
| 672 | |||
| 621 | [[package]] | 673 | [[package]] |
| 622 | name = "regex-syntax" | 674 | name = "regex-syntax" |
| 623 | version = "0.8.11" | 675 | version = "0.8.11" |
Cargo.toml +2 −1
| @@ -1,6 +1,6 @@ | |||
| 1 | [package] | 1 | [package] |
| 2 | name = "org-ssg" | 2 | name = "org-ssg" |
| 3 | version = "0.4.0" | 3 | version = "0.5.0" |
| 4 | edition = "2021" | 4 | edition = "2021" |
| 5 | description = "Org-mode static site generator that renders the org element tree straight to HTML" | 5 | description = "Org-mode static site generator that renders the org element tree straight to HTML" |
| 6 | license = "MIT" | 6 | license = "MIT" |
| @@ -30,6 +30,7 @@ blake3 = "1" | |||
| 30 | clap = { version = "4", features = ["derive"] } | 30 | clap = { version = "4", features = ["derive"] } |
| 31 | anyhow = "1" | 31 | anyhow = "1" |
| 32 | thiserror = "2" | 32 | thiserror = "2" |
| 33 | rayon = "1.12.0" | ||
| 33 | 34 | ||
| 34 | [dev-dependencies] | 35 | [dev-dependencies] |
| 35 | insta = { version = "1", features = ["json"] } | 36 | insta = { version = "1", features = ["json"] } |
README.md +60 −6
| @@ -71,7 +71,7 @@ all-of-org. Phase 0 checked this line against a real 179-file corpus and found i | |||
| 71 | | 4 | Rendering to HTML — tree walk, tables, footnote two-pass, minijinja templating, syntect highlighting | done | | 71 | | 4 | Rendering to HTML — tree walk, tables, footnote two-pass, minijinja templating, syntect highlighting | done | |
| 72 | | 5 | Link resolution + symbol table (INDEX + RESOLVE, used-target list, broken-link reporting) | done | | 72 | | 5 | Link resolution + symbol table (INDEX + RESOLVE, used-target list, broken-link reporting) | done | |
| 73 | | 6 | Incremental build layer (hashing, dep graph, invalidation) done; `watch` is a simple poll loop | done | | 73 | | 6 | Incremental build layer (hashing, dep graph, invalidation) done; `watch` is a simple poll loop | done | |
| 74 | | 7 | Hardening: rayon parallelism, error locations in parse diagnostics | todo | | 74 | | **7** | **Hardening: rayon parallelism, error locations in parse diagnostics** | **done** | |
| 75 | 75 | ||
| 76 | ### v0.2 in / out | 76 | ### v0.2 in / out |
| 77 | 77 | ||
| @@ -164,9 +164,9 @@ excluded construct to a specific degradation: babel is never executed *and* a ch | |||
| 164 | as literal text; drawers other than PROPERTIES are captured and dropped; unmodelled block | 164 | as literal text; drawers other than PROPERTIES are captured and dropped; unmodelled block |
| 165 | types keep their content verbatim. | 165 | types keep their content verbatim. |
| 166 | 166 | ||
| 167 | **Still out at v0.4:** rayon parallelism; parse errors carrying source locations; `#+TODO:` | 167 | **Still out:** `#+TODO:` per-file keyword sequences; planning lines |
| 168 | per-file keyword sequences; planning lines (`SCHEDULED:`/`DEADLINE:`), which render as | 168 | (`SCHEDULED:`/`DEADLINE:`), which render as ordinary paragraphs; fixed-width `: ` lines; |
| 169 | ordinary paragraphs; fixed-width `: ` lines; and the `watch` fs-notify integration. | 169 | and the `watch` fs-notify integration. |
| 170 | 170 | ||
| 171 | ## Phase 0: the corpus audit and the Emacs oracle | 171 | ## Phase 0: the corpus audit and the Emacs oracle |
| 172 | 172 | ||
| @@ -241,6 +241,59 @@ corrupted (it trimmed each of syntect's per-token text runs, turning `def greet` | |||
| 241 | were measurement artifacts. A differential harness is a piece of software like any other, | 241 | were measurement artifacts. A differential harness is a piece of software like any other, |
| 242 | and the first divergences it reports are usually its own. | 242 | and the first divergences it reports are usually its own. |
| 243 | 243 | ||
| 244 | ## Phase 7: hardening | ||
| 245 | |||
| 246 | ### Parse diagnostics (`file:line: message`) | ||
| 247 | |||
| 248 | The parser's contract is that it always returns a document — out-of-scope and malformed | ||
| 249 | constructs degrade rather than crash. The gap was that they degraded *silently*, and in the | ||
| 250 | worst cases the degradation is severe: an unterminated `#+BEGIN_SRC` reads the rest of the | ||
| 251 | file as block content, and an unterminated drawer does the same but renders to nothing, so | ||
| 252 | one missing line deletes most of a page from a build that reports success. | ||
| 253 | |||
| 254 | `parse` now returns `Document::diagnostics`, each carrying a 1-based source line, and the | ||
| 255 | build prints them as `file:line: message`. `--strict` turns them (and unresolved links) into | ||
| 256 | a non-zero exit. Line numbers are threaded as an absolute offset through every nested parse, | ||
| 257 | so a block inside a list item inside a section still reports its real file line — there is a | ||
| 258 | test for exactly that, because reconstructed and re-indented nested slices are precisely | ||
| 259 | where an off-by-N hides. The 179-file corpus produces zero diagnostics. | ||
| 260 | |||
| 261 | ### Parallelism | ||
| 262 | |||
| 263 | PARSE, RESOLVE and RENDER/EMIT run under rayon. PARSE is a pure function of one file's bytes | ||
| 264 | and RESOLVE only reads the shared symbol table, which is what makes both safe to parallelize | ||
| 265 | at all; INDEX stays sequential. | ||
| 266 | |||
| 267 | | corpus | before | after | speedup | | ||
| 268 | |---|---|---|---| | ||
| 269 | | 179 files (real) | 0.23s | 0.07s | 3.3× | | ||
| 270 | | 1,790 files (10× copy) | 3.98s | 0.82s | 4.9× | | ||
| 271 | |||
| 272 | Measured on 12 cores. `RAYON_NUM_THREADS=1` reproduces the old 3.98s exactly, so the gain is | ||
| 273 | parallelism rather than incidental change, and the output is byte-identical to the sequential | ||
| 274 | build across the whole corpus. | ||
| 275 | |||
| 276 | **Parallelism must not be observable in the result.** `par_iter().collect()` preserves input | ||
| 277 | order, so the emitted bytes are unaffected — but the build *report* is the fragile half: | ||
| 278 | pushing to `rendered`/`skipped` from inside the parallel pass would order them by thread | ||
| 279 | scheduling, producing a non-deterministic report over a deterministic site. The parallel pass | ||
| 280 | therefore returns only what was written, and the report is assembled sequentially afterwards. | ||
| 281 | `parallel_builds_are_deterministic_in_output_and_report_order` holds that line, and it was | ||
| 282 | verified by reintroducing the bug and watching it fail. | ||
| 283 | |||
| 284 | ### The real scaling limit is not the CPU | ||
| 285 | |||
| 286 | Going 10× on corpus size cost 17× in time before parallelism, which is superlinear — and | ||
| 287 | parallelism moves that constant without fixing it. The cause is the nav bar: it lists **every** | ||
| 288 | page, so an *n*-page site emits *n*² nav links. At 1,790 pages each page carries 1,799 links | ||
| 289 | and the output is 284 MB, against 5.5 MB for the 179-page corpus — 52× the bytes for 10× the | ||
| 290 | input. Even at the real corpus size this is already visible: 18 KB pages whose nav dwarfs the | ||
| 291 | prose, where the live site's nav has about six links. | ||
| 292 | |||
| 293 | This is a template and configuration question rather than a bug — *which* pages belong in a | ||
| 294 | nav is a decision this project has not made yet — so it is recorded here rather than guessed | ||
| 295 | at. Until it is made, a build's cost is dominated by chrome nobody asked for. | ||
| 296 | |||
| 244 | **From v0.1 (core subset):** headings with nesting and anchors (every heading is now | 297 | **From v0.1 (core subset):** headings with nesting and anchors (every heading is now |
| 245 | anchored — `:CUSTOM_ID:`/`:ID:` else a slug of its text) and trailing tags; paragraphs; | 298 | anchored — `:CUSTOM_ID:`/`:ID:` else a slug of its text) and trailing tags; paragraphs; |
| 246 | plain lists (unordered + ordered) with checkboxes; source blocks; inline markup (`*bold*`, | 299 | plain lists (unordered + ordered) with checkboxes; source blocks; inline markup (`*bold*`, |
| @@ -251,8 +304,9 @@ plain lists (unordered + ordered) with checkboxes; source blocks; inline markup | |||
| 251 | Parser is hand-written recursive descent (not `nom`/`chumsky`/`pest` — org is | 304 | Parser is hand-written recursive descent (not `nom`/`chumsky`/`pest` — org is |
| 252 | line-oriented and context-sensitive, not clean CFG). Key crates: `syntect` (syntax | 305 | line-oriented and context-sensitive, not clean CFG). Key crates: `syntect` (syntax |
| 253 | highlighting, behind a `Highlighter` trait so tree-sitter can be swapped in later), | 306 | highlighting, behind a `Highlighter` trait so tree-sitter can be swapped in later), |
| 254 | `minijinja` (runtime templates), `blake3` (content/cache hashing), `chrono`, | 307 | `minijinja` (runtime templates), `blake3` (content/cache hashing), `rayon` (parallel |
| 255 | `camino`, `walkdir`, `clap`, `anyhow`/`thiserror`. `insta` for snapshot tests. | 308 | PARSE/RESOLVE/RENDER), `chrono`, `camino`, `walkdir`, `clap`, `anyhow`/`thiserror`. |
| 309 | `insta` for snapshot tests, and `emacs --batch` — optional, and only for the oracle. | ||
| 256 | 310 | ||
| 257 | ## Build & test | 311 | ## Build & test |
| 258 | 312 | ||
src/main.rs +6 −2
| @@ -72,14 +72,15 @@ fn main() -> Result<()> { | |||
| 72 | let opts = BuildOptions { no_cache, strict }; | 72 | let opts = BuildOptions { no_cache, strict }; |
| 73 | let report = build_site(&input, &out, &opts)?; | 73 | let report = build_site(&input, &out, &opts)?; |
| 74 | println!( | 74 | println!( |
| 75 | "built {} page(s) ({} rendered, {} cached), copied {} asset(s) from {} -> {} ({} unresolved link(s))", | 75 | "built {} page(s) ({} rendered, {} cached), copied {} asset(s) from {} -> {} ({} unresolved link(s), {} diagnostic(s))", |
| 76 | report.pages.len(), | 76 | report.pages.len(), |
| 77 | report.rendered.len(), | 77 | report.rendered.len(), |
| 78 | report.skipped.len(), | 78 | report.skipped.len(), |
| 79 | report.assets.len(), | 79 | report.assets.len(), |
| 80 | input, | 80 | input, |
| 81 | out, | 81 | out, |
| 82 | report.broken.len() | 82 | report.broken.len(), |
| 83 | report.diagnostics.len() | ||
| 83 | ); | 84 | ); |
| 84 | } else { | 85 | } else { |
| 85 | let output = output.unwrap_or_else(|| input.with_extension("html")); | 86 | let output = output.unwrap_or_else(|| input.with_extension("html")); |
| @@ -170,6 +171,9 @@ fn build_file(input: &Utf8Path, output: &Utf8Path) -> Result<()> { | |||
| 170 | let source = fs::read_to_string(input) | 171 | let source = fs::read_to_string(input) |
| 171 | .with_context(|| format!("reading source file {input}"))?; | 172 | .with_context(|| format!("reading source file {input}"))?; |
| 172 | let document = parse(input, &source).with_context(|| format!("parsing {input}"))?; | 173 | let document = parse(input, &source).with_context(|| format!("parsing {input}"))?; |
| 174 | for d in &document.diagnostics { | ||
| 175 | eprintln!("warning: {input}:{}: {}", d.line, d.message); | ||
| 176 | } | ||
| 173 | 177 | ||
| 174 | let title = document | 178 | let title = document |
| 175 | .keywords | 179 | .keywords |
src/model.rs +16
| @@ -36,6 +36,19 @@ pub struct TodoKeyword { | |||
| 36 | pub done: bool, | 36 | pub done: bool, |
| 37 | } | 37 | } |
| 38 | 38 | ||
| 39 | /// A problem found while parsing, carrying the 1-based source line it was found on. | ||
| 40 | /// | ||
| 41 | /// Diagnostics are warnings, not errors: the parser's contract is that it always returns | ||
| 42 | /// a document (spec §1 — out-of-scope constructs degrade, never crash). What a warning | ||
| 43 | /// buys is that degrading stops being *silent*, which matters most exactly where the | ||
| 44 | /// damage is largest — an unterminated `#+BEGIN_SRC` swallows the rest of the file. | ||
| 45 | #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] | ||
| 46 | pub struct Diagnostic { | ||
| 47 | /// 1-based line number in the source file. | ||
| 48 | pub line: usize, | ||
| 49 | pub message: String, | ||
| 50 | } | ||
| 51 | |||
| 39 | /// One source file → one Document. This is the unit of parsing and caching (spec §2.3). | 52 | /// One source file → one Document. This is the unit of parsing and caching (spec §2.3). |
| 40 | #[derive(Debug, Clone, Serialize, Deserialize)] | 53 | #[derive(Debug, Clone, Serialize, Deserialize)] |
| 41 | pub struct Document { | 54 | pub struct Document { |
| @@ -44,6 +57,9 @@ pub struct Document { | |||
| 44 | pub keywords: Keywords, | 57 | pub keywords: Keywords, |
| 45 | /// Pre-first-heading content plus child headings. | 58 | /// Pre-first-heading content plus child headings. |
| 46 | pub root: Section, | 59 | pub root: Section, |
| 60 | /// Non-fatal problems found while parsing this file. | ||
| 61 | #[serde(default)] | ||
| 62 | pub diagnostics: Vec<Diagnostic>, | ||
| 47 | } | 63 | } |
| 48 | 64 | ||
| 49 | /// A section = content directly under a heading (or the file preamble), followed by | 65 | /// A section = content directly under a heading (or the file preamble), followed by |
src/parser.rs +109 −25
| @@ -25,8 +25,8 @@ use chrono::{NaiveDate, NaiveDateTime, NaiveTime}; | |||
| 25 | 25 | ||
| 26 | use crate::model::{ | 26 | use crate::model::{ |
| 27 | BlockParams, Bullet, Checkbox, ContentHash, Document, Element, Heading, Keywords, Link, | 27 | BlockParams, Bullet, Checkbox, ContentHash, Document, Element, Heading, Keywords, Link, |
| 28 | LinkTarget, List, ListItem, ListKind, Object, Properties, Section, Table, TableRow, Timestamp, | 28 | Diagnostic, LinkTarget, List, ListItem, ListKind, Object, Properties, Section, Table, TableRow, |
| 29 | TodoKeyword, | 29 | Timestamp, TodoKeyword, |
| 30 | }; | 30 | }; |
| 31 | 31 | ||
| 32 | #[derive(Debug, thiserror::Error)] | 32 | #[derive(Debug, thiserror::Error)] |
| @@ -104,6 +104,7 @@ pub fn parse(path: &Utf8Path, source: &str) -> Result<Document, ParseError> { | |||
| 104 | let lines: Vec<&str> = source.lines().collect(); | 104 | let lines: Vec<&str> = source.lines().collect(); |
| 105 | let classes = line_lexer(source); | 105 | let classes = line_lexer(source); |
| 106 | 106 | ||
| 107 | let mut diagnostics: Vec<Diagnostic> = Vec::new(); | ||
| 107 | let mut keywords = Keywords::default(); | 108 | let mut keywords = Keywords::default(); |
| 108 | let mut root = Section { | 109 | let mut root = Section { |
| 109 | heading: None, | 110 | heading: None, |
| @@ -135,7 +136,7 @@ pub fn parse(path: &Utf8Path, source: &str) -> Result<Document, ParseError> { | |||
| 135 | } | 136 | } |
| 136 | } | 137 | } |
| 137 | } | 138 | } |
| 138 | root.content = parse_elements(&lines[..first]); | 139 | root.content = parse_elements(&lines[..first], 0, &mut diagnostics); |
| 139 | } | 140 | } |
| 140 | 141 | ||
| 141 | // Each heading segment runs from its own line up to (but excluding) the next heading. | 142 | // Each heading segment runs from its own line up to (but excluding) the next heading. |
| @@ -144,7 +145,8 @@ pub fn parse(path: &Utf8Path, source: &str) -> Result<Document, ParseError> { | |||
| 144 | let end = heading_idxs.get(k + 1).copied().unwrap_or(lines.len()); | 145 | let end = heading_idxs.get(k + 1).copied().unwrap_or(lines.len()); |
| 145 | let heading = parse_heading(lines[h_idx]); | 146 | let heading = parse_heading(lines[h_idx]); |
| 146 | let level = heading.level; | 147 | let level = heading.level; |
| 147 | let (heading, content) = parse_section_body(heading, &lines[h_idx + 1..end]); | 148 | let (heading, content) = |
| 149 | parse_section_body(heading, &lines[h_idx + 1..end], h_idx + 1, &mut diagnostics); | ||
| 148 | flat.push(( | 150 | flat.push(( |
| 149 | level, | 151 | level, |
| 150 | Section { | 152 | Section { |
| @@ -158,11 +160,13 @@ pub fn parse(path: &Utf8Path, source: &str) -> Result<Document, ParseError> { | |||
| 158 | let mut pos = 0; | 160 | let mut pos = 0; |
| 159 | root.children = build_children(&mut flat, &mut pos, 0); | 161 | root.children = build_children(&mut flat, &mut pos, 0); |
| 160 | 162 | ||
| 163 | diagnostics.sort_by_key(|d| d.line); | ||
| 161 | Ok(Document { | 164 | Ok(Document { |
| 162 | source_path: path.to_owned(), | 165 | source_path: path.to_owned(), |
| 163 | content_hash, | 166 | content_hash, |
| 164 | keywords, | 167 | keywords, |
| 165 | root, | 168 | root, |
| 169 | diagnostics, | ||
| 166 | }) | 170 | }) |
| 167 | } | 171 | } |
| 168 | 172 | ||
| @@ -320,17 +324,25 @@ fn is_tag_cluster(s: &str) -> bool { | |||
| 320 | // Section body: property drawer + block content | 324 | // Section body: property drawer + block content |
| 321 | // --------------------------------------------------------------------------- | 325 | // --------------------------------------------------------------------------- |
| 322 | 326 | ||
| 323 | fn parse_section_body(mut heading: Heading, body: &[&str]) -> (Heading, Vec<Element>) { | 327 | fn parse_section_body( |
| 328 | mut heading: Heading, | ||
| 329 | body: &[&str], | ||
| 330 | base: usize, | ||
| 331 | diags: &mut Vec<Diagnostic>, | ||
| 332 | ) -> (Heading, Vec<Element>) { | ||
| 324 | let mut idx = 0; | 333 | let mut idx = 0; |
| 325 | while idx < body.len() && body[idx].trim().is_empty() { | 334 | while idx < body.len() && body[idx].trim().is_empty() { |
| 326 | idx += 1; | 335 | idx += 1; |
| 327 | } | 336 | } |
| 328 | if idx < body.len() && body[idx].trim().eq_ignore_ascii_case(":PROPERTIES:") { | 337 | if idx < body.len() && body[idx].trim().eq_ignore_ascii_case(":PROPERTIES:") { |
| 338 | let opened_at = base + idx; | ||
| 339 | let mut terminated = false; | ||
| 329 | idx += 1; | 340 | idx += 1; |
| 330 | while idx < body.len() { | 341 | while idx < body.len() { |
| 331 | let t = body[idx].trim(); | 342 | let t = body[idx].trim(); |
| 332 | if t.eq_ignore_ascii_case(":END:") { | 343 | if t.eq_ignore_ascii_case(":END:") { |
| 333 | idx += 1; | 344 | idx += 1; |
| 345 | terminated = true; | ||
| 334 | break; | 346 | break; |
| 335 | } | 347 | } |
| 336 | if let Some((k, v)) = parse_property(t) { | 348 | if let Some((k, v)) = parse_property(t) { |
| @@ -343,8 +355,16 @@ fn parse_section_body(mut heading: Heading, body: &[&str]) -> (Heading, Vec<Elem | |||
| 343 | } | 355 | } |
| 344 | idx += 1; | 356 | idx += 1; |
| 345 | } | 357 | } |
| 358 | if !terminated { | ||
| 359 | diags.push(Diagnostic { | ||
| 360 | line: opened_at + 1, | ||
| 361 | message: "unterminated :PROPERTIES: drawer (no :END:); the rest of the \ | ||
| 362 | section was read as properties" | ||
| 363 | .to_string(), | ||
| 364 | }); | ||
| 365 | } | ||
| 346 | } | 366 | } |
| 347 | let content = parse_elements(&body[idx..]); | 367 | let content = parse_elements(&body[idx..], base + idx, diags); |
| 348 | (heading, content) | 368 | (heading, content) |
| 349 | } | 369 | } |
| 350 | 370 | ||
| @@ -365,7 +385,10 @@ fn parse_property(line: &str) -> Option<(String, String)> { | |||
| 365 | // Block-level element builder | 385 | // Block-level element builder |
| 366 | // --------------------------------------------------------------------------- | 386 | // --------------------------------------------------------------------------- |
| 367 | 387 | ||
| 368 | fn parse_elements(lines: &[&str]) -> Vec<Element> { | 388 | /// Build the block elements of `lines`. `base` is the absolute 0-based index of |
| 389 | /// `lines[0]` in the source file, so diagnostics can name a real line number however | ||
| 390 | /// deeply nested the construct is. | ||
| 391 | fn parse_elements(lines: &[&str], base: usize, diags: &mut Vec<Diagnostic>) -> Vec<Element> { | ||
| 369 | let mut out = Vec::new(); | 392 | let mut out = Vec::new(); |
| 370 | // Affiliated keywords (`#+CAPTION:` and friends) belong to the element that follows | 393 | // Affiliated keywords (`#+CAPTION:` and friends) belong to the element that follows |
| 371 | // them, so they are held aside until that element is built. | 394 | // them, so they are held aside until that element is built. |
| @@ -393,7 +416,7 @@ fn parse_elements(lines: &[&str]) -> Vec<Element> { | |||
| 393 | i += 1; | 416 | i += 1; |
| 394 | continue; | 417 | continue; |
| 395 | } | 418 | } |
| 396 | let (element, next) = parse_one_element(lines, i); | 419 | let (element, next) = parse_one_element(lines, i, base, diags); |
| 397 | i = next; | 420 | i = next; |
| 398 | if std::mem::take(&mut drop_next) { | 421 | if std::mem::take(&mut drop_next) { |
| 399 | affiliated.clear(); | 422 | affiliated.clear(); |
| @@ -409,17 +432,22 @@ fn parse_elements(lines: &[&str]) -> Vec<Element> { | |||
| 409 | /// Build the single element starting at `lines[start]`, returning it with the index of | 432 | /// Build the single element starting at `lines[start]`, returning it with the index of |
| 410 | /// the first line past it. `None` means the lines were consumed without producing an | 433 | /// the first line past it. `None` means the lines were consumed without producing an |
| 411 | /// element. `start` is guaranteed non-blank and not an affiliated keyword. | 434 | /// element. `start` is guaranteed non-blank and not an affiliated keyword. |
| 412 | fn parse_one_element(lines: &[&str], start: usize) -> (Option<Element>, usize) { | 435 | fn parse_one_element( |
| 436 | lines: &[&str], | ||
| 437 | start: usize, | ||
| 438 | base: usize, | ||
| 439 | diags: &mut Vec<Diagnostic>, | ||
| 440 | ) -> (Option<Element>, usize) { | ||
| 413 | let line = lines[start]; | 441 | let line = lines[start]; |
| 414 | if let Some(text) = comment_text(line) { | 442 | if let Some(text) = comment_text(line) { |
| 415 | return (Some(Element::Comment(text)), start + 1); | 443 | return (Some(Element::Comment(text)), start + 1); |
| 416 | } | 444 | } |
| 417 | if let Some((kind, after)) = block_begin(line) { | 445 | if let Some((kind, after)) = block_begin(line) { |
| 418 | let (el, next) = parse_block(lines, start, &kind, &after); | 446 | let (el, next) = parse_block(lines, start, &kind, &after, base, diags); |
| 419 | return (Some(el), next); | 447 | return (Some(el), next); |
| 420 | } | 448 | } |
| 421 | if let Some(name) = drawer_begin_name(line) { | 449 | if let Some(name) = drawer_begin_name(line) { |
| 422 | let (el, next) = parse_drawer(lines, start, name); | 450 | let (el, next) = parse_drawer(lines, start, name, base, diags); |
| 423 | return (Some(el), next); | 451 | return (Some(el), next); |
| 424 | } | 452 | } |
| 425 | if is_rule(line) { | 453 | if is_rule(line) { |
| @@ -434,7 +462,7 @@ fn parse_one_element(lines: &[&str], start: usize) -> (Option<Element>, usize) { | |||
| 434 | return (Some(def), next); | 462 | return (Some(def), next); |
| 435 | } | 463 | } |
| 436 | if is_list_item(line.trim_start()).is_some() { | 464 | if is_list_item(line.trim_start()).is_some() { |
| 437 | let (list, next) = parse_list(lines, start); | 465 | let (list, next) = parse_list(lines, start, base, diags); |
| 438 | return (Some(Element::List(list)), next); | 466 | return (Some(Element::List(list)), next); |
| 439 | } | 467 | } |
| 440 | // Paragraph: gather consecutive soft-wrapped text lines. | 468 | // Paragraph: gather consecutive soft-wrapped text lines. |
| @@ -449,8 +477,19 @@ fn parse_one_element(lines: &[&str], start: usize) -> (Option<Element>, usize) { | |||
| 449 | i += 1; | 477 | i += 1; |
| 450 | } | 478 | } |
| 451 | if para.is_empty() { | 479 | if para.is_empty() { |
| 452 | // `is_structural` said this line begins a construct that no branch above claimed | 480 | // `is_structural` said this line begins a construct that no branch above claimed. |
| 453 | // (a stray `#+END_`); skip it rather than looping forever. | 481 | // In practice that is a stray `#+END_`: a block terminator with nothing open. |
| 482 | // Skip it rather than looping forever, but say so — it usually means a `#+BEGIN_` | ||
| 483 | // above it is misspelled, and silence would leave the author hunting. | ||
| 484 | if is_block_end(line) { | ||
| 485 | diags.push(Diagnostic { | ||
| 486 | line: base + start + 1, | ||
| 487 | message: format!( | ||
| 488 | "stray `{}` with no matching `#+BEGIN_`", | ||
| 489 | line.split_whitespace().next().unwrap_or("#+END_") | ||
| 490 | ), | ||
| 491 | }); | ||
| 492 | } | ||
| 454 | return (None, start + 1); | 493 | return (None, start + 1); |
| 455 | } | 494 | } |
| 456 | (Some(Element::Paragraph(inline(¶.join(" ")))), i) | 495 | (Some(Element::Paragraph(inline(¶.join(" ")))), i) |
| @@ -478,13 +517,34 @@ fn is_structural(line: &str) -> bool { | |||
| 478 | /// Consume `#+BEGIN_<KIND> … #+END_<KIND>`. Matching is on the *specific* kind so a | 517 | /// Consume `#+BEGIN_<KIND> … #+END_<KIND>`. Matching is on the *specific* kind so a |
| 479 | /// source block can sit inside a quote block; an unterminated block runs to end of | 518 | /// source block can sit inside a quote block; an unterminated block runs to end of |
| 480 | /// input rather than failing. | 519 | /// input rather than failing. |
| 481 | fn parse_block(lines: &[&str], start: usize, kind: &str, after: &str) -> (Element, usize) { | 520 | fn parse_block( |
| 521 | lines: &[&str], | ||
| 522 | start: usize, | ||
| 523 | kind: &str, | ||
| 524 | after: &str, | ||
| 525 | base: usize, | ||
| 526 | diags: &mut Vec<Diagnostic>, | ||
| 527 | ) -> (Element, usize) { | ||
| 482 | let mut inner: Vec<&str> = Vec::new(); | 528 | let mut inner: Vec<&str> = Vec::new(); |
| 483 | let mut j = start + 1; | 529 | let mut j = start + 1; |
| 484 | while j < lines.len() && !is_block_end_of(lines[j], kind) { | 530 | while j < lines.len() && !is_block_end_of(lines[j], kind) { |
| 485 | inner.push(lines[j]); | 531 | inner.push(lines[j]); |
| 486 | j += 1; | 532 | j += 1; |
| 487 | } | 533 | } |
| 534 | if j >= lines.len() { | ||
| 535 | // Everything to the end of input was swallowed by the block. This is the single | ||
| 536 | // most destructive malformation in org: one missing line silently deletes the | ||
| 537 | // rest of the document from the output. | ||
| 538 | diags.push(Diagnostic { | ||
| 539 | line: base + start + 1, | ||
| 540 | message: format!( | ||
| 541 | "unterminated `#+BEGIN_{}` block (no `#+END_{}`); \ | ||
| 542 | everything to the end of the file was read as block content", | ||
| 543 | kind.to_ascii_uppercase(), | ||
| 544 | kind.to_ascii_uppercase() | ||
| 545 | ), | ||
| 546 | }); | ||
| 547 | } | ||
| 488 | let next = if j < lines.len() { j + 1 } else { j }; | 548 | let next = if j < lines.len() { j + 1 } else { j }; |
| 489 | let element = match kind.to_ascii_uppercase().as_str() { | 549 | let element = match kind.to_ascii_uppercase().as_str() { |
| 490 | "SRC" => { | 550 | "SRC" => { |
| @@ -496,8 +556,8 @@ fn parse_block(lines: &[&str], start: usize, kind: &str, after: &str) -> (Elemen | |||
| 496 | } | 556 | } |
| 497 | } | 557 | } |
| 498 | "EXAMPLE" => Element::ExampleBlock(inner.join("\n")), | 558 | "EXAMPLE" => Element::ExampleBlock(inner.join("\n")), |
| 499 | "QUOTE" => Element::QuoteBlock(parse_elements(&inner)), | 559 | "QUOTE" => Element::QuoteBlock(parse_elements(&inner, base + start + 1, diags)), |
| 500 | "CENTER" => Element::CenterBlock(parse_elements(&inner)), | 560 | "CENTER" => Element::CenterBlock(parse_elements(&inner, base + start + 1, diags)), |
| 501 | "EXPORT" => Element::ExportBlock { | 561 | "EXPORT" => Element::ExportBlock { |
| 502 | backend: after.split_whitespace().next().unwrap_or("").to_string(), | 562 | backend: after.split_whitespace().next().unwrap_or("").to_string(), |
| 503 | raw: inner.join("\n"), | 563 | raw: inner.join("\n"), |
| @@ -512,18 +572,35 @@ fn parse_block(lines: &[&str], start: usize, kind: &str, after: &str) -> (Elemen | |||
| 512 | /// `:NAME:` … `:END:` at block level. A PROPERTIES drawer directly under a heading is | 572 | /// `:NAME:` … `:END:` at block level. A PROPERTIES drawer directly under a heading is |
| 513 | /// consumed by [`parse_section_body`]; anything reaching here is a generic drawer, | 573 | /// consumed by [`parse_section_body`]; anything reaching here is a generic drawer, |
| 514 | /// which the renderer drops (README §OUT). | 574 | /// which the renderer drops (README §OUT). |
| 515 | fn parse_drawer(lines: &[&str], start: usize, name: String) -> (Element, usize) { | 575 | fn parse_drawer( |
| 576 | lines: &[&str], | ||
| 577 | start: usize, | ||
| 578 | name: String, | ||
| 579 | base: usize, | ||
| 580 | diags: &mut Vec<Diagnostic>, | ||
| 581 | ) -> (Element, usize) { | ||
| 516 | let mut inner: Vec<&str> = Vec::new(); | 582 | let mut inner: Vec<&str> = Vec::new(); |
| 517 | let mut j = start + 1; | 583 | let mut j = start + 1; |
| 518 | while j < lines.len() && !lines[j].trim().eq_ignore_ascii_case(":END:") { | 584 | while j < lines.len() && !lines[j].trim().eq_ignore_ascii_case(":END:") { |
| 519 | inner.push(lines[j]); | 585 | inner.push(lines[j]); |
| 520 | j += 1; | 586 | j += 1; |
| 521 | } | 587 | } |
| 588 | if j >= lines.len() { | ||
| 589 | // Drawers render to nothing, so an unterminated one deletes the rest of the file | ||
| 590 | // from the output just as thoroughly as an unterminated block — and more quietly. | ||
| 591 | diags.push(Diagnostic { | ||
| 592 | line: base + start + 1, | ||
| 593 | message: format!( | ||
| 594 | "unterminated `:{name}:` drawer (no `:END:`); everything to the end of \ | ||
| 595 | the file was read as drawer content and will not be rendered" | ||
| 596 | ), | ||
| 597 | }); | ||
| 598 | } | ||
| 522 | let next = if j < lines.len() { j + 1 } else { j }; | 599 | let next = if j < lines.len() { j + 1 } else { j }; |
| 523 | ( | 600 | ( |
| 524 | Element::Drawer { | 601 | Element::Drawer { |
| 525 | name, | 602 | name, |
| 526 | content: parse_elements(&inner), | 603 | content: parse_elements(&inner, base + start + 1, diags), |
| 527 | }, | 604 | }, |
| 528 | next, | 605 | next, |
| 529 | ) | 606 | ) |
| @@ -698,8 +775,13 @@ fn parse_footnote_def( | |||
| 698 | /// column; everything indented further is that item's body, re-parsed as block content — | 775 | /// column; everything indented further is that item's body, re-parsed as block content — |
| 699 | /// which is what makes lists nest. A single blank line does not end a list, but a blank | 776 | /// which is what makes lists nest. A single blank line does not end a list, but a blank |
| 700 | /// line followed by anything that is not a sibling bullet does. | 777 | /// line followed by anything that is not a sibling bullet does. |
| 701 | fn parse_list(lines: &[&str], start: usize) -> (List, usize) { | 778 | fn parse_list( |
| 702 | let base = indent_of(lines[start]); | 779 | lines: &[&str], |
| 780 | start: usize, | ||
| 781 | base: usize, | ||
| 782 | diags: &mut Vec<Diagnostic>, | ||
| 783 | ) -> (List, usize) { | ||
| 784 | let base_indent = indent_of(lines[start]); | ||
| 703 | let family = bullet_family(&is_list_item(lines[start].trim_start()).expect("list item")); | 785 | let family = bullet_family(&is_list_item(lines[start].trim_start()).expect("list item")); |
| 704 | // A list is a description list when its FIRST item carries a `::` term separator. | 786 | // A list is a description list when its FIRST item carries a `::` term separator. |
| 705 | let kind = match (&family, split_term(item_text(lines[start].trim_start()))) { | 787 | let kind = match (&family, split_term(item_text(lines[start].trim_start()))) { |
| @@ -716,7 +798,7 @@ fn parse_list(lines: &[&str], start: usize) -> (List, usize) { | |||
| 716 | while j < lines.len() && lines[j].trim().is_empty() { | 798 | while j < lines.len() && lines[j].trim().is_empty() { |
| 717 | j += 1; | 799 | j += 1; |
| 718 | } | 800 | } |
| 719 | if j >= lines.len() || indent_of(lines[j]) != base { | 801 | if j >= lines.len() || indent_of(lines[j]) != base_indent { |
| 720 | break; | 802 | break; |
| 721 | } | 803 | } |
| 722 | let Some(bullet) = is_list_item(lines[j].trim_start()) else { | 804 | let Some(bullet) = is_list_item(lines[j].trim_start()) else { |
| @@ -747,14 +829,14 @@ fn parse_list(lines: &[&str], start: usize) -> (List, usize) { | |||
| 747 | while k < lines.len() && lines[k].trim().is_empty() { | 829 | while k < lines.len() && lines[k].trim().is_empty() { |
| 748 | k += 1; | 830 | k += 1; |
| 749 | } | 831 | } |
| 750 | if k < lines.len() && indent_of(lines[k]) > base { | 832 | if k < lines.len() && indent_of(lines[k]) > base_indent { |
| 751 | body.resize(body.len() + (k - i), String::new()); | 833 | body.resize(body.len() + (k - i), String::new()); |
| 752 | i = k; | 834 | i = k; |
| 753 | continue; | 835 | continue; |
| 754 | } | 836 | } |
| 755 | break; | 837 | break; |
| 756 | } | 838 | } |
| 757 | if indent_of(lines[i]) <= base { | 839 | if indent_of(lines[i]) <= base_indent { |
| 758 | break; | 840 | break; |
| 759 | } | 841 | } |
| 760 | body.push(lines[i].to_string()); | 842 | body.push(lines[i].to_string()); |
| @@ -765,7 +847,9 @@ fn parse_list(lines: &[&str], start: usize) -> (List, usize) { | |||
| 765 | bullet, | 847 | bullet, |
| 766 | checkbox, | 848 | checkbox, |
| 767 | term, | 849 | term, |
| 768 | content: parse_elements(&dedent(&body)), | 850 | // The item body starts at the bullet line, so `base + j` is exact even after |
| 851 | // the body has been dedented into fresh strings. | ||
| 852 | content: parse_elements(&dedent(&body), base + j, diags), | ||
| 769 | }); | 853 | }); |
| 770 | } | 854 | } |
| 771 | (List { kind, items }, i) | 855 | (List { kind, items }, i) |
src/site.rs +92 −37
| @@ -14,6 +14,7 @@ use std::fs; | |||
| 14 | 14 | ||
| 15 | use anyhow::{Context, Result}; | 15 | use anyhow::{Context, Result}; |
| 16 | use camino::{Utf8Path, Utf8PathBuf}; | 16 | use camino::{Utf8Path, Utf8PathBuf}; |
| 17 | use rayon::prelude::*; | ||
| 17 | use walkdir::WalkDir; | 18 | use walkdir::WalkDir; |
| 18 | 19 | ||
| 19 | use crate::incremental::{ | 20 | use crate::incremental::{ |
| @@ -21,7 +22,7 @@ use crate::incremental::{ | |||
| 21 | template_hash, BuildConfig, DepGraph, Hash, Manifest, PageRecord, CACHE_FORMAT_VERSION, | 22 | template_hash, BuildConfig, DepGraph, Hash, Manifest, PageRecord, CACHE_FORMAT_VERSION, |
| 22 | }; | 23 | }; |
| 23 | use crate::index::{document_targets, SymbolTable, TargetId}; | 24 | use crate::index::{document_targets, SymbolTable, TargetId}; |
| 24 | use crate::model::{ContentHash, Document}; | 25 | use crate::model::{ContentHash, Diagnostic, Document}; |
| 25 | use crate::parser::parse; | 26 | use crate::parser::parse; |
| 26 | use crate::render::{render, syntax_css, Html, SyntectHighlighter}; | 27 | use crate::render::{render, syntax_css, Html, SyntectHighlighter}; |
| 27 | use crate::resolve::resolve; | 28 | use crate::resolve::resolve; |
| @@ -62,6 +63,26 @@ pub struct SiteReport { | |||
| 62 | pub assets: Vec<Utf8PathBuf>, | 63 | pub assets: Vec<Utf8PathBuf>, |
| 63 | /// Unresolved internal links: `(page, target)`. Warnings, not failures (spec §4.3.4). | 64 | /// Unresolved internal links: `(page, target)`. Warnings, not failures (spec §4.3.4). |
| 64 | pub broken: Vec<(Utf8PathBuf, TargetId)>, | 65 | pub broken: Vec<(Utf8PathBuf, TargetId)>, |
| 66 | /// Parse diagnostics: `(source file, diagnostic)`, in file then line order. | ||
| 67 | pub diagnostics: Vec<(Utf8PathBuf, Diagnostic)>, | ||
| 68 | } | ||
| 69 | |||
| 70 | impl SiteReport { | ||
| 71 | /// Every diagnostic and broken link, formatted one per line as | ||
| 72 | /// `file:line: message` — the form an editor can jump to. | ||
| 73 | pub fn warnings(&self) -> Vec<String> { | ||
| 74 | let mut out: Vec<String> = self | ||
| 75 | .diagnostics | ||
| 76 | .iter() | ||
| 77 | .map(|(path, d)| format!("{path}:{}: {}", d.line, d.message)) | ||
| 78 | .collect(); | ||
| 79 | out.extend( | ||
| 80 | self.broken | ||
| 81 | .iter() | ||
| 82 | .map(|(page, target)| format!("{page}: unresolved link {target}")), | ||
| 83 | ); | ||
| 84 | out | ||
| 85 | } | ||
| 65 | } | 86 | } |
| 66 | 87 | ||
| 67 | /// Everything a build needs about one page *before* the decision to render it: its | 88 | /// Everything a build needs about one page *before* the decision to render it: its |
| @@ -75,6 +96,7 @@ struct PagePrep { | |||
| 75 | used: HashSet<TargetId>, | 96 | used: HashSet<TargetId>, |
| 76 | defines: HashSet<TargetId>, | 97 | defines: HashSet<TargetId>, |
| 77 | broken: Vec<TargetId>, | 98 | broken: Vec<TargetId>, |
| 99 | diagnostics: Vec<Diagnostic>, | ||
| 78 | nav: Vec<NavItem>, | 100 | nav: Vec<NavItem>, |
| 79 | } | 101 | } |
| 80 | 102 | ||
| @@ -86,13 +108,18 @@ fn prepare_pages(src: &Utf8Path) -> Result<(Vec<PagePrep>, SymbolTable)> { | |||
| 86 | let (org_rel, _assets) = discover(src)?; | 108 | let (org_rel, _assets) = discover(src)?; |
| 87 | 109 | ||
| 88 | // PARSE every file (relative paths keep snapshots and links machine-independent). | 110 | // PARSE every file (relative paths keep snapshots and links machine-independent). |
| 89 | let mut docs: Vec<Document> = Vec::new(); | 111 | // PARSE is a pure function of one file's bytes (spec §2.1), which is exactly the |
| 90 | for rel in &org_rel { | 112 | // property that makes it safe to run in parallel. `par_iter().collect()` preserves |
| 91 | let abs = src.join(rel); | 113 | // input order, so the document list — and everything downstream of it — is identical |
| 92 | let source = fs::read_to_string(&abs).with_context(|| format!("reading {abs}"))?; | 114 | // to the sequential build regardless of how the work was scheduled. |
| 93 | let doc = parse(rel.as_path(), &source).with_context(|| format!("parsing {rel}"))?; | 115 | let docs: Vec<Document> = org_rel |
| 94 | docs.push(doc); | 116 | .par_iter() |
| 95 | } | 117 | .map(|rel| { |
| 118 | let abs = src.join(rel); | ||
| 119 | let source = fs::read_to_string(&abs).with_context(|| format!("reading {abs}"))?; | ||
| 120 | parse(rel.as_path(), &source).with_context(|| format!("parsing {rel}")) | ||
| 121 | }) | ||
| 122 | .collect::<Result<Vec<_>>>()?; | ||
| 96 | 123 | ||
| 97 | // INDEX: collect every link target across the corpus. | 124 | // INDEX: collect every link target across the corpus. |
| 98 | let mut symbols = SymbolTable::new(); | 125 | let mut symbols = SymbolTable::new(); |
| @@ -121,8 +148,11 @@ fn prepare_pages(src: &Utf8Path) -> Result<(Vec<PagePrep>, SymbolTable)> { | |||
| 121 | } | 148 | } |
| 122 | } | 149 | } |
| 123 | 150 | ||
| 124 | let mut pages = Vec::new(); | 151 | // RESOLVE reads the shared symbol table and writes only into its own page's output, |
| 125 | for doc in &docs { | 152 | // so it parallelizes for free once INDEX has finished building the table. |
| 153 | let pages: Vec<PagePrep> = docs | ||
| 154 | .par_iter() | ||
| 155 | .map(|doc| { | ||
| 126 | let out = resolve(doc, &symbols); | 156 | let out = resolve(doc, &symbols); |
| 127 | let used: HashSet<TargetId> = out.used_targets.iter().cloned().collect(); | 157 | let used: HashSet<TargetId> = out.used_targets.iter().cloned().collect(); |
| 128 | let broken: Vec<TargetId> = out.broken.iter().map(|b| b.target.clone()).collect(); | 158 | let broken: Vec<TargetId> = out.broken.iter().map(|b| b.target.clone()).collect(); |
| @@ -139,18 +169,20 @@ fn prepare_pages(src: &Utf8Path) -> Result<(Vec<PagePrep>, SymbolTable)> { | |||
| 139 | }) | 169 | }) |
| 140 | .collect(); | 170 | .collect(); |
| 141 | 171 | ||
| 142 | pages.push(PagePrep { | 172 | PagePrep { |
| 143 | source: doc.source_path.clone(), | 173 | source: doc.source_path.clone(), |
| 144 | output, | 174 | output, |
| 145 | title: page_title(doc), | 175 | title: page_title(doc), |
| 146 | content_hash: doc.content_hash, | 176 | content_hash: doc.content_hash, |
| 147 | resolved: out.resolved, | 177 | resolved: out.resolved, |
| 148 | used, | 178 | used, |
| 149 | defines, | 179 | defines, |
| 150 | broken, | 180 | broken, |
| 151 | nav, | 181 | diagnostics: doc.diagnostics.clone(), |
| 152 | }); | 182 | nav, |
| 153 | } | 183 | } |
| 184 | }) | ||
| 185 | .collect(); | ||
| 154 | 186 | ||
| 155 | Ok((pages, symbols)) | 187 | Ok((pages, symbols)) |
| 156 | } | 188 | } |
| @@ -270,22 +302,42 @@ pub fn build_site(src: &Utf8Path, out: &Utf8Path, opts: &BuildOptions) -> Result | |||
| 270 | let templater = Templater::new(); | 302 | let templater = Templater::new(); |
| 271 | let mut report = SiteReport::default(); | 303 | let mut report = SiteReport::default(); |
| 272 | 304 | ||
| 273 | for p in &preps { | 305 | // RENDER + TEMPLATE + EMIT, in parallel. This is where a build's time actually goes |
| 274 | for t in &p.broken { | 306 | // (syntect highlighting and templating dominate), and each page writes only its own |
| 275 | report.broken.push((p.source.clone(), t.clone())); | 307 | // file, so the pages are independent. |
| 276 | } | 308 | // |
| 277 | report.pages.push(p.output.clone()); | 309 | // The parallel pass returns whether each page was written; the report is assembled |
| 278 | 310 | // sequentially afterwards from `preps` order. Pushing to the report from inside the | |
| 279 | let dest = out.join(&p.output); | 311 | // parallel pass would make `rendered`/`skipped` ordering depend on thread scheduling, |
| 280 | if rebuild.contains(&p.source) { | 312 | // which would be a non-deterministic build report over a deterministic build. |
| 313 | let written: Vec<bool> = preps | ||
| 314 | .par_iter() | ||
| 315 | .map(|p| { | ||
| 316 | if !rebuild.contains(&p.source) { | ||
| 317 | // Skip: the on-disk output is already correct (spec §4.1). Leave it alone. | ||
| 318 | return Ok(false); | ||
| 319 | } | ||
| 320 | let dest = out.join(&p.output); | ||
| 281 | if let Some(parent) = dest.parent() { | 321 | if let Some(parent) = dest.parent() { |
| 282 | fs::create_dir_all(parent).with_context(|| format!("creating {parent}"))?; | 322 | fs::create_dir_all(parent).with_context(|| format!("creating {parent}"))?; |
| 283 | } | 323 | } |
| 284 | let html = render_page(&templater, &highlighter, p)?; | 324 | let html = render_page(&templater, &highlighter, p)?; |
| 285 | fs::write(&dest, &html).with_context(|| format!("writing {dest}"))?; | 325 | fs::write(&dest, &html).with_context(|| format!("writing {dest}"))?; |
| 326 | Ok(true) | ||
| 327 | }) | ||
| 328 | .collect::<Result<Vec<_>>>()?; | ||
| 329 | |||
| 330 | for (p, was_written) in preps.iter().zip(&written) { | ||
| 331 | for t in &p.broken { | ||
| 332 | report.broken.push((p.source.clone(), t.clone())); | ||
| 333 | } | ||
| 334 | for d in &p.diagnostics { | ||
| 335 | report.diagnostics.push((p.source.clone(), d.clone())); | ||
| 336 | } | ||
| 337 | report.pages.push(p.output.clone()); | ||
| 338 | if *was_written { | ||
| 286 | report.rendered.push(p.output.clone()); | 339 | report.rendered.push(p.output.clone()); |
| 287 | } else { | 340 | } else { |
| 288 | // Skip: the on-disk output is already correct (spec §4.1). Leave it untouched. | ||
| 289 | report.skipped.push(p.output.clone()); | 341 | report.skipped.push(p.output.clone()); |
| 290 | } | 342 | } |
| 291 | } | 343 | } |
| @@ -322,17 +374,20 @@ pub fn build_site(src: &Utf8Path, out: &Utf8Path, opts: &BuildOptions) -> Result | |||
| 322 | incremental::save_manifest(out, &manifest) | 374 | incremental::save_manifest(out, &manifest) |
| 323 | .with_context(|| format!("writing cache manifest under {out}"))?; | 375 | .with_context(|| format!("writing cache manifest under {out}"))?; |
| 324 | 376 | ||
| 325 | if opts.strict && !report.broken.is_empty() { | 377 | let warnings = report.warnings(); |
| 326 | for (page, target) in &report.broken { | 378 | if opts.strict && !warnings.is_empty() { |
| 327 | eprintln!("error: {page}: unresolved link {target}"); | 379 | for w in &warnings { |
| 380 | eprintln!("error: {w}"); | ||
| 328 | } | 381 | } |
| 329 | anyhow::bail!( | 382 | anyhow::bail!( |
| 330 | "{} unresolved internal link(s) under --strict", | 383 | "{} problem(s) under --strict ({} parse diagnostic(s), {} unresolved link(s))", |
| 384 | warnings.len(), | ||
| 385 | report.diagnostics.len(), | ||
| 331 | report.broken.len() | 386 | report.broken.len() |
| 332 | ); | 387 | ); |
| 333 | } | 388 | } |
| 334 | for (page, target) in &report.broken { | 389 | for w in &warnings { |
| 335 | eprintln!("warning: {page}: unresolved link {target}"); | 390 | eprintln!("warning: {w}"); |
| 336 | } | 391 | } |
| 337 | 392 | ||
| 338 | Ok(report) | 393 | Ok(report) |
tests/constructs.rs +99
| @@ -290,3 +290,102 @@ fn include_is_not_expanded() { | |||
| 290 | "`#+INCLUDE:` must not be expanded or echoed:\n{html}" | 290 | "`#+INCLUDE:` must not be expanded or echoed:\n{html}" |
| 291 | ); | 291 | ); |
| 292 | } | 292 | } |
| 293 | |||
| 294 | // --------------------------------------------------------------------------- | ||
| 295 | // Parse diagnostics: degrading is fine, degrading *silently* is not | ||
| 296 | // --------------------------------------------------------------------------- | ||
| 297 | |||
| 298 | fn diagnostics(source: &str) -> Vec<String> { | ||
| 299 | let document = parse(Utf8PathBuf::from("t.org").as_path(), source).expect("parse"); | ||
| 300 | document | ||
| 301 | .diagnostics | ||
| 302 | .iter() | ||
| 303 | .map(|d| format!("{}: {}", d.line, d.message)) | ||
| 304 | .collect() | ||
| 305 | } | ||
| 306 | |||
| 307 | /// An unterminated block swallows the rest of the file. The parser's contract is to | ||
| 308 | /// degrade rather than crash, so it still returns a document — but a silent one would | ||
| 309 | /// mean a build that reports success while deleting most of a page. | ||
| 310 | #[test] | ||
| 311 | fn unterminated_block_is_reported_with_its_line() { | ||
| 312 | let source = "#+TITLE: T\n\nIntro.\n\n* Section\n\n#+BEGIN_SRC rust\nfn main() {}\n\n* Vanishes\n"; | ||
| 313 | let found = diagnostics(source); | ||
| 314 | assert_eq!(found.len(), 1, "exactly one diagnostic: {found:?}"); | ||
| 315 | assert!( | ||
| 316 | found[0].starts_with("7: unterminated `#+BEGIN_SRC` block"), | ||
| 317 | "must name the line the block opened on: {found:?}" | ||
| 318 | ); | ||
| 319 | } | ||
| 320 | |||
| 321 | /// The same failure mode, and quieter: drawers render to nothing, so an unterminated one | ||
| 322 | /// deletes the rest of the file without even leaving a code block behind. | ||
| 323 | #[test] | ||
| 324 | fn unterminated_drawer_is_reported_with_its_line() { | ||
| 325 | let found = diagnostics("#+TITLE: T\n\n* Head\n:LOGBOOK:\nCLOCK: x\n\n* Lost\n"); | ||
| 326 | assert_eq!(found.len(), 1, "exactly one diagnostic: {found:?}"); | ||
| 327 | assert!( | ||
| 328 | found[0].starts_with("4: unterminated `:LOGBOOK:` drawer"), | ||
| 329 | "must name the drawer and its line: {found:?}" | ||
| 330 | ); | ||
| 331 | } | ||
| 332 | |||
| 333 | /// A stray terminator usually means the matching `#+BEGIN_` above it is misspelled. | ||
| 334 | #[test] | ||
| 335 | fn stray_block_end_is_reported_with_its_line() { | ||
| 336 | let found = diagnostics("#+TITLE: T\n\nText.\n\n#+END_SRC\n\nMore.\n"); | ||
| 337 | assert_eq!(found.len(), 1, "exactly one diagnostic: {found:?}"); | ||
| 338 | assert!( | ||
| 339 | found[0].starts_with("5: stray `#+END_SRC`"), | ||
| 340 | "must name the stray terminator and its line: {found:?}" | ||
| 341 | ); | ||
| 342 | } | ||
| 343 | |||
| 344 | /// Line numbers must survive nesting. A block inside a list item inside a section is | ||
| 345 | /// several levels of re-parsed, re-indented, reconstructed lines away from the file, and | ||
| 346 | /// a diagnostic that points at the wrong line is worse than none. | ||
| 347 | #[test] | ||
| 348 | fn diagnostic_lines_survive_nesting() { | ||
| 349 | let source = concat!( | ||
| 350 | "#+TITLE: T\n", // 1 | ||
| 351 | "\n", // 2 | ||
| 352 | "* Section\n", // 3 | ||
| 353 | "\n", // 4 | ||
| 354 | "- an item\n", // 5 | ||
| 355 | "\n", // 6 | ||
| 356 | " #+BEGIN_SRC sh\n", // 7 | ||
| 357 | " echo hi\n", // 8 | ||
| 358 | ); | ||
| 359 | let found = diagnostics(source); | ||
| 360 | assert_eq!(found.len(), 1, "exactly one diagnostic: {found:?}"); | ||
| 361 | assert!( | ||
| 362 | found[0].starts_with("7: unterminated"), | ||
| 363 | "the line must be the real file line, not an offset into a nested slice: {found:?}" | ||
| 364 | ); | ||
| 365 | } | ||
| 366 | |||
| 367 | /// Every fixture that is meant to be well-formed must parse without complaint — | ||
| 368 | /// otherwise the diagnostics are crying wolf on ordinary documents. | ||
| 369 | #[test] | ||
| 370 | fn well_formed_fixtures_produce_no_diagnostics() { | ||
| 371 | for name in [ | ||
| 372 | "minimal.org", | ||
| 373 | "core.org", | ||
| 374 | "elements.org", | ||
| 375 | "table.org", | ||
| 376 | "footnote.org", | ||
| 377 | "headings.org", | ||
| 378 | "lists.org", | ||
| 379 | "blocks.org", | ||
| 380 | "timestamps.org", | ||
| 381 | "images.org", | ||
| 382 | "outofscope.org", | ||
| 383 | ] { | ||
| 384 | let document = parse_fixture(name); | ||
| 385 | assert!( | ||
| 386 | document.diagnostics.is_empty(), | ||
| 387 | "{name} should parse cleanly, got {:?}", | ||
| 388 | document.diagnostics | ||
| 389 | ); | ||
| 390 | } | ||
| 391 | } | ||
tests/incremental.rs +67
| @@ -294,3 +294,70 @@ fn corrupt_cache_falls_back_without_crashing() { | |||
| 294 | let r = build_site(&src, &out_dir, &BuildOptions::default()).unwrap(); | 294 | let r = build_site(&src, &out_dir, &BuildOptions::default()).unwrap(); |
| 295 | assert_eq!(r.rendered.len(), 2, "a corrupt cache is never a correctness dependency"); | 295 | assert_eq!(r.rendered.len(), 2, "a corrupt cache is never a correctness dependency"); |
| 296 | } | 296 | } |
| 297 | |||
| 298 | /// PARSE, RESOLVE and RENDER/EMIT all run in parallel (rayon). Parallelism must not be | ||
| 299 | /// observable in the result: the emitted bytes and the *ordering* of the build report | ||
| 300 | /// have to be identical run to run, or a build stops being reproducible. | ||
| 301 | /// | ||
| 302 | /// The report ordering is the fragile half. Pushing to `rendered`/`skipped` from inside | ||
| 303 | /// the parallel pass would order them by thread scheduling, giving a non-deterministic | ||
| 304 | /// report over a deterministic site — so the report is assembled sequentially afterwards, | ||
| 305 | /// and this test is what holds that line. Enough pages to make a race likely if one exists. | ||
| 306 | #[test] | ||
| 307 | fn parallel_builds_are_deterministic_in_output_and_report_order() { | ||
| 308 | let root = tmpdir("parallel"); | ||
| 309 | let src = root.join("src"); | ||
| 310 | std::fs::create_dir_all(src.join("deep")).unwrap(); | ||
| 311 | |||
| 312 | for i in 0..40 { | ||
| 313 | // Cross-link every page to its neighbour so RESOLVE has real work, and give each | ||
| 314 | // a source block so RENDER does too. | ||
| 315 | let body = format!( | ||
| 316 | "#+TITLE: Page {i}\n#+SLUG: page-{i}\n\n\ | ||
| 317 | See [[#anchor-{next}][the next page]].\n\n\ | ||
| 318 | * Heading {i}\n:PROPERTIES:\n:CUSTOM_ID: anchor-{i}\n:END:\n\n\ | ||
| 319 | #+BEGIN_SRC rust\nfn page_{i}() -> u32 {{ {i} }}\n#+END_SRC\n", | ||
| 320 | next = (i + 1) % 40 | ||
| 321 | ); | ||
| 322 | let dir = if i % 3 == 0 { src.join("deep") } else { src.clone() }; | ||
| 323 | std::fs::write(dir.join(format!("p{i}.org")), body).unwrap(); | ||
| 324 | } | ||
| 325 | |||
| 326 | let build = |out: &Utf8PathBuf| { | ||
| 327 | build_site( | ||
| 328 | &src, | ||
| 329 | out, | ||
| 330 | &BuildOptions { | ||
| 331 | no_cache: true, | ||
| 332 | strict: false, | ||
| 333 | }, | ||
| 334 | ) | ||
| 335 | .unwrap() | ||
| 336 | }; | ||
| 337 | |||
| 338 | let first_out = root.join("first"); | ||
| 339 | let first = build(&first_out); | ||
| 340 | assert_eq!(first.rendered.len(), 40, "every page renders"); | ||
| 341 | |||
| 342 | for _ in 0..3 { | ||
| 343 | let out = tmpdir("parallel-again").join("out"); | ||
| 344 | let again = build(&out); | ||
| 345 | assert_eq!( | ||
| 346 | first.pages, again.pages, | ||
| 347 | "page ordering in the report must be deterministic" | ||
| 348 | ); | ||
| 349 | assert_eq!( | ||
| 350 | first.rendered, again.rendered, | ||
| 351 | "rendered ordering in the report must be deterministic" | ||
| 352 | ); | ||
| 353 | assert_eq!( | ||
| 354 | first.skipped, again.skipped, | ||
| 355 | "skipped ordering in the report must be deterministic" | ||
| 356 | ); | ||
| 357 | assert_eq!( | ||
| 358 | output_files(&first_out), | ||
| 359 | output_files(&out), | ||
| 360 | "emitted bytes must be identical across runs" | ||
| 361 | ); | ||
| 362 | } | ||
| 363 | } | ||