Commit d6911e2a61
Verified · cmc
Layout: unified · split
README.md +32 −12
| @@ -281,18 +281,38 @@ therefore returns only what was written, and the report is assembled sequentiall | |||
| 281 | `parallel_builds_are_deterministic_in_output_and_report_order` holds that line, and it was | 281 | `parallel_builds_are_deterministic_in_output_and_report_order` holds that line, and it was |
| 282 | verified by reintroducing the bug and watching it fail. | 282 | verified by reintroducing the bug and watching it fail. |
| 283 | 283 | ||
| 284 | ### The real scaling limit is not the CPU | 284 | ### The real scaling limit was not the CPU |
| 285 | 285 | ||
| 286 | Going 10× on corpus size cost 17× in time before parallelism, which is superlinear — and | 286 | Going 10× on corpus size cost 17× in time, which parallelism improves without fixing: the |
| 287 | parallelism moves that constant without fixing it. The cause is the nav bar: it lists **every** | 287 | cause was the nav bar listing **every** page, so an *n*-page site emitted *n*² nav links. At |
| 288 | page, so an *n*-page site emits *n*² nav links. At 1,790 pages each page carries 1,799 links | 288 | 1,790 pages each page carried 1,799 links and the output was 284 MB, against 5.5 MB for the |
| 289 | and the output is 284 MB, against 5.5 MB for the 179-page corpus — 52× the bytes for 10× the | 289 | 179-page corpus — 52× the bytes for 10× the input. |
| 290 | input. Even at the real corpus size this is already visible: 18 KB pages whose nav dwarfs the | 290 | |
| 291 | prose, where the live site's nav has about six links. | 291 | The nav is now built from **top-level pages only** ([`is_top_level`](src/site.rs)): a nav is a |
| 292 | 292 | map of the site's top level, not an index of its contents, and section pages reach their | |
| 293 | This is a template and configuration question rather than a bug — *which* pages belong in a | 293 | siblings through that section's landing page. Nav size becomes a function of the top level |
| 294 | nav is a decision this project has not made yet — so it is recorded here rather than guessed | 294 | rather than of the corpus, and the quadratic disappears. |
| 295 | at. Until it is made, a build's cost is dominated by chrome nobody asked for. | 295 | |
| 296 | | 1,790-page corpus (6 top-level pages) | before | after | | ||
| 297 | |---|---|---| | ||
| 298 | | full build | 0.82s | 0.39s | | ||
| 299 | | total output | 284 MB | 34 MB | | ||
| 300 | | nav links per page | 1,799 | 6 | | ||
| 301 | |||
| 302 | Scaling is now linear: 179 pages in 0.07s and 1,796 in 0.39s, where the small case is mostly | ||
| 303 | the fixed cost of loading syntect's syntax definitions. | ||
| 304 | |||
| 305 | The same rule sharpened the incremental build, which is the larger win. The site-structure | ||
| 306 | hash — the thing that forces a global re-render — now covers only the pages that appear in | ||
| 307 | the nav, because those are the only ones whose title or URL affects another page. **Adding a | ||
| 308 | blog post used to re-render the entire site; now it renders one page.** A top-level page's | ||
| 309 | title still invalidates everything, correctly, since every page displays it. | ||
| 310 | |||
| 311 | **Trade-off worth knowing:** on a site whose sections live in subdirectories, only genuinely | ||
| 312 | root-level pages appear. cleberg.net keeps its landing pages at `content/salary/index.org` | ||
| 313 | and friends, so its nav comes out as a single `index.org` entry where the live site shows | ||
| 314 | four. Treating a directory's `index.org` as top-level too is a one-line change to | ||
| 315 | `is_top_level` if that is the behaviour you want. | ||
| 296 | 316 | ||
| 297 | **From v0.1 (core subset):** headings with nesting and anchors (every heading is now | 317 | **From v0.1 (core subset):** headings with nesting and anchors (every heading is now |
| 298 | anchored — `:CUSTOM_ID:`/`:ID:` else a slug of its text) and trailing tags; paragraphs; | 318 | anchored — `:CUSTOM_ID:`/`:ID:` else a slug of its text) and trailing tags; paragraphs; |
src/site.rs +28 −6
| @@ -127,18 +127,23 @@ fn prepare_pages(src: &Utf8Path) -> Result<(Vec<PagePrep>, SymbolTable)> { | |||
| 127 | symbols.index_document(doc); | 127 | symbols.index_document(doc); |
| 128 | } | 128 | } |
| 129 | 129 | ||
| 130 | // Nav is global; titles come from #+TITLE (falling back to the file stem) and URLs | 130 | // Nav is global chrome; titles come from #+TITLE (falling back to the file stem) and |
| 131 | // from each page's output path, which `#+SLUG:` can rename. | 131 | // URLs from each page's output path, which `#+SLUG:` can rename. |
| 132 | let entries: Vec<(Utf8PathBuf, String)> = docs | 132 | let all_pages: Vec<(Utf8PathBuf, String)> = docs |
| 133 | .iter() | 133 | .iter() |
| 134 | .map(|d| (output_path(&d.source_path, &d.keywords), page_title(d))) | 134 | .map(|d| (output_path(&d.source_path, &d.keywords), page_title(d))) |
| 135 | .collect(); | 135 | .collect(); |
| 136 | let entries: Vec<(Utf8PathBuf, String)> = all_pages | ||
| 137 | .iter() | ||
| 138 | .filter(|(out, _)| is_top_level(out)) | ||
| 139 | .cloned() | ||
| 140 | .collect(); | ||
| 136 | 141 | ||
| 137 | // Two sources emitting one page would silently drop a page — and with slugs, a | 142 | // Two sources emitting one page would silently drop a page — and with slugs, a |
| 138 | // collision is a typo away and invisible in the source filenames. | 143 | // collision is a typo away and invisible in the source filenames. |
| 139 | let mut claimed: std::collections::HashMap<&Utf8PathBuf, &Utf8PathBuf> = | 144 | let mut claimed: std::collections::HashMap<&Utf8PathBuf, &Utf8PathBuf> = |
| 140 | std::collections::HashMap::new(); | 145 | std::collections::HashMap::new(); |
| 141 | for (doc, (out, _)) in docs.iter().zip(&entries) { | 146 | for (doc, (out, _)) in docs.iter().zip(&all_pages) { |
| 142 | if let Some(other) = claimed.insert(out, &doc.source_path) { | 147 | if let Some(other) = claimed.insert(out, &doc.source_path) { |
| 143 | anyhow::bail!( | 148 | anyhow::bail!( |
| 144 | "output collision: {} and {} both build to {out} (check their #+SLUG:)", | 149 | "output collision: {} and {} both build to {out} (check their #+SLUG:)", |
| @@ -239,10 +244,16 @@ pub fn build_site(src: &Utf8Path, out: &Utf8Path, opts: &BuildOptions) -> Result | |||
| 239 | // chrome on every page — is built from every page's (path, title), so a title/path | 244 | // chrome on every page — is built from every page's (path, title), so a title/path |
| 240 | // change or a page add/remove must re-render every page (else stale nav on disk). | 245 | // change or a page add/remove must re-render every page (else stale nav on disk). |
| 241 | let cfg = BuildConfig::default(); | 246 | let cfg = BuildConfig::default(); |
| 242 | // Keyed on the *output* path: a `#+SLUG:` change moves a page's URL, which changes | 247 | // Only the pages that actually appear in the nav belong in the site-structure hash, |
| 243 | // the nav on every other page even though no source filename moved. | 248 | // because the nav is the only global chrome a page carries. Hashing *every* page |
| 249 | // here would mean adding one blog post re-rendered the entire site — correct, but | ||
| 250 | // needlessly: a nested page cannot change any other page's nav. | ||
| 251 | // | ||
| 252 | // Keyed on the *output* path, since a `#+SLUG:` change moves a page's URL — and so | ||
| 253 | // its nav link — even though no source filename moved. | ||
| 244 | let nav_entries: Vec<(String, String)> = preps | 254 | let nav_entries: Vec<(String, String)> = preps |
| 245 | .iter() | 255 | .iter() |
| 256 | .filter(|p| is_top_level(&p.output)) | ||
| 246 | .map(|p| (p.output.to_string(), p.title.clone())) | 257 | .map(|p| (p.output.to_string(), p.title.clone())) |
| 247 | .collect(); | 258 | .collect(); |
| 248 | let cfg_hash = combine(config_hash(&cfg), site_structure_hash(&nav_entries)); | 259 | let cfg_hash = combine(config_hash(&cfg), site_structure_hash(&nav_entries)); |
| @@ -489,6 +500,17 @@ fn discover(src: &Utf8Path) -> Result<(Vec<Utf8PathBuf>, Vec<Utf8PathBuf>)> { | |||
| 489 | Ok((org, assets)) | 500 | Ok((org, assets)) |
| 490 | } | 501 | } |
| 491 | 502 | ||
| 503 | /// Does this output path sit at the site root? | ||
| 504 | /// | ||
| 505 | /// The nav is the site's global chrome, and listing *every* page in it makes an `n`-page | ||
| 506 | /// site emit `n²` nav links — 1,790 pages produced 284 MB of output, most of it nav. A | ||
| 507 | /// nav is a map of the site's top level, not an index of its contents, so it is built | ||
| 508 | /// from root-level pages only. Section pages reach their siblings through that section's | ||
| 509 | /// own landing page. | ||
| 510 | fn is_top_level(output: &Utf8Path) -> bool { | ||
| 511 | output.parent().is_none_or(|p| p.as_str().is_empty()) | ||
| 512 | } | ||
| 513 | |||
| 492 | fn page_title(doc: &Document) -> String { | 514 | fn page_title(doc: &Document) -> String { |
| 493 | doc.keywords | 515 | doc.keywords |
| 494 | .entries | 516 | .entries |
tests/incremental.rs +87
| @@ -361,3 +361,90 @@ fn parallel_builds_are_deterministic_in_output_and_report_order() { | |||
| 361 | ); | 361 | ); |
| 362 | } | 362 | } |
| 363 | } | 363 | } |
| 364 | |||
| 365 | /// A site with pages in subdirectories. | ||
| 366 | fn write_nested_site(src: &Utf8PathBuf) { | ||
| 367 | std::fs::create_dir_all(src.join("blog")).unwrap(); | ||
| 368 | write(src, "index.org", "#+TITLE: Home\n\nWelcome.\n"); | ||
| 369 | write(src, "about.org", "#+TITLE: About\n\nAbout me.\n"); | ||
| 370 | write(&src.join("blog"), "first.org", "#+TITLE: First Post\n\nPost body.\n"); | ||
| 371 | write(&src.join("blog"), "second.org", "#+TITLE: Second Post\n\nPost body.\n"); | ||
| 372 | } | ||
| 373 | |||
| 374 | /// The nav is a map of the site's top level, not an index of its contents. Listing every | ||
| 375 | /// page made an n-page site emit n² nav links: 1,790 pages produced 284 MB of output, | ||
| 376 | /// nearly all of it nav. | ||
| 377 | #[test] | ||
| 378 | fn nav_lists_only_top_level_pages() { | ||
| 379 | let root = tmpdir("navtop"); | ||
| 380 | let src = root.join("src"); | ||
| 381 | std::fs::create_dir_all(&src).unwrap(); | ||
| 382 | write_nested_site(&src); | ||
| 383 | let out_dir = root.join("out"); | ||
| 384 | |||
| 385 | build_site(&src, &out_dir, &BuildOptions::default()).unwrap(); | ||
| 386 | let home = std::fs::read_to_string(out_dir.join("index.html")).unwrap(); | ||
| 387 | let nav = home | ||
| 388 | .split("<nav>") | ||
| 389 | .nth(1) | ||
| 390 | .and_then(|s| s.split("</nav>").next()) | ||
| 391 | .expect("a nav element"); | ||
| 392 | |||
| 393 | assert!(nav.contains("About"), "a root-level page belongs in the nav:\n{nav}"); | ||
| 394 | assert!(nav.contains("Home"), "the index page belongs in the nav:\n{nav}"); | ||
| 395 | assert!( | ||
| 396 | !nav.contains("First Post") && !nav.contains("Second Post"), | ||
| 397 | "pages in subdirectories must not appear in the nav:\n{nav}" | ||
| 398 | ); | ||
| 399 | |||
| 400 | // Nested pages still get the nav — they just are not *in* it. | ||
| 401 | let post = std::fs::read_to_string(out_dir.join("blog/first.html")).unwrap(); | ||
| 402 | assert!( | ||
| 403 | post.contains("href=\"../about.html\"") && post.contains("href=\"../index.html\""), | ||
| 404 | "a nested page links up to the top-level nav:\n{post}" | ||
| 405 | ); | ||
| 406 | } | ||
| 407 | |||
| 408 | /// The payoff for narrowing the site-structure hash to nav entries. Adding a blog post | ||
| 409 | /// cannot change any other page's nav, so it must not re-render the site — which is what | ||
| 410 | /// hashing *every* page's (path, title) used to force. | ||
| 411 | #[test] | ||
| 412 | fn adding_a_nested_page_does_not_rebuild_the_site() { | ||
| 413 | let root = tmpdir("navadd"); | ||
| 414 | let src = root.join("src"); | ||
| 415 | std::fs::create_dir_all(&src).unwrap(); | ||
| 416 | write_nested_site(&src); | ||
| 417 | let out_dir = root.join("out"); | ||
| 418 | |||
| 419 | build_site(&src, &out_dir, &BuildOptions::default()).unwrap(); | ||
| 420 | |||
| 421 | write(&src.join("blog"), "third.org", "#+TITLE: Third Post\n\nBody.\n"); | ||
| 422 | let r = build_site(&src, &out_dir, &BuildOptions::default()).unwrap(); | ||
| 423 | |||
| 424 | assert_eq!( | ||
| 425 | r.rendered, | ||
| 426 | vec![out("blog/third.html")], | ||
| 427 | "only the new nested page renders, got: {:?}", | ||
| 428 | r.rendered | ||
| 429 | ); | ||
| 430 | assert_eq!(r.skipped.len(), 4, "every pre-existing page is reused"); | ||
| 431 | } | ||
| 432 | |||
| 433 | /// The other half of the same rule: a page that IS in the nav still invalidates | ||
| 434 | /// everything when its title changes, because every page renders that title. | ||
| 435 | #[test] | ||
| 436 | fn retitling_a_top_level_page_still_rebuilds_the_site() { | ||
| 437 | let root = tmpdir("navretitle"); | ||
| 438 | let src = root.join("src"); | ||
| 439 | std::fs::create_dir_all(&src).unwrap(); | ||
| 440 | write_nested_site(&src); | ||
| 441 | let out_dir = root.join("out"); | ||
| 442 | |||
| 443 | build_site(&src, &out_dir, &BuildOptions::default()).unwrap(); | ||
| 444 | write(&src, "about.org", "#+TITLE: Colophon\n\nAbout me.\n"); | ||
| 445 | let r = build_site(&src, &out_dir, &BuildOptions::default()).unwrap(); | ||
| 446 | |||
| 447 | assert_eq!(r.rendered.len(), 4, "a nav title change re-renders every page"); | ||
| 448 | let post = std::fs::read_to_string(out_dir.join("blog/first.html")).unwrap(); | ||
| 449 | assert!(post.contains("Colophon"), "nested pages show the updated nav title"); | ||
| 450 | } | ||