Commit 71e9f793b9
Unsigned
Layout: unified · split
README.md +2 −1
| @@ -94,7 +94,8 @@ each, measured back to back on one machine: | ||
| 94 | 94 | |
| 95 | 95 | That middle row is the interesting one. weblorg alone does not group a blog index by year, |
| 96 | 96 | write a tags page, rewrite image URLs, minify CSS or emit a sitemap — so I |
| 97 | wrote ~600 lines of Python to do those on top of it. orgo does the first three natively. | |
| 97 | wrote ~600 lines of Python to do those on top of it. orgo does four of the five natively — | |
| 98 | the sitemap included, since writing this table is what prompted it. | |
| 98 | 99 | |
| 99 | 100 | Read the numbers with three things in mind. The weblorg figures include Emacs starting and |
| 100 | 101 | loading its packages, which you pay on every publish and cannot avoid. orgo emits 13 pages |
docs/guide/02-configuration.org +12
| @@ -34,6 +34,7 @@ syntaxes_dir = "syntaxes" | ||
| 34 | 34 | [build] |
| 35 | 35 | drafts = false |
| 36 | 36 | assets = [] |
| 37 | sitemap = true | |
| 37 | 38 | |
| 38 | 39 | [html] |
| 39 | 40 | heading_offset = 1 |
| @@ -192,10 +193,21 @@ should not stop a site from building. | ||
| 192 | 193 | |-----+---------+---------| |
| 193 | 194 | | =drafts= | =false= | Include pages marked =#+DRAFT:=. | |
| 194 | 195 | | =assets= | =[]= | Extra directories copied to the *site root*. | |
| 196 | | =sitemap= | =true= | Write =sitemap.xml=. Needs =site.base_url=. | | |
| 195 | 197 | |
| 196 | 198 | =--drafts= on the command line turns this on for one run. The flag can only turn drafts |
| 197 | 199 | on; it never turns off a config that asked for them. |
| 198 | 200 | |
| 201 | ** sitemap.xml | |
| 202 | ||
| 203 | Every page the build emits, generated ones included — a crawler has no other way to learn | |
| 204 | that =/blog/= exists. =lastmod= is the page's own =#+DATE:= where it has one, and absent | |
| 205 | where it does not: a filesystem timestamp would say the day you cloned the repository. | |
| 206 | ||
| 207 | *Nothing is written until =site.base_url= is set.* A sitemap has nowhere to put a relative | |
| 208 | URL, so a zero-config build produces no sitemap rather than an invalid one. Set a base URL | |
| 209 | and it appears; set =sitemap = false= and it does not. | |
| 210 | ||
| 199 | 211 | ** Static files that live elsewhere |
| 200 | 212 | |
| 201 | 213 | A site's static files do not always sit where its writing does. weblorg publishes |
docs/guide/10-deploying.org +12
| @@ -113,6 +113,18 @@ Both zeros matter. Unresolved links are internal links pointing at nothing; diag | ||
| 113 | 113 | are malformed org that degraded rather than failing. With =--strict= neither can reach |
| 114 | 114 | this line, because either would have failed the build. |
| 115 | 115 | |
| 116 | * Telling a search engine where things are | |
| 117 | ||
| 118 | A build with =site.base_url= set writes =sitemap.xml= at the site root, listing every | |
| 119 | page. Point a =robots.txt= at it if you want one: | |
| 120 | ||
| 121 | #+BEGIN_EXAMPLE | |
| 122 | Sitemap: https://example.com/sitemap.xml | |
| 123 | #+END_EXAMPLE | |
| 124 | ||
| 125 | =robots.txt= is an ordinary file — put it beside your org files, or in a directory named | |
| 126 | by =[build] assets=, and it is copied through. | |
| 127 | ||
| 116 | 128 | * After an upgrade |
| 117 | 129 | |
| 118 | 130 | The first build on a new version is worth running with =--no-cache=, so you compare the |
src/config.rs +21 −1
| @@ -200,7 +200,7 @@ pub enum SortOrder { | ||
| 200 | 200 | Asc, |
| 201 | 201 | } |
| 202 | 202 | |
| 203 | #[derive(Debug, Clone, Default, PartialEq, Serialize, Deserialize)] | |
| 203 | #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] | |
| 204 | 204 | #[serde(default, deny_unknown_fields)] |
| 205 | 205 | pub struct Build { |
| 206 | 206 | /// Include pages marked `#+DRAFT:` in the build. |
| @@ -217,6 +217,23 @@ pub struct Build { | ||
| 217 | 217 | /// does: weblorg publishes `theme/static/` to `/`, and a repository migrating from it |
| 218 | 218 | /// should not have to move `robots.txt` next to its blog posts to keep the URL. |
| 219 | 219 | pub assets: Vec<Utf8PathBuf>, |
| 220 | /// Write `sitemap.xml` listing every published page. | |
| 221 | /// | |
| 222 | /// On, but a sitemap requires absolute URLs — the format has nowhere to put a | |
| 223 | /// relative one — so nothing is written until `site.base_url` is set. That is why a | |
| 224 | /// zero-config build produces no sitemap and no complaint: there is no URL to give a | |
| 225 | /// search engine yet. | |
| 226 | pub sitemap: bool, | |
| 227 | } | |
| 228 | ||
| 229 | impl Default for Build { | |
| 230 | fn default() -> Self { | |
| 231 | Build { | |
| 232 | drafts: false, | |
| 233 | assets: Vec::new(), | |
| 234 | sitemap: true, | |
| 235 | } | |
| 236 | } | |
| 220 | 237 | } |
| 221 | 238 | |
| 222 | 239 | #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] |
| @@ -585,6 +602,9 @@ drafts = false | ||
| 585 | 602 | # source directory. `assets = ["../theme/static"]` publishes that directory's contents at |
| 586 | 603 | # `/`, not at `/static/`. |
| 587 | 604 | assets = [] |
| 605 | # Write sitemap.xml. Needs site.base_url — a sitemap has nowhere to put a relative URL — | |
| 606 | # so nothing is written until you set one. | |
| 607 | sitemap = true | |
| 588 | 608 | |
| 589 | 609 | [html] |
| 590 | 610 | # How far to push heading levels down: a level-1 org heading becomes <h(1 + offset)>. |
src/site.rs +57
| @@ -893,6 +893,50 @@ fn render_page( | ||
| 893 | 893 | /// Site-root-relative name of the generated syntax stylesheet. Every page links to it. |
| 894 | 894 | pub const SYNTAX_STYLESHEET: &str = "syntax.css"; |
| 895 | 895 | |
| 896 | /// Site-root-relative name of the generated sitemap. | |
| 897 | pub const SITEMAP: &str = "sitemap.xml"; | |
| 898 | ||
| 899 | /// `sitemap.xml` for every HTML page in `pages`, in URL order. | |
| 900 | /// | |
| 901 | /// Only HTML: a sitemap is a list of pages for a crawler to read, and a feed or a | |
| 902 | /// stylesheet is neither. `lastmod` is the page's own `#+DATE:` where it has one — the | |
| 903 | /// nearest honest thing available without trusting a filesystem timestamp that a fresh | |
| 904 | /// clone would reset. | |
| 905 | fn sitemap(base_url: &str, pages: &[Utf8PathBuf], dated: &HashMap<&Utf8Path, &str>) -> String { | |
| 906 | let mut urls: Vec<&Utf8PathBuf> = pages | |
| 907 | .iter() | |
| 908 | .filter(|p| p.extension() == Some("html")) | |
| 909 | .collect(); | |
| 910 | urls.sort(); | |
| 911 | urls.dedup(); | |
| 912 | ||
| 913 | let base = base_url.trim_end_matches('/'); | |
| 914 | let mut out = String::from( | |
| 915 | "<?xml version=\"1.0\" encoding=\"UTF-8\"?>\n\ | |
| 916 | <urlset xmlns=\"http://www.sitemaps.org/schemas/sitemap/0.9\">\n", | |
| 917 | ); | |
| 918 | for url in urls { | |
| 919 | out.push_str("<url>\n"); | |
| 920 | out.push_str(&format!("<loc>{base}/{}</loc>\n", escape_xml(url.as_str()))); | |
| 921 | if let Some(date) = dated.get(url.as_path()) { | |
| 922 | out.push_str(&format!("<lastmod>{date}</lastmod>\n")); | |
| 923 | } | |
| 924 | out.push_str("</url>\n"); | |
| 925 | } | |
| 926 | out.push_str("</urlset>\n"); | |
| 927 | out | |
| 928 | } | |
| 929 | ||
| 930 | /// The five XML predefined entities. A `&` in a URL is the common one, from a query | |
| 931 | /// string that survived into a filename. | |
| 932 | fn escape_xml(s: &str) -> String { | |
| 933 | s.replace('&', "&") | |
| 934 | .replace('<', "<") | |
| 935 | .replace('>', ">") | |
| 936 | .replace('"', """) | |
| 937 | .replace('\'', "'") | |
| 938 | } | |
| 939 | ||
| 896 | 940 | /// Full site build with the incremental layer (spec §4). Renders only the pages whose |
| 897 | 941 | /// `render_key` changed or that link into a changed file's targets; reuses the on-disk |
| 898 | 942 | /// output of everything else; persists an updated cache manifest. |
| @@ -1123,6 +1167,19 @@ pub fn build_site(src: &Utf8Path, out: &Utf8Path, opts: &BuildOptions) -> Result | ||
| 1123 | 1167 | fs::write(out.join(SYNTAX_STYLESHEET), &syntax_css) |
| 1124 | 1168 | .with_context(|| format!("writing {SYNTAX_STYLESHEET} under {out}"))?; |
| 1125 | 1169 | |
| 1170 | // A sitemap covers every page the build emits, authored and generated alike, so it is | |
| 1171 | // written here rather than declared as a collection: a collection lists the pages it | |
| 1172 | // was pointed at, and this one has to know about all of them including itself. | |
| 1173 | if cfg.build.sitemap && !cfg.site.base_url.is_empty() { | |
| 1174 | let dated: HashMap<&Utf8Path, &str> = preps | |
| 1175 | .iter() | |
| 1176 | .filter_map(|p| Some((p.output.as_path(), p.context.date_iso.as_deref()?))) | |
| 1177 | .collect(); | |
| 1178 | let xml = sitemap(&cfg.site.base_url, &report.pages, &dated); | |
| 1179 | fs::write(out.join(SITEMAP), xml) | |
| 1180 | .with_context(|| format!("writing {SITEMAP} under {out}"))?; | |
| 1181 | } | |
| 1182 | ||
| 1126 | 1183 | // Assets are a dumb copy in v0.3 (spec §8 Q11): copy every run. Cheap, and keeps the |
| 1127 | 1184 | // full-vs-incremental byte equivalence trivially true for non-`.org` files. |
| 1128 | 1185 | for asset in &assets { |
tests/config.rs +85
| @@ -2378,3 +2378,88 @@ fn entries_carry_no_content_unless_asked() { | ||
| 2378 | 2378 | let html = page(&out, "blog/index.html"); |
| 2379 | 2379 | assert!(!html.contains("false"), "no entry carries a body:\n{html}"); |
| 2380 | 2380 | } |
| 2381 | ||
| 2382 | // --------------------------------------------------------------------------- | |
| 2383 | // Sitemap | |
| 2384 | // --------------------------------------------------------------------------- | |
| 2385 | ||
| 2386 | fn sitemap_site(src: &Utf8PathBuf, config: &str) { | |
| 2387 | std::fs::create_dir_all(src.join("blog")).unwrap(); | |
| 2388 | std::fs::create_dir_all(src.join("templates")).unwrap(); | |
| 2389 | std::fs::write(src.join("index.org"), "#+TITLE: Home\n\nWelcome.\n").unwrap(); | |
| 2390 | std::fs::write( | |
| 2391 | src.join("blog/post.org"), | |
| 2392 | "#+TITLE: Post\n#+DATE: <2026-01-15 Thu>\n\nBody.\n", | |
| 2393 | ) | |
| 2394 | .unwrap(); | |
| 2395 | std::fs::write(src.join("blog/undated.org"), "#+TITLE: Undated\n\nBody.\n").unwrap(); | |
| 2396 | std::fs::write(src.join("style.css"), "body{}").unwrap(); | |
| 2397 | std::fs::write( | |
| 2398 | src.join("templates/list.html"), | |
| 2399 | "<html><body>{% for p in pages %}<li>{{ p.title }}</li>{% endfor %}</body></html>", | |
| 2400 | ) | |
| 2401 | .unwrap(); | |
| 2402 | std::fs::write(src.join("orgo.toml"), config).unwrap(); | |
| 2403 | } | |
| 2404 | ||
| 2405 | /// A sitemap covers every page the build emits, generated ones included — a crawler has no | |
| 2406 | /// other way to learn that `/blog/` exists. | |
| 2407 | #[test] | |
| 2408 | fn a_sitemap_lists_every_page_including_generated_ones() { | |
| 2409 | let root = tmpdir("sitemap"); | |
| 2410 | let src = root.join("src"); | |
| 2411 | std::fs::create_dir_all(&src).unwrap(); | |
| 2412 | sitemap_site( | |
| 2413 | &src, | |
| 2414 | "[site]\nbase_url = \"https://example.com\"\n\n\ | |
| 2415 | [[collections]]\nsource = \"blog\"\noutput = \"blog/index.html\"\n\ | |
| 2416 | template = \"list.html\"\ntitle = \"Blog\"\n", | |
| 2417 | ); | |
| 2418 | let out = root.join("out"); | |
| 2419 | build(&src, &out); | |
| 2420 | ||
| 2421 | let xml = page(&out, "sitemap.xml"); | |
| 2422 | for url in [ | |
| 2423 | "https://example.com/index.html", | |
| 2424 | "https://example.com/blog/post.html", | |
| 2425 | "https://example.com/blog/index.html", | |
| 2426 | ] { | |
| 2427 | assert!(xml.contains(url), "{url} is in the sitemap:\n{xml}"); | |
| 2428 | } | |
| 2429 | // A date the author wrote is the only honest `lastmod` available; a page without one | |
| 2430 | // gets no element rather than a filesystem timestamp a fresh clone would reset. | |
| 2431 | assert!(xml.contains("<lastmod>2026-01-15</lastmod>"), "{xml}"); | |
| 2432 | assert_eq!(xml.matches("<lastmod>").count(), 1, "only the dated page:\n{xml}"); | |
| 2433 | // Assets and the stylesheet are not pages. | |
| 2434 | assert!(!xml.contains("style.css") && !xml.contains("syntax.css"), "{xml}"); | |
| 2435 | } | |
| 2436 | ||
| 2437 | /// A sitemap has nowhere to put a relative URL, so without a base URL there is nothing | |
| 2438 | /// honest to write — and a build with no `base_url` set is the zero-config default. | |
| 2439 | #[test] | |
| 2440 | fn no_base_url_means_no_sitemap() { | |
| 2441 | let root = tmpdir("sitemapnobase"); | |
| 2442 | let src = root.join("src"); | |
| 2443 | std::fs::create_dir_all(&src).unwrap(); | |
| 2444 | sitemap_site(&src, ""); | |
| 2445 | let out = root.join("out"); | |
| 2446 | build(&src, &out); | |
| 2447 | ||
| 2448 | assert!(!out.join("sitemap.xml").exists(), "no base_url, no sitemap"); | |
| 2449 | } | |
| 2450 | ||
| 2451 | /// And it can be turned off outright. | |
| 2452 | #[test] | |
| 2453 | fn the_sitemap_can_be_disabled() { | |
| 2454 | let root = tmpdir("sitemapoff"); | |
| 2455 | let src = root.join("src"); | |
| 2456 | std::fs::create_dir_all(&src).unwrap(); | |
| 2457 | sitemap_site( | |
| 2458 | &src, | |
| 2459 | "[site]\nbase_url = \"https://example.com\"\n\n[build]\nsitemap = false\n", | |
| 2460 | ); | |
| 2461 | let out = root.join("out"); | |
| 2462 | build(&src, &out); | |
| 2463 | ||
| 2464 | assert!(!out.join("sitemap.xml").exists(), "disabled means absent"); | |
| 2465 | } | |