Commit 20cb94bb4e
Verified · cmc
Layout: unified · split
Cargo.lock +1 −1
| @@ -657,7 +657,7 @@ dependencies = [ | |||
| 657 | 657 | ||
| 658 | [[package]] | 658 | [[package]] |
| 659 | name = "org-ssg" | 659 | name = "org-ssg" |
| 660 | version = "0.11.0" | 660 | version = "0.12.0" |
| 661 | dependencies = [ | 661 | dependencies = [ |
| 662 | "anyhow", | 662 | "anyhow", |
| 663 | "blake3", | 663 | "blake3", |
Cargo.toml +1 −1
| @@ -1,6 +1,6 @@ | |||
| 1 | [package] | 1 | [package] |
| 2 | name = "org-ssg" | 2 | name = "org-ssg" |
| 3 | version = "0.11.0" | 3 | version = "0.12.0" |
| 4 | edition = "2021" | 4 | edition = "2021" |
| 5 | description = "Org-mode static site generator that renders the org element tree straight to HTML" | 5 | description = "Org-mode static site generator that renders the org element tree straight to HTML" |
| 6 | license = "MIT" | 6 | license = "MIT" |
README.md +22 −2
| @@ -72,7 +72,7 @@ receive: | |||
| 72 | | Variable | What it is | | 72 | | Variable | What it is | |
| 73 | |---|---| | 73 | |---|---| |
| 74 | | `body` | the rendered page HTML — use `{{ body \| safe }}` | | 74 | | `body` | the rendered page HTML — use `{{ body \| safe }}` | |
| 75 | | `page` | `.title`, `.url`, `.source`, `.date`, `.tags`, `.keywords` | | 75 | | `page` | `.title`, `.url`, `.source`, `.date`, `.date_iso`, `.tags`, `.excerpt`, `.word_count`, `.reading_time`, `.keywords` | |
| 76 | | `site` | `.title`, `.base_url`, `.description`, `.language` | | 76 | | `site` | `.title`, `.base_url`, `.description`, `.language` | |
| 77 | | `nav` | list of `{title, url}`, relative to this page | | 77 | | `nav` | list of `{title, url}`, relative to this page | |
| 78 | | `root` | `../`-prefix back to the site root from this page | | 78 | | `root` | `../`-prefix back to the site root from this page | |
| @@ -218,6 +218,7 @@ broken. Set `site.base_url` and use the `absolute` filter: | |||
| 218 | |---|---| | 218 | |---|---| |
| 219 | | `absolute` | site-root-relative path → absolute URL; already-absolute URLs pass through | | 219 | | `absolute` | site-root-relative path → absolute URL; already-absolute URLs pass through | |
| 220 | | `rfc822` | any org or ISO date → the format RSS `pubDate` requires | | 220 | | `rfc822` | any org or ISO date → the format RSS `pubDate` requires | |
| 221 | | `truncate(n)` | shorten to at most `n` characters on a word boundary, with an ellipsis | | ||
| 221 | 222 | ||
| 222 | Apply `absolute` to the site-root-relative values — `page.url`, `pages[].url`, | 223 | Apply `absolute` to the site-root-relative values — `page.url`, `pages[].url`, |
| 223 | `group.url` — and not to `nav[].url`, `paginator.*_url`, `stylesheet` or `root`, which | 224 | `group.url` — and not to `nav[].url`, `paginator.*_url`, `stylesheet` or `root`, which |
| @@ -230,6 +231,24 @@ The default layout also emits `<link rel="canonical">` when a base URL is set. | |||
| 230 | Listing pages are cached on the entries they list, so adding a post re-renders that | 231 | Listing pages are cached on the entries they list, so adding a post re-renders that |
| 231 | section's index and nothing else. | 232 | section's index and nothing else. |
| 232 | 233 | ||
| 234 | ### Excerpts and drafts | ||
| 235 | |||
| 236 | `page.excerpt` is a page's `#+DESCRIPTION:` when it sets one and its first paragraph | ||
| 237 | otherwise, so a listing has something to show whether or not the author thought about | ||
| 238 | summaries. `page.word_count` and `page.reading_time` (minutes at 200 wpm) count prose | ||
| 239 | only — a post that is mostly a shell transcript should not read as an hour's work. | ||
| 240 | `truncate` exists because an excerpt is usually a whole paragraph and minijinja has no | ||
| 241 | such filter. | ||
| 242 | |||
| 243 | `#+DRAFT:` keeps a page out of the build entirely — no page, and absent from listings and | ||
| 244 | the nav rather than merely unlinked. `--drafts` includes them, which is what you want | ||
| 245 | under `watch` while writing one. A draft is out of the symbol table too, so a link *to* | ||
| 246 | one is reported as the dead link it would be once published. | ||
| 247 | |||
| 248 | The keyword is read forgivingly: `t`, `yes`, `1` and a bare `#+DRAFT:` all mean draft, | ||
| 249 | because writing the keyword at all is the signal. Only an explicit `nil`, `false`, `no`, | ||
| 250 | `0` or `off` means published. | ||
| 251 | |||
| 233 | ### `#+SLUG:` | 252 | ### `#+SLUG:` |
| 234 | 253 | ||
| 235 | A page's output filename comes from its `#+SLUG:` when it has one, so | 254 | A page's output filename comes from its `#+SLUG:` when it has one, so |
| @@ -303,6 +322,7 @@ all-of-org. Phase 0 checked this line against a real 179-file corpus and found i | |||
| 303 | | **11** | **Pagination: numbered pages with a `paginator` context, composing with grouping** | **done** | | 322 | | **11** | **Pagination: numbered pages with a `paginator` context, composing with grouping** | **done** | |
| 304 | | **12** | **`base_url`: `absolute`/`rfc822` filters, a valid RSS feed in the scaffold, canonical links** | **done** | | 323 | | **12** | **`base_url`: `absolute`/`rfc822` filters, a valid RSS feed in the scaffold, canonical links** | **done** | |
| 305 | | **13** | **`watch` on OS filesystem events, debounced, with the feedback loop closed** | **done** | | 324 | | **13** | **`watch` on OS filesystem events, debounced, with the feedback loop closed** | **done** | |
| 325 | | **14** | **Authoring: excerpts, word count, reading time, `truncate`, and draft pages** | **done** | | ||
| 306 | 326 | ||
| 307 | ### v0.2 in / out | 327 | ### v0.2 in / out |
| 308 | 328 | ||
| @@ -590,7 +610,7 @@ PARSE/RESOLVE/RENDER), `notify` (filesystem events for `watch`), `toml` (config) | |||
| 590 | 610 | ||
| 591 | ``` | 611 | ``` |
| 592 | cargo build | 612 | cargo build |
| 593 | cargo test # 128 tests | 613 | cargo test # 135 tests |
| 594 | cargo run -- init my-site # scaffold a new site | 614 | cargo run -- init my-site # scaffold a new site |
| 595 | cargo run -- build fixtures/minimal.org -o minimal.html # single file | 615 | cargo run -- build fixtures/minimal.org -o minimal.html # single file |
| 596 | cargo run -- build fixtures/site -o _site # whole site (incremental) | 616 | cargo run -- build fixtures/site -o _site # whole site (incremental) |
src/config.rs +17
| @@ -30,6 +30,7 @@ pub struct Config { | |||
| 30 | pub templates: Templates, | 30 | pub templates: Templates, |
| 31 | pub highlight: Highlight, | 31 | pub highlight: Highlight, |
| 32 | pub html: HtmlOutput, | 32 | pub html: HtmlOutput, |
| 33 | pub build: Build, | ||
| 33 | /// Generated listing pages. Each produces one output file that has no source `.org` | 34 | /// Generated listing pages. Each produces one output file that has no source `.org` |
| 34 | /// file behind it — a blog index, an archive, a feed. | 35 | /// file behind it — a blog index, an archive, a feed. |
| 35 | pub collections: Vec<Collection>, | 36 | pub collections: Vec<Collection>, |
| @@ -130,6 +131,17 @@ pub enum SortOrder { | |||
| 130 | Asc, | 131 | Asc, |
| 131 | } | 132 | } |
| 132 | 133 | ||
| 134 | #[derive(Debug, Clone, Default, PartialEq, Serialize, Deserialize)] | ||
| 135 | #[serde(default, deny_unknown_fields)] | ||
| 136 | pub struct Build { | ||
| 137 | /// Include pages marked `#+DRAFT:` in the build. | ||
| 138 | /// | ||
| 139 | /// Off by default, because the point of marking something a draft is that it is not | ||
| 140 | /// ready to be read. `--drafts` turns it on for a session, which is what you want | ||
| 141 | /// under `watch` while writing one. | ||
| 142 | pub drafts: bool, | ||
| 143 | } | ||
| 144 | |||
| 133 | #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] | 145 | #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] |
| 134 | #[serde(default, deny_unknown_fields)] | 146 | #[serde(default, deny_unknown_fields)] |
| 135 | pub struct HtmlOutput { | 147 | pub struct HtmlOutput { |
| @@ -400,6 +412,11 @@ expose_page_list = false | |||
| 400 | # base16-eighties.dark, base16-mocha.dark, base16-ocean.light. | 412 | # base16-eighties.dark, base16-mocha.dark, base16-ocean.light. |
| 401 | theme = "InspiredGitHub" | 413 | theme = "InspiredGitHub" |
| 402 | 414 | ||
| 415 | [build] | ||
| 416 | # Include pages marked `#+DRAFT:`. Off by default — the point of marking a draft is that | ||
| 417 | # it is not ready to be read. `--drafts` turns it on for one run, handy under `watch`. | ||
| 418 | drafts = false | ||
| 419 | |||
| 403 | [html] | 420 | [html] |
| 404 | # How far to push heading levels down: a level-1 org heading becomes <h(1 + offset)>. | 421 | # How far to push heading levels down: a level-1 org heading becomes <h(1 + offset)>. |
| 405 | # The default of 1 matches Emacs, and assumes your layout renders the page title as the | 422 | # The default of 1 matches Emacs, and assumes your layout renders the page title as the |
src/main.rs +13
| @@ -39,6 +39,9 @@ enum Command { | |||
| 39 | /// Config file to use, overriding `org-ssg.toml` in the source directory. | 39 | /// Config file to use, overriding `org-ssg.toml` in the source directory. |
| 40 | #[arg(long, value_name = "FILE")] | 40 | #[arg(long, value_name = "FILE")] |
| 41 | config: Option<Utf8PathBuf>, | 41 | config: Option<Utf8PathBuf>, |
| 42 | /// Include pages marked `#+DRAFT:`. | ||
| 43 | #[arg(long)] | ||
| 44 | drafts: bool, | ||
| 42 | }, | 45 | }, |
| 43 | /// Watch a source directory and rebuild incrementally on change, driven by OS | 46 | /// Watch a source directory and rebuild incrementally on change, driven by OS |
| 44 | /// filesystem events. | 47 | /// filesystem events. |
| @@ -57,6 +60,9 @@ enum Command { | |||
| 57 | /// Config file to use, overriding `org-ssg.toml` in the source directory. | 60 | /// Config file to use, overriding `org-ssg.toml` in the source directory. |
| 58 | #[arg(long, value_name = "FILE")] | 61 | #[arg(long, value_name = "FILE")] |
| 59 | config: Option<Utf8PathBuf>, | 62 | config: Option<Utf8PathBuf>, |
| 63 | /// Include pages marked `#+DRAFT:`. Handy while writing one. | ||
| 64 | #[arg(long)] | ||
| 65 | drafts: bool, | ||
| 60 | }, | 66 | }, |
| 61 | /// Remove the build output directory (which holds the cache manifest). | 67 | /// Remove the build output directory (which holds the cache manifest). |
| 62 | Clean { | 68 | Clean { |
| @@ -87,6 +93,7 @@ fn main() -> Result<()> { | |||
| 87 | no_cache, | 93 | no_cache, |
| 88 | strict, | 94 | strict, |
| 89 | config, | 95 | config, |
| 96 | drafts, | ||
| 90 | } => { | 97 | } => { |
| 91 | if input.is_dir() { | 98 | if input.is_dir() { |
| 92 | let out = output | 99 | let out = output |
| @@ -95,6 +102,7 @@ fn main() -> Result<()> { | |||
| 95 | no_cache, | 102 | no_cache, |
| 96 | strict, | 103 | strict, |
| 97 | config_path: config.clone(), | 104 | config_path: config.clone(), |
| 105 | drafts, | ||
| 98 | }; | 106 | }; |
| 99 | let report = build_site(&input, &out, &opts)?; | 107 | let report = build_site(&input, &out, &opts)?; |
| 100 | println!( | 108 | println!( |
| @@ -121,6 +129,7 @@ fn main() -> Result<()> { | |||
| 121 | no_cache, | 129 | no_cache, |
| 122 | strict, | 130 | strict, |
| 123 | config, | 131 | config, |
| 132 | drafts, | ||
| 124 | } => org_ssg::watch::run( | 133 | } => org_ssg::watch::run( |
| 125 | &input, | 134 | &input, |
| 126 | &output, | 135 | &output, |
| @@ -128,6 +137,7 @@ fn main() -> Result<()> { | |||
| 128 | no_cache, | 137 | no_cache, |
| 129 | strict, | 138 | strict, |
| 130 | config_path: config, | 139 | config_path: config, |
| 140 | drafts, | ||
| 131 | }, | 141 | }, |
| 132 | ), | 142 | ), |
| 133 | Command::Audit { input } => { | 143 | Command::Audit { input } => { |
| @@ -266,6 +276,9 @@ fn build_file(input: &Utf8Path, output: &Utf8Path) -> Result<()> { | |||
| 266 | date: None, | 276 | date: None, |
| 267 | date_iso: None, | 277 | date_iso: None, |
| 268 | tags: Vec::new(), | 278 | tags: Vec::new(), |
| 279 | excerpt: String::new(), | ||
| 280 | word_count: 0, | ||
| 281 | reading_time: 0, | ||
| 269 | keywords: Default::default(), | 282 | keywords: Default::default(), |
| 270 | }; | 283 | }; |
| 271 | let mut ctx = RenderContext::new(&site, &page_ctx, &[], SYNTAX_STYLESHEET, ""); | 284 | let mut ctx = RenderContext::new(&site, &page_ctx, &[], SYNTAX_STYLESHEET, ""); |
src/site.rs +32 −2
| @@ -32,7 +32,14 @@ use crate::template::{ | |||
| 32 | GroupContext, NavItem, PageContext, Paginator, PaginatorPage, RenderContext, SiteContext, | 32 | GroupContext, NavItem, PageContext, Paginator, PaginatorPage, RenderContext, SiteContext, |
| 33 | Templater, | 33 | Templater, |
| 34 | }; | 34 | }; |
| 35 | use crate::util::{iso_date, output_path, output_url, relative_root, slugify}; | 35 | use crate::util::{ |
| 36 | document_text, first_paragraph, is_draft, iso_date, output_path, output_url, relative_root, | ||
| 37 | slugify, | ||
| 38 | }; | ||
| 39 | |||
| 40 | /// Reading speed for [`PageContext::reading_time`]. 200 wpm is the conventional figure | ||
| 41 | /// for prose on screen. | ||
| 42 | const WORDS_PER_MINUTE: usize = 200; | ||
| 36 | 43 | ||
| 37 | /// A fully built page: source and output paths (relative to their roots) and its | 44 | /// A fully built page: source and output paths (relative to their roots) and its |
| 38 | /// final templated HTML. | 45 | /// final templated HTML. |
| @@ -56,6 +63,8 @@ pub struct BuildOptions { | |||
| 56 | pub strict: bool, | 63 | pub strict: bool, |
| 57 | /// Explicit config file, overriding `org-ssg.toml` in the source directory. | 64 | /// Explicit config file, overriding `org-ssg.toml` in the source directory. |
| 58 | pub config_path: Option<Utf8PathBuf>, | 65 | pub config_path: Option<Utf8PathBuf>, |
| 66 | /// Include pages marked `#+DRAFT:`, overriding `build.drafts` when set. | ||
| 67 | pub drafts: bool, | ||
| 59 | } | 68 | } |
| 60 | 69 | ||
| 61 | /// Summary of a site build. | 70 | /// Summary of a site build. |
| @@ -468,6 +477,9 @@ fn listing_context(listing: &Listing) -> PageContext { | |||
| 468 | date: None, | 477 | date: None, |
| 469 | date_iso: None, | 478 | date_iso: None, |
| 470 | tags: Vec::new(), | 479 | tags: Vec::new(), |
| 480 | excerpt: String::new(), | ||
| 481 | word_count: 0, | ||
| 482 | reading_time: 0, | ||
| 471 | keywords: Default::default(), | 483 | keywords: Default::default(), |
| 472 | } | 484 | } |
| 473 | } | 485 | } |
| @@ -517,6 +529,15 @@ fn prepare_pages( | |||
| 517 | }) | 529 | }) |
| 518 | .collect::<Result<Vec<_>>>()?; | 530 | .collect::<Result<Vec<_>>>()?; |
| 519 | 531 | ||
| 532 | // Drop drafts before anything else sees them. Removing them here rather than at emit | ||
| 533 | // time means they are absent from listings, the nav and the symbol table too — so a | ||
| 534 | // link *to* a draft is reported as broken, which is exactly what it would be on the | ||
| 535 | // published site. | ||
| 536 | let docs: Vec<Document> = docs | ||
| 537 | .into_iter() | ||
| 538 | .filter(|d| config.build.drafts || !is_draft(&d.keywords)) | ||
| 539 | .collect(); | ||
| 540 | |||
| 520 | // INDEX: collect every link target across the corpus. | 541 | // INDEX: collect every link target across the corpus. |
| 521 | let mut symbols = SymbolTable::new(); | 542 | let mut symbols = SymbolTable::new(); |
| 522 | for doc in &docs { | 543 | for doc in &docs { |
| @@ -692,11 +713,13 @@ pub const SYNTAX_STYLESHEET: &str = "syntax.css"; | |||
| 692 | /// `render_key` changed or that link into a changed file's targets; reuses the on-disk | 713 | /// `render_key` changed or that link into a changed file's targets; reuses the on-disk |
| 693 | /// output of everything else; persists an updated cache manifest. | 714 | /// output of everything else; persists an updated cache manifest. |
| 694 | pub fn build_site(src: &Utf8Path, out: &Utf8Path, opts: &BuildOptions) -> Result<SiteReport> { | 715 | pub fn build_site(src: &Utf8Path, out: &Utf8Path, opts: &BuildOptions) -> Result<SiteReport> { |
| 695 | let cfg = match &opts.config_path { | 716 | let mut cfg = match &opts.config_path { |
| 696 | Some(path) => Config::load_file(path)?, | 717 | Some(path) => Config::load_file(path)?, |
| 697 | None => Config::load(src)?, | 718 | None => Config::load(src)?, |
| 698 | }; | 719 | }; |
| 699 | cfg.validate()?; | 720 | cfg.validate()?; |
| 721 | // The flag turns drafts on; it never turns off a config that asked for them. | ||
| 722 | cfg.build.drafts |= opts.drafts; | ||
| 700 | 723 | ||
| 701 | // Create the output directory up front so it can be recognised and excluded when it | 724 | // Create the output directory up front so it can be recognised and excluded when it |
| 702 | // lives inside the source tree. | 725 | // lives inside the source tree. |
| @@ -1141,6 +1164,7 @@ fn is_top_level(output: &Utf8Path) -> bool { | |||
| 1141 | /// under its lowercased name, so a template can use metadata this crate has never heard | 1164 | /// under its lowercased name, so a template can use metadata this crate has never heard |
| 1142 | /// of without the crate needing a release to support it. | 1165 | /// of without the crate needing a release to support it. |
| 1143 | fn page_context(doc: &Document, output: &Utf8Path) -> PageContext { | 1166 | fn page_context(doc: &Document, output: &Utf8Path) -> PageContext { |
| 1167 | let words = document_text(&doc.root).split_whitespace().count(); | ||
| 1144 | let keyword = |name: &str| { | 1168 | let keyword = |name: &str| { |
| 1145 | doc.keywords | 1169 | doc.keywords |
| 1146 | .entries | 1170 | .entries |
| @@ -1154,6 +1178,12 @@ fn page_context(doc: &Document, output: &Utf8Path) -> PageContext { | |||
| 1154 | source: doc.source_path.to_string(), | 1178 | source: doc.source_path.to_string(), |
| 1155 | date_iso: keyword("DATE").as_deref().and_then(iso_date), | 1179 | date_iso: keyword("DATE").as_deref().and_then(iso_date), |
| 1156 | date: keyword("DATE"), | 1180 | date: keyword("DATE"), |
| 1181 | excerpt: keyword("DESCRIPTION") | ||
| 1182 | .filter(|d| !d.trim().is_empty()) | ||
| 1183 | .or_else(|| first_paragraph(&doc.root)) | ||
| 1184 | .unwrap_or_default(), | ||
| 1185 | word_count: words, | ||
| 1186 | reading_time: words.div_ceil(WORDS_PER_MINUTE).max(usize::from(words > 0)), | ||
| 1157 | tags: keyword("FILETAGS") | 1187 | tags: keyword("FILETAGS") |
| 1158 | .unwrap_or_default() | 1188 | .unwrap_or_default() |
| 1159 | .split(':') | 1189 | .split(':') |
src/template.rs +31
| @@ -52,6 +52,14 @@ pub struct PageContext { | |||
| 52 | pub date_iso: Option<String>, | 52 | pub date_iso: Option<String>, |
| 53 | /// `#+FILETAGS:` split on `:`. | 53 | /// `#+FILETAGS:` split on `:`. |
| 54 | pub tags: Vec<String>, | 54 | pub tags: Vec<String>, |
| 55 | /// A short summary for listings: `#+DESCRIPTION:` when the page sets one, otherwise | ||
| 56 | /// its first paragraph. Empty only when the page has neither. | ||
| 57 | pub excerpt: String, | ||
| 58 | /// Words of prose, excluding code and example blocks. | ||
| 59 | pub word_count: usize, | ||
| 60 | /// Minutes to read at 200 words per minute, rounded up; at least 1 for a page with | ||
| 61 | /// any prose at all. | ||
| 62 | pub reading_time: usize, | ||
| 55 | /// Every `#+KEYWORD:` in the file, keyed by lowercased name, so a template can use | 63 | /// Every `#+KEYWORD:` in the file, keyed by lowercased name, so a template can use |
| 56 | /// project-specific metadata this crate has never heard of. | 64 | /// project-specific metadata this crate has never heard of. |
| 57 | pub keywords: BTreeMap<String, String>, | 65 | pub keywords: BTreeMap<String, String>, |
| @@ -375,6 +383,10 @@ pub const STARTER_LIST_TEMPLATE: &str = r#"<!DOCTYPE html> | |||
| 375 | <li> | 383 | <li> |
| 376 | {%- if post.date_iso %}<time datetime="{{ post.date_iso }}">{{ post.date_iso }}</time> {% endif %} | 384 | {%- if post.date_iso %}<time datetime="{{ post.date_iso }}">{{ post.date_iso }}</time> {% endif %} |
| 377 | <a href="{{ root }}{{ post.url }}">{{ post.title }}</a> | 385 | <a href="{{ root }}{{ post.url }}">{{ post.title }}</a> |
| 386 | {%- if post.excerpt %} | ||
| 387 | <p class="excerpt">{{ post.excerpt | truncate(180) }}</p> | ||
| 388 | {%- endif %} | ||
| 389 | <span class="reading-time">{{ post.reading_time }} min read</span> | ||
| 378 | </li> | 390 | </li> |
| 379 | {%- endfor %} | 391 | {%- endfor %} |
| 380 | </ul> | 392 | </ul> |
| @@ -427,6 +439,25 @@ fn add_filters(env: &mut Environment<'static>, base_url: &str) { | |||
| 427 | }, | 439 | }, |
| 428 | ); | 440 | ); |
| 429 | 441 | ||
| 442 | // `truncate`: shorten to at most N characters, on a word boundary, with an ellipsis. | ||
| 443 | // | ||
| 444 | // minijinja ships no truncate, and an excerpt is usually a whole first paragraph — | ||
| 445 | // so without this the only options in a listing are the full paragraph or nothing. | ||
| 446 | env.add_filter( | ||
| 447 | "truncate", | ||
| 448 | |text: &str, limit: Option<usize>| -> String { | ||
| 449 | let limit = limit.unwrap_or(160); | ||
| 450 | if text.chars().count() <= limit { | ||
| 451 | return text.to_string(); | ||
| 452 | } | ||
| 453 | let head: String = text.chars().take(limit).collect(); | ||
| 454 | // Cut at the last space so a word is never sliced in half; if there is no | ||
| 455 | // space at all, the hard cut is the only option. | ||
| 456 | let cut = head.rfind(char::is_whitespace).unwrap_or(head.len()); | ||
| 457 | format!("{}…", head[..cut].trim_end()) | ||
| 458 | }, | ||
| 459 | ); | ||
| 460 | |||
| 430 | // `rfc822`: an org or ISO date → the format RSS `pubDate` requires. | 461 | // `rfc822`: an org or ISO date → the format RSS `pubDate` requires. |
| 431 | env.add_filter("rfc822", |raw: &str| -> Result<String, minijinja::Error> { | 462 | env.add_filter("rfc822", |raw: &str| -> Result<String, minijinja::Error> { |
| 432 | let iso = crate::util::iso_date(raw).ok_or_else(|| { | 463 | let iso = crate::util::iso_date(raw).ok_or_else(|| { |
src/util.rs +100 −1
| @@ -4,7 +4,7 @@ | |||
| 4 | 4 | ||
| 5 | use camino::{Utf8Path, Utf8PathBuf}; | 5 | use camino::{Utf8Path, Utf8PathBuf}; |
| 6 | 6 | ||
| 7 | use crate::model::{Keywords, Object}; | 7 | use crate::model::{Element, Keywords, Object, Section, TableRow}; |
| 8 | 8 | ||
| 9 | /// The output path for a document, relative to the site root. | 9 | /// The output path for a document, relative to the site root. |
| 10 | /// | 10 | /// |
| @@ -76,6 +76,105 @@ fn plain_text_into(objs: &[Object], out: &mut String) { | |||
| 76 | } | 76 | } |
| 77 | } | 77 | } |
| 78 | 78 | ||
| 79 | /// Is this document marked as a draft? | ||
| 80 | /// | ||
| 81 | /// `#+DRAFT:` counts as true by its mere presence — writing the keyword at all is the | ||
| 82 | /// signal — unless the value explicitly says otherwise. Someone who types `#+DRAFT: t`, | ||
| 83 | /// `#+DRAFT: yes` or a bare `#+DRAFT:` means the same thing, and publishing an unfinished | ||
| 84 | /// post because the value was not the expected spelling is the wrong way to be strict. | ||
| 85 | pub fn is_draft(keywords: &Keywords) -> bool { | ||
| 86 | keywords | ||
| 87 | .entries | ||
| 88 | .iter() | ||
| 89 | .find(|(k, _)| k.eq_ignore_ascii_case("DRAFT")) | ||
| 90 | .map(|(_, v)| { | ||
| 91 | !matches!( | ||
| 92 | v.trim().to_ascii_lowercase().as_str(), | ||
| 93 | "nil" | "false" | "no" | "0" | "off" | ||
| 94 | ) | ||
| 95 | }) | ||
| 96 | .unwrap_or(false) | ||
| 97 | } | ||
| 98 | |||
| 99 | /// The document's prose as plain text, for word counts and excerpts. | ||
| 100 | /// | ||
| 101 | /// Source and example blocks are excluded on purpose: a reading-time estimate over a | ||
| 102 | /// post that is mostly a shell transcript should describe the prose someone reads, not | ||
| 103 | /// the code they skim. Headings are included — they are read. | ||
| 104 | pub fn document_text(root: &Section) -> String { | ||
| 105 | let mut out = String::new(); | ||
| 106 | section_text(root, &mut out); | ||
| 107 | out | ||
| 108 | } | ||
| 109 | |||
| 110 | fn section_text(section: &Section, out: &mut String) { | ||
| 111 | if let Some(heading) = §ion.heading { | ||
| 112 | push_words(&plain_text(&heading.title), out); | ||
| 113 | } | ||
| 114 | elements_text(§ion.content, out); | ||
| 115 | for child in §ion.children { | ||
| 116 | section_text(child, out); | ||
| 117 | } | ||
| 118 | } | ||
| 119 | |||
| 120 | fn elements_text(elements: &[Element], out: &mut String) { | ||
| 121 | for element in elements { | ||
| 122 | match element { | ||
| 123 | Element::Paragraph(objs) => push_words(&plain_text(objs), out), | ||
| 124 | Element::List(list) => { | ||
| 125 | for item in &list.items { | ||
| 126 | if let Some(term) = &item.term { | ||
| 127 | push_words(&plain_text(term), out); | ||
| 128 | } | ||
| 129 | elements_text(&item.content, out); | ||
| 130 | } | ||
| 131 | } | ||
| 132 | Element::Table(table) => { | ||
| 133 | for row in &table.rows { | ||
| 134 | if let TableRow::Cells(cells) = row { | ||
| 135 | for cell in cells { | ||
| 136 | push_words(&plain_text(cell), out); | ||
| 137 | } | ||
| 138 | } | ||
| 139 | } | ||
| 140 | } | ||
| 141 | Element::QuoteBlock(inner) | Element::CenterBlock(inner) => elements_text(inner, out), | ||
| 142 | Element::Figure { caption, .. } => push_words(&plain_text(caption), out), | ||
| 143 | Element::FootnoteDefinition { content, .. } => elements_text(content, out), | ||
| 144 | // Code, drawers, comments, keywords and raw export blocks are not prose. | ||
| 145 | _ => {} | ||
| 146 | } | ||
| 147 | } | ||
| 148 | } | ||
| 149 | |||
| 150 | fn push_words(text: &str, out: &mut String) { | ||
| 151 | let text = text.trim(); | ||
| 152 | if text.is_empty() { | ||
| 153 | return; | ||
| 154 | } | ||
| 155 | if !out.is_empty() { | ||
| 156 | out.push(' '); | ||
| 157 | } | ||
| 158 | out.push_str(text); | ||
| 159 | } | ||
| 160 | |||
| 161 | /// The document's first paragraph as plain text — the fallback excerpt for a page with | ||
| 162 | /// no `#+DESCRIPTION:`. | ||
| 163 | pub fn first_paragraph(root: &Section) -> Option<String> { | ||
| 164 | fn find(section: &Section) -> Option<String> { | ||
| 165 | for element in §ion.content { | ||
| 166 | if let Element::Paragraph(objs) = element { | ||
| 167 | let text = plain_text(objs); | ||
| 168 | if !text.trim().is_empty() { | ||
| 169 | return Some(text.trim().to_string()); | ||
| 170 | } | ||
| 171 | } | ||
| 172 | } | ||
| 173 | section.children.iter().find_map(find) | ||
| 174 | } | ||
| 175 | find(root) | ||
| 176 | } | ||
| 177 | |||
| 79 | /// Turn heading text into a URL-safe anchor slug. | 178 | /// Turn heading text into a URL-safe anchor slug. |
| 80 | pub fn slugify(text: &str) -> String { | 179 | pub fn slugify(text: &str) -> String { |
| 81 | let mut out = String::new(); | 180 | let mut out = String::new(); |
tests/config.rs +221
| @@ -1469,3 +1469,224 @@ fn changing_base_url_re_renders_the_site() { | |||
| 1469 | assert_eq!(report.rendered.len(), 3, "every page carries the base URL"); | 1469 | assert_eq!(report.rendered.len(), 3, "every page carries the base URL"); |
| 1470 | assert!(page(&out, "index.html").contains("https://moved.example/index.html")); | 1470 | assert!(page(&out, "index.html").contains("https://moved.example/index.html")); |
| 1471 | } | 1471 | } |
| 1472 | |||
| 1473 | // --------------------------------------------------------------------------- | ||
| 1474 | // Excerpts, reading metadata, and drafts | ||
| 1475 | // --------------------------------------------------------------------------- | ||
| 1476 | |||
| 1477 | /// Posts with and without a `#+DESCRIPTION:`, and a template that prints the metadata. | ||
| 1478 | fn write_excerpt_site(src: &Utf8PathBuf, extra_config: &str) { | ||
| 1479 | std::fs::create_dir_all(src.join("blog")).unwrap(); | ||
| 1480 | std::fs::create_dir_all(src.join("templates")).unwrap(); | ||
| 1481 | std::fs::write(src.join("index.org"), "#+TITLE: Home\n\nWelcome.\n").unwrap(); | ||
| 1482 | std::fs::write( | ||
| 1483 | src.join("blog/described.org"), | ||
| 1484 | "#+TITLE: Described\n#+DATE: 2024-02-02\n#+DESCRIPTION: A hand-written summary.\n\n\ | ||
| 1485 | The body's first paragraph, which is not the excerpt here.\n", | ||
| 1486 | ) | ||
| 1487 | .unwrap(); | ||
| 1488 | std::fs::write( | ||
| 1489 | src.join("blog/plain.org"), | ||
| 1490 | "#+TITLE: Plain\n#+DATE: 2024-01-01\n\nThe opening paragraph stands in for a summary.\n\n\ | ||
| 1491 | A second paragraph that should not appear in the excerpt.\n", | ||
| 1492 | ) | ||
| 1493 | .unwrap(); | ||
| 1494 | std::fs::write( | ||
| 1495 | src.join("templates/list.html"), | ||
| 1496 | "<html><body>{% for p in pages %}<li>{{ p.title }}|{{ p.excerpt }}|\ | ||
| 1497 | {{ p.word_count }}|{{ p.reading_time }}</li>{% endfor %}</body></html>", | ||
| 1498 | ) | ||
| 1499 | .unwrap(); | ||
| 1500 | std::fs::write( | ||
| 1501 | src.join("org-ssg.toml"), | ||
| 1502 | format!( | ||
| 1503 | "[[collections]]\nsource = \"blog\"\noutput = \"blog/index.html\"\n\ | ||
| 1504 | template = \"list.html\"\ntitle = \"Blog\"\n{extra_config}" | ||
| 1505 | ), | ||
| 1506 | ) | ||
| 1507 | .unwrap(); | ||
| 1508 | } | ||
| 1509 | |||
| 1510 | /// A listing of bare titles is thin. 176 of the 179 corpus files set a | ||
| 1511 | /// `#+DESCRIPTION:`, so that is the excerpt when it exists — and the first paragraph | ||
| 1512 | /// when it does not, so a page that never thought about summaries still has one. | ||
| 1513 | #[test] | ||
| 1514 | fn excerpts_prefer_the_description_and_fall_back_to_the_first_paragraph() { | ||
| 1515 | let root = tmpdir("excerpt"); | ||
| 1516 | let src = root.join("src"); | ||
| 1517 | std::fs::create_dir_all(&src).unwrap(); | ||
| 1518 | write_excerpt_site(&src, ""); | ||
| 1519 | let out = root.join("out"); | ||
| 1520 | build(&src, &out); | ||
| 1521 | |||
| 1522 | let listing = page(&out, "blog/index.html"); | ||
| 1523 | assert!( | ||
| 1524 | listing.contains("Described|A hand-written summary.|"), | ||
| 1525 | "an explicit description wins:\n{listing}" | ||
| 1526 | ); | ||
| 1527 | assert!( | ||
| 1528 | listing.contains("Plain|The opening paragraph stands in for a summary.|"), | ||
| 1529 | "otherwise the first paragraph:\n{listing}" | ||
| 1530 | ); | ||
| 1531 | assert!( | ||
| 1532 | !listing.contains("A second paragraph"), | ||
| 1533 | "only the *first* paragraph:\n{listing}" | ||
| 1534 | ); | ||
| 1535 | } | ||
| 1536 | |||
| 1537 | /// Reading time should describe the prose someone reads, not the code they skim. | ||
| 1538 | #[test] | ||
| 1539 | fn word_count_and_reading_time_ignore_code_blocks() { | ||
| 1540 | let root = tmpdir("wordcount"); | ||
| 1541 | let src = root.join("src"); | ||
| 1542 | std::fs::create_dir_all(&src).unwrap(); | ||
| 1543 | write_excerpt_site(&src, ""); | ||
| 1544 | let prose = "word ".repeat(400); | ||
| 1545 | std::fs::write( | ||
| 1546 | src.join("blog/plain.org"), | ||
| 1547 | format!( | ||
| 1548 | "#+TITLE: Plain\n#+DATE: 2024-01-01\n\n{prose}\n\n\ | ||
| 1549 | #+BEGIN_SRC rust\n{}\n#+END_SRC\n", | ||
| 1550 | "let noise = 1; ".repeat(200) | ||
| 1551 | ), | ||
| 1552 | ) | ||
| 1553 | .unwrap(); | ||
| 1554 | let out = root.join("out"); | ||
| 1555 | build(&src, &out); | ||
| 1556 | |||
| 1557 | let listing = page(&out, "blog/index.html"); | ||
| 1558 | // Exactly the 400 prose words: the 600+ words of code are not prose, and neither is | ||
| 1559 | // `#+TITLE:`, which is metadata the layout renders as chrome rather than body text. | ||
| 1560 | assert!( | ||
| 1561 | listing.contains("|400|2</li>"), | ||
| 1562 | "code must not inflate the count or the estimate:\n{listing}" | ||
| 1563 | ); | ||
| 1564 | } | ||
| 1565 | |||
| 1566 | /// An excerpt is usually a whole paragraph, and minijinja ships no `truncate`, so | ||
| 1567 | /// without one a listing's only options are the full paragraph or nothing. | ||
| 1568 | #[test] | ||
| 1569 | fn the_truncate_filter_cuts_on_a_word_boundary() { | ||
| 1570 | let root = tmpdir("truncate"); | ||
| 1571 | let src = root.join("src"); | ||
| 1572 | std::fs::create_dir_all(&src).unwrap(); | ||
| 1573 | write_excerpt_site(&src, ""); | ||
| 1574 | std::fs::write( | ||
| 1575 | src.join("templates/list.html"), | ||
| 1576 | "<html><body>{% for p in pages %}<li>{{ p.excerpt | truncate(20) }}</li>\ | ||
| 1577 | {% endfor %}<x>{{ \"short\" | truncate(20) }}</x></body></html>", | ||
| 1578 | ) | ||
| 1579 | .unwrap(); | ||
| 1580 | let out = root.join("out"); | ||
| 1581 | build(&src, &out); | ||
| 1582 | |||
| 1583 | let listing = page(&out, "blog/index.html"); | ||
| 1584 | assert!( | ||
| 1585 | listing.contains("<li>A hand-written…</li>"), | ||
| 1586 | "cut at a space, not mid-word:\n{listing}" | ||
| 1587 | ); | ||
| 1588 | assert!( | ||
| 1589 | listing.contains("<x>short</x>"), | ||
| 1590 | "text under the limit is untouched:\n{listing}" | ||
| 1591 | ); | ||
| 1592 | } | ||
| 1593 | |||
| 1594 | /// The point of marking something a draft is that it is not ready to be read. | ||
| 1595 | #[test] | ||
| 1596 | fn drafts_are_excluded_from_the_build_by_default() { | ||
| 1597 | let root = tmpdir("draft"); | ||
| 1598 | let src = root.join("src"); | ||
| 1599 | std::fs::create_dir_all(&src).unwrap(); | ||
| 1600 | write_excerpt_site(&src, ""); | ||
| 1601 | std::fs::write( | ||
| 1602 | src.join("blog/wip.org"), | ||
| 1603 | "#+TITLE: Unfinished\n#+DRAFT: t\n#+DATE: 2024-03-03\n\nNot ready.\n", | ||
| 1604 | ) | ||
| 1605 | .unwrap(); | ||
| 1606 | let out = root.join("out"); | ||
| 1607 | build(&src, &out); | ||
| 1608 | |||
| 1609 | assert!(!out.join("blog/wip.html").exists(), "no page is written"); | ||
| 1610 | assert!( | ||
| 1611 | !page(&out, "blog/index.html").contains("Unfinished"), | ||
| 1612 | "and it is absent from listings, not merely unlinked" | ||
| 1613 | ); | ||
| 1614 | } | ||
| 1615 | |||
| 1616 | /// `--drafts` is for previewing one while writing it, typically under `watch`. | ||
| 1617 | #[test] | ||
| 1618 | fn the_drafts_flag_includes_them() { | ||
| 1619 | let root = tmpdir("draftflag"); | ||
| 1620 | let src = root.join("src"); | ||
| 1621 | std::fs::create_dir_all(&src).unwrap(); | ||
| 1622 | write_excerpt_site(&src, ""); | ||
| 1623 | std::fs::write( | ||
| 1624 | src.join("blog/wip.org"), | ||
| 1625 | "#+TITLE: Unfinished\n#+DRAFT: t\n#+DATE: 2024-03-03\n\nNot ready.\n", | ||
| 1626 | ) | ||
| 1627 | .unwrap(); | ||
| 1628 | let out = root.join("out"); | ||
| 1629 | build_site( | ||
| 1630 | &src, | ||
| 1631 | &out, | ||
| 1632 | &BuildOptions { | ||
| 1633 | drafts: true, | ||
| 1634 | ..Default::default() | ||
| 1635 | }, | ||
| 1636 | ) | ||
| 1637 | .expect("build"); | ||
| 1638 | |||
| 1639 | assert!(out.join("blog/wip.html").exists()); | ||
| 1640 | assert!(page(&out, "blog/index.html").contains("Unfinished")); | ||
| 1641 | } | ||
| 1642 | |||
| 1643 | /// A draft is absent from the symbol table too, so a link to one is reported as the dead | ||
| 1644 | /// link it would be on the published site — rather than silently pointing at nothing. | ||
| 1645 | #[test] | ||
| 1646 | fn a_link_to_a_draft_is_reported_as_broken() { | ||
| 1647 | let root = tmpdir("draftlink"); | ||
| 1648 | let src = root.join("src"); | ||
| 1649 | std::fs::create_dir_all(&src).unwrap(); | ||
| 1650 | write_excerpt_site(&src, ""); | ||
| 1651 | std::fs::write( | ||
| 1652 | src.join("blog/wip.org"), | ||
| 1653 | "#+TITLE: Unfinished\n#+DRAFT: t\n\nNot ready.\n", | ||
| 1654 | ) | ||
| 1655 | .unwrap(); | ||
| 1656 | std::fs::write( | ||
| 1657 | src.join("index.org"), | ||
| 1658 | "#+TITLE: Home\n\nSee [[file:blog/wip.org][the draft]].\n", | ||
| 1659 | ) | ||
| 1660 | .unwrap(); | ||
| 1661 | let out = root.join("out"); | ||
| 1662 | let report = build(&src, &out); | ||
| 1663 | |||
| 1664 | assert!( | ||
| 1665 | report.warnings().iter().any(|w| w.contains("wip.org")), | ||
| 1666 | "linking to a draft must be reported: {:?}", | ||
| 1667 | report.warnings() | ||
| 1668 | ); | ||
| 1669 | } | ||
| 1670 | |||
| 1671 | /// Writing the keyword at all is the signal. Publishing an unfinished post because the | ||
| 1672 | /// value was not the expected spelling is the wrong way to be strict — but an explicit | ||
| 1673 | /// "no" has to mean no. | ||
| 1674 | #[test] | ||
| 1675 | fn draft_truthiness_is_forgiving_but_respects_an_explicit_negative() { | ||
| 1676 | use org_ssg::model::Keywords; | ||
| 1677 | let draft = |value: &str| { | ||
| 1678 | org_ssg::util::is_draft(&Keywords { | ||
| 1679 | entries: vec![("DRAFT".to_string(), value.to_string())], | ||
| 1680 | }) | ||
| 1681 | }; | ||
| 1682 | for yes in ["t", "true", "yes", "1", "", " ", "anything"] { | ||
| 1683 | assert!(draft(yes), "{yes:?} should mean draft"); | ||
| 1684 | } | ||
| 1685 | for no in ["nil", "false", "no", "0", "off", "NIL"] { | ||
| 1686 | assert!(!draft(no), "{no:?} should mean published"); | ||
| 1687 | } | ||
| 1688 | assert!( | ||
| 1689 | !org_ssg::util::is_draft(&Keywords::default()), | ||
| 1690 | "no keyword at all means published" | ||
| 1691 | ); | ||
| 1692 | } | ||