| @@ -1,726 +1,107 @@ |
| 1 | * orgo |
1 | * orgo |
| 2 | An org-mode static site generator, in Rust. Org is treated as the /source language/, |
| |
| 3 | not an inconvenient input to be normalized into markdown. The org element tree — |
| |
| 4 | headings, drawers, blocks, links with their org-specific semantics — *is* the |
| |
| 5 | document model, and we render that tree straight to HTML. We never round-trip through |
| |
| 6 | a markdown-shaped intermediate representation, because the point is to preserve what |
| |
| 7 | markdown cannot express: property drawers, TODO/priority/tag metadata on headings, |
| |
| 8 | =#+= directives, ID links, named/captioned blocks, footnote semantics. |
| |
| 9 | |
2 | |
| 10 | The one non-obvious early commitment is *incremental builds keyed on content |
3 | Turn a folder of org files into a website. |
| 11 | hashing*, treated as a first-class architectural concern from day one. The discipline |
| |
| 12 | it imposes on the data model — pure, hashable, dependency-tracked units — is the real |
| |
| 13 | deliverable, even while the corpus is small enough that a full rebuild is instant. |
| |
| 14 | |
4 | |
| 15 | *Full documentation is in [[file:docs/][=docs/=]]* — a site written in org and built by |
5 | You write posts the way you already do — an =.org= file per page, in whatever directory |
| 16 | orgo itself. Build and read it with: |
6 | structure suits you — and orgo builds a complete site from them: pages, navigation, a blog |
| |
7 | index, tags, an RSS feed, syntax-highlighted code. It is one binary with nothing to |
| |
8 | install alongside it, and *you do not need Emacs to build your site*, only to write in a |
| |
9 | format Emacs made. |
| 17 | |
10 | |
| 18 | #+begin_src sh |
11 | Org is the source language here, not something to convert away from first. Tools that |
| 19 | cargo run -- serve docs -o docs/_site |
12 | route org through markdown lose what markdown has no words for — property drawers, a |
| 20 | #+end_src |
13 | heading's TODO state and tags, =#+= keywords, ID links, captions on images. orgo keeps all |
| 21 | |
14 | of it, and its output is checked page by page against what Emacs' own exporter produces |
| 22 | ** Quick start |
15 | from the same file. |
| 23 | #+begin_src sh |
| |
| 24 | cargo run -- init my-site # config + an editable copy of the layout + a page |
| |
| 25 | cargo run -- build my-site -o _site |
| |
| 26 | #+end_src |
| |
| 27 | |
| |
| 28 | Or skip the scaffolding entirely — point it at any directory of =.org= files: |
| |
| 29 | |
| |
| 30 | #+begin_src sh |
| |
| 31 | cargo run -- build ~/notes -o _site |
| |
| 32 | #+end_src |
| |
| 33 | |
| |
| 34 | *Zero configuration is a supported path, not a demo.* With no =orgo.toml=, no |
| |
| 35 | templates and no orgo-specific markup in your files, you get a complete site: pages, |
| |
| 36 | navigation, syntax-highlighted code and the stylesheet to colour it. Configuration |
| |
| 37 | changes what you get; it is never what makes it work. |
| |
| 38 | |
| |
| 39 | Discovery skips what should not be published — dot-directories such as =.git=, the config |
| |
| 40 | file, the templates directory, and the output directory when it sits inside the source, so |
| |
| 41 | =orgo build . -o _site= does the obvious thing. |
| |
| 42 | |
| |
| 43 | ** Configuration |
| |
| 44 | Everything is optional. =orgo init= writes a fully commented =orgo.toml=; every |
| |
| 45 | value below is the default. |
| |
| 46 | |
| |
| 47 | #+begin_src toml |
| |
| 48 | [site] |
| |
| 49 | title = "orgo site" |
| |
| 50 | base_url = "" # absolute URL, no trailing slash; needed for feeds/canonical links |
| |
| 51 | description = "" |
| |
| 52 | language = "en" |
| |
| 53 | |
| |
| 54 | [nav] |
| |
| 55 | mode = "top-level" # top-level | all | explicit | none |
| |
| 56 | # pages = ["index.org", "about.org"] # for mode = "explicit"; order is preserved |
| |
| 57 | |
| |
| 58 | [templates] |
| |
| 59 | dir = "templates" # base.html replaces the built-in layout |
| |
| 60 | expose_page_list = false |
| |
| 61 | |
| |
| 62 | # [[pages]] # which layout a section renders through; base.html by default |
| |
| 63 | # match = "blog" # a source directory or one .org file; most specific rule wins |
| |
| 64 | # template = "post.html" |
| |
| 65 | |
| |
| 66 | [highlight] |
| |
| 67 | theme = "InspiredGitHub" |
| |
| 68 | |
| |
| 69 | [build] |
| |
| 70 | drafts = false |
| |
| 71 | assets = [] # extra directories copied to the site root, e.g. ["../theme/static"] |
| |
| 72 | |
| |
| 73 | [html] |
| |
| 74 | heading_offset = 1 # a level-1 org heading becomes <h2>, beneath the layout's <h1> |
| |
| 75 | #+end_src |
| |
| 76 | |
| |
| 77 | *** Templates |
| |
| 78 | Drop a =base.html= into the templates directory and it replaces the built-in layout |
| |
| 79 | entirely. Any other =.html= file there is available to ={% include %}= and |
| |
| 80 | ={% extends %}=. Templates are [[https://docs.rs/minijinja][minijinja]] (Jinja2 syntax) and |
| |
| 81 | receive: |
| |
| 82 | |
| |
| 83 | | Variable | What it is | |
| |
| 84 | |————--+————————————————————————————————————————————————--| |
| |
| 85 | | =body= | the rendered page HTML — use ={{ body \| safe }}= | |
| |
| 86 | | =page= | =.title=, =.url=, =.source=, =.date=, =.date_iso=, =.year=, =.tags=, =.content=, =.excerpt=, =.word_count=, =.reading_time=, =.toc=, =.keywords= | |
| |
| 87 | | =site= | =.title=, =.base_url=, =.description=, =.language= | |
| |
| 88 | | =nav= | list of ={title, url}=, relative to this page | |
| |
| 89 | | =root= | =../=-prefix back to the site root from this page | |
| |
| 90 | | =stylesheet= | URL of the generated =syntax.css= | |
| |
| 91 | | =pages= | every page's metadata — only when =expose_page_list = true= | |
| |
| 92 | |
| |
| 93 | =page.keywords= carries *every* =#+KEYWORD:= in the file under its lowercased name, so |
| |
| 94 | your own metadata works without this crate knowing about it: =#+CUSTOM_THING: x= is |
| |
| 95 | ={{ page.keywords.custom_thing }}=. |
| |
| 96 | |
| |
| 97 | =base.html= is the default layout, not the only one. A =[[pages]]= rule gives a section |
| |
| 98 | its own — =match = "blog"=, =template = "post.html"= — and =#+TEMPLATE: wide.html= gives |
| |
| 99 | one page its own, which wins over any rule. A second layout usually starts with |
| |
| 100 | ={% extends "base.html" %}=. |
| |
| 101 | |
| |
| 102 | Editing a template re-renders the pages that use it — template sources are a hash input, |
| |
| 103 | so a design change never leaves a site half-updated. |
| |
| 104 | |
| |
| 105 | *** Generated listing pages |
| |
| 106 | A blog index, an archive, a feed — output files with no source =.org= behind them. |
| |
| 107 | Repeat the block for each one: |
| |
| 108 | |
| |
| 109 | #+begin_src toml |
| |
| 110 | [[collections]] |
| |
| 111 | source = "blog" # directory to list; empty means every page |
| |
| 112 | output = "blog/index.html" # where to write it |
| |
| 113 | template = "list.html" |
| |
| 114 | title = "Blog" |
| |
| 115 | sort = "date" # date | title | path |
| |
| 116 | order = "desc" # desc | asc |
| |
| 117 | nav = true # put this listing page in the nav |
| |
| 118 | #+end_src |
| |
| 119 | |
| |
| 120 | The template gets the collection's entries as =pages=, already sorted, plus the usual |
| |
| 121 | =site=/=nav=/=root=. It can ={% extends "base.html" %}= to inherit the site chrome: |
| |
| 122 | |
| |
| 123 | #+begin_src jinja |
| |
| 124 | {% extends "base.html" %} |
| |
| 125 | {% block main %} |
| |
| 126 | <ul>{% for p in pages %} |
| |
| 127 | <li><time datetime="{{ p.date_iso }}">{{ p.date_iso }}</time> |
| |
| 128 | <a href="{{ root }}{{ p.url }}">{{ p.title }}</a></li> |
| |
| 129 | {% endfor %}</ul> |
| |
| 130 | {% endblock %} |
| |
| 131 | #+end_src |
| |
| 132 | |
| |
| 133 | =p.date_iso= is the =YYYY-MM-DD= extracted from =#+DATE:=, whatever org syntax it was |
| |
| 134 | written in — =[2025-09-05 Fri 10:21:00]=, =<2024-05-01 Wed>= or bare =2024-05-01=. It is |
| |
| 135 | also the sort key; pages without a parseable date sort last, so an undated draft never |
| |
| 136 | leads a dated archive. |
| |
| 137 | |
| |
| 138 | **** Pagination |
| |
| 139 | Set =paginate= to split a long listing across numbered pages: |
| |
| 140 | |
| |
| 141 | #+begin_src toml |
| |
| 142 | [[collections]] |
| |
| 143 | source = "blog" |
| |
| 144 | output = "blog/index.html" |
| |
| 145 | paginate = 10 |
| |
| 146 | paginate_output = "blog/page/{n}.html" # {n} is the 1-based page number |
| |
| 147 | #+end_src |
| |
| 148 | |
| |
| 149 | Page 1 stays at =output=, so a section's canonical URL never moves as its page count |
| |
| 150 | changes; only pages 2..N are named by =paginate_output=. The template gets a =paginator=: |
| |
| 151 | |
| |
| 152 | #+begin_src jinja |
| |
| 153 | {% if paginator and paginator.total > 1 %} |
| |
| 154 | <nav> |
| |
| 155 | {% if paginator.prev_url %}<a href="{{ paginator.prev_url }}">Newer</a>{% endif %} |
| |
| 156 | {% for pg in paginator.pages %} |
| |
| 157 | <a href="{{ pg.url }}"{% if pg.current %} aria-current="page"{% endif %}>{{ pg.number }}</a> |
| |
| 158 | {% endfor %} |
| |
| 159 | {% if paginator.next_url %}<a href="{{ paginator.next_url }}">Older</a>{% endif %} |
| |
| 160 | </nav> |
| |
| 161 | {% endif %} |
| |
| 162 | #+end_src |
| |
| 163 | |
| |
| 164 | =paginator= carries =current=, =total=, =per_page=, =total_entries=, =prev_url=, |
| |
| 165 | =next_url=, =first_url=, =last_url=, and =pages=. Every URL is relative to the page |
| |
| 166 | carrying it, so links work from page 1 (=page/2.html=) and from page 5 (=../index.html=, |
| |
| 167 | =6.html=) without the template knowing where it sits. An unpaginated collection has no |
| |
| 168 | =paginator= at all, so ={% if paginator %}= is a reliable test in a shared template. |
| |
| 169 | |
| |
| 170 | Grouping and pagination compose: each group paginates independently, which is why |
| |
| 171 | =paginate_output= needs ={tag}= as well as ={n}= on a grouped collection. An empty |
| |
| 172 | collection still emits page 1 — a section that exists but has nothing in it should say so |
| |
| 173 | rather than 404. When the entry count shrinks, pages that no longer exist are deleted |
| |
| 174 | instead of being left serving stale posts. |
| |
| 175 | |
| |
| 176 | **** Tag pages |
| |
| 177 | Add =group_by= and the collection emits one page /per group/ instead of one page total, |
| |
| 178 | plus an optional index of the groups: |
| |
| 179 | |
| |
| 180 | #+begin_src toml |
| |
| 181 | [[collections]] |
| |
| 182 | source = "blog" |
| |
| 183 | group_by = "tags" # "tags", or any #+KEYWORD: name to group by its value |
| |
| 184 | output = "tags/{tag}.html" # {tag} is replaced by each group's slug |
| |
| 185 | template = "tag.html" |
| |
| 186 | title = "Tagged: {tag}" |
| |
| 187 | index_output = "tags/index.html" # the tag index |
| |
| 188 | index_template = "tags.html" |
| |
| 189 | index_title = "Tags" |
| |
| 190 | nav = true # adds the *index*, not every tag |
| |
| 191 | #+end_src |
| |
| 192 | |
| |
| 193 | A group page receives its own posts as =pages= and itself as =group= |
| |
| 194 | (=.name=, =.slug=, =.url=, =.count=). The index receives =groups= — every group, sorted |
| |
| 195 | by name: |
| |
| 196 | |
| |
| 197 | #+begin_src jinja |
| |
| 198 | <ul>{% for tag in groups %} |
| |
| 199 | <li><a href="{{ root }}{{ tag.url }}">{{ tag.name }}</a> ({{ tag.count }})</li> |
| |
| 200 | {% endfor %}</ul> |
| |
| 201 | #+end_src |
| |
| 202 | |
| |
| 203 | =group_by = "tags"= is multi-valued: a post appears under every tag it carries. Any other |
| |
| 204 | value names a single-valued =#+KEYWORD:=, so =group_by = "category"= buckets by |
| |
| 205 | =#+CATEGORY:=. |
| |
| 206 | |
| |
| 207 | Two tags that would produce the same URL (=web_dev= and =web@dev= both slugify to |
| |
| 208 | =web-dev=) are a build error rather than one page silently overwriting the other. |
| |
| 209 | |
| |
| 210 | A tag page depends on its own posts and nothing else, so adding a post tagged =rust= |
| |
| 211 | re-renders that post, its section index, =tags/rust.html=, and the tag index whose counts |
| |
| 212 | changed — four pages, not one per tag. That precision is why =groups= is given to the |
| |
| 213 | index and not to every group page: a page that can see every group depends on every |
| |
| 214 | group. |
| |
| 215 | |
| |
| 216 | **** Feeds and absolute URLs |
| |
| 217 | *A feed is a listing page with an XML template*, not a separate feature — templates are |
| |
| 218 | loaded by full filename and any extension, so =output = "feed.xml"= with |
| |
| 219 | =template = "feed.xml"= is all it takes. =orgo init= writes a working RSS template. |
| |
| 220 | |
| |
| 221 | A feed is read away from the site that served it, so relative links in one are simply |
| |
| 222 | broken. Set =site.base_url= and use the =absolute= filter: |
| |
| 223 | |
| |
| 224 | #+begin_src jinja |
| |
| 225 | <link>{{ post.url | absolute }}</link> |
| |
| 226 | <pubDate>{{ post.date_iso | rfc822 }}</pubDate> |
| |
| 227 | #+end_src |
| |
| 228 | |
16 | |
| 229 | | Filter | Does | |
17 | *Documentation: https://ccleberg.github.io/orgo/* — that site is written in org and built |
| 230 | |—————+—————————————————————————-| |
18 | by orgo, so it doubles as the longest worked example available. |
| 231 | | =absolute= | site-root-relative path → absolute URL; already-absolute URLs pass through | |
| |
| 232 | | =rfc822= | any org or ISO date → the format RSS =pubDate= requires | |
| |
| 233 | | =truncate(n)= | shorten to at most =n= characters on a word boundary, with an ellipsis | |
| |
| 234 | |
19 | |
| 235 | Apply =absolute= to the site-root-relative values — =page.url=, =pages[].url=, |
20 | ** Install |
| 236 | =group.url= — and not to =nav[].url=, =paginator.*_url=, =stylesheet= or =root=, which |
| |
| 237 | are relative to the page carrying them and already correct there. |
| |
| 238 | |
21 | |
| 239 | With no =base_url=, =absolute= is an *error* naming the setting, rather than quietly |
22 | You need [[https://rustup.rs][Rust]] (1.88 or newer). Nothing else — syntax highlighting |
| 240 | emitting a relative URL that would make the feed invalid everywhere while looking fine. |
23 | and its themes are compiled in. |
| 241 | The default layout also emits =<link rel="canonical">= when a base URL is set. |
| |
| 242 | |
24 | |
| 243 | Listing pages are cached on the entries they list, so adding a post re-renders that |
25 | #+begin_src sh |
| 244 | section's index and nothing else. |
26 | git clone https://github.com/ccleberg/orgo |
| 245 | |
27 | cd orgo |
| 246 | *** Table of contents and =#+OPTIONS:= |
28 | cargo install --path . |
| 247 | =page.toc= is the page's headings as a *tree* — ={title, anchor, level, children}= — |
| |
| 248 | because a table of contents is one, and rebuilding a tree from a flat list of levels |
| |
| 249 | inside a template is what Jinja is worst at. Its anchors come from the same function the |
| |
| 250 | renderer uses to emit heading =id=s, so a TOC link cannot drift from the heading it |
| |
| 251 | points at. |
| |
| 252 | |
| |
| 253 | #+begin_src jinja |
| |
| 254 | {% macro toc_list(entries) %} |
| |
| 255 | <ul>{% for e in entries %} |
| |
| 256 | <li><a href="#{{ e.anchor }}">{{ e.title }}</a> |
| |
| 257 | {%- if e.children %}{{ toc_list(e.children) }}{% endif %}</li> |
| |
| 258 | {% endfor %}</ul> |
| |
| 259 | {% endmacro %} |
| |
| 260 | {% if page.toc %}{{ toc_list(page.toc) }}{% endif %} |
| |
| 261 | #+end_src |
29 | #+end_src |
| 262 | |
30 | |
| 263 | Org's own per-file export switches are honoured, so a document can turn a feature off for |
31 | That puts an =orgo= command on your =PATH=. Full notes, including how to run it without |
| 264 | itself the way its author already knows: |
32 | installing anything: https://ccleberg.github.io/orgo/install.html |
| 265 | |
| |
| 266 | | Switch | Effect | Site default | |
| |
| 267 | |———————-+————————————+———————————-| |
| |
| 268 | | =#+OPTIONS: toc:nil= | empties =page.toc= for this page | =[html] toc = true= | |
| |
| 269 | | =#+OPTIONS: num:t= | numbers headings =1.=, =1.1.=, … | =[html] section_numbers = false= | |
| |
| 270 | |
| |
| 271 | *Section numbers default to off, which differs from Emacs on purpose.* |
| |
| 272 | =org-export-with-section-numbers= is on there, so an org-published site inherits numbered |
| |
| 273 | headings whether or not anyone chose them. Most sites do not want them; =num:t= or |
| |
| 274 | =section_numbers = true= gets Emacs' behaviour back, with Emacs' own |
| |
| 275 | =section-number-N= classes so the output stays diffable against the oracle. |
| |
| 276 | |
| |
| 277 | *** Excerpts and drafts |
| |
| 278 | =page.excerpt= is a page's =#+DESCRIPTION:= when it sets one and its first paragraph |
| |
| 279 | otherwise, so a listing has something to show whether or not the author thought about |
| |
| 280 | summaries. =page.word_count= and =page.reading_time= (minutes at 200 wpm) count prose |
| |
| 281 | only — a post that is mostly a shell transcript should not read as an hour's work. |
| |
| 282 | =truncate= exists because an excerpt is usually a whole paragraph and minijinja has no |
| |
| 283 | such filter. |
| |
| 284 | |
| |
| 285 | =#+DRAFT:= keeps a page out of the build entirely — no page, and absent from listings and |
| |
| 286 | the nav rather than merely unlinked. =--drafts= includes them, which is what you want |
| |
| 287 | under =watch= while writing one. A draft is out of the symbol table too, so a link /to/ |
| |
| 288 | one is reported as the dead link it would be once published. |
| |
| 289 | |
| |
| 290 | The keyword is read forgivingly: =t=, =yes=, =1= and a bare =#+DRAFT:= all mean draft, |
| |
| 291 | because writing the keyword at all is the signal. Only an explicit =nil=, =false=, =no=, |
| |
| 292 | =0= or =off= means published. |
| |
| 293 | |
| |
| 294 | *** =#+SLUG:= |
| |
| 295 | A page's output filename comes from its =#+SLUG:= when it has one, so |
| |
| 296 | =2018-11-28-aes-encryption.org= can publish as =aes-encryption.html=. Without one the |
| |
| 297 | source filename is used. Slugs are sanitized to a single safe path component, and two |
| |
| 298 | pages claiming one URL is a build error rather than a silently dropped page. |
| |
| 299 | |
| |
| 300 | ** Pipeline |
| |
| 301 | #+begin_example |
| |
| 302 | DISCOVER → PARSE → INDEX → RESOLVE → RENDER → TEMPLATE → EMIT |
| |
| 303 | #+end_example |
| |
| 304 | |
| |
| 305 | PARSE and RENDER are pure functions of their inputs (cacheable, hashable). INDEX/RESOLVE |
| |
| 306 | is the only inherently global stage — it is where the link dependency graph is born. |
| |
| 307 | |
| |
| 308 | | Stage | Module | Notes | |
| |
| 309 | |————-+———————-+———————————————————————————-| |
| |
| 310 | | config | =src/config.rs= | =orgo.toml=: site metadata, nav mode, templates, theme. A hash input. | |
| |
| 311 | | PARSE | =src/parser.rs= | Hand-written recursive descent: line lexer → element builder → inline tokenizer. | |
| |
| 312 | | audit | =src/audit.rs= | Phase 0 corpus audit: construct frequencies against the IN/OUT line. | |
| |
| 313 | | model | =src/model.rs= | The org element tree — Elements (block) vs Objects (inline). | |
| |
| 314 | | INDEX | =src/index.rs= | Collect link targets into a symbol table. | |
| |
| 315 | | RESOLVE | =src/resolve.rs= | Rewrite links to URLs; return the used-target list (dependency edges). | |
| |
| 316 | | RENDER | =src/render.rs= | Tree → HTML fragment; syntect highlighting; footnote two-pass. | |
| |
| 317 | | TEMPLATE | =src/template.rs= | minijinja: fragment + metadata → full page. | |
| |
| 318 | | incremental | =src/incremental.rs= | Content/config/template hashing, dep graph, cache manifest, invalidation. | |
| |
| 319 | |
| |
| 320 | ** v1 scope (delivered as of v0.4; still to be reconciled against a corpus audit) |
| |
| 321 | *IN — v1 must handle:* headings with nesting, at levels relative to the document's |
| |
| 322 | shallowest; TODO keywords; priorities =[#A]=; tags; property drawers; plain lists |
| |
| 323 | (unordered/ordered/description, checkboxes, =[@N]= counters, nesting); tables (with rule |
| |
| 324 | rows and org's special marker column, no =#+TBLFM:=); source blocks with syntax |
| |
| 325 | highlighting; example/quote/center/verse blocks and named special blocks; links (external, |
| |
| 326 | internal =[[*Heading]]=/=[[#custom-id]]=, =id:=); footnotes (inline and referenced); =#+= |
| |
| 327 | keywords/directives; inline markup (bold/italic/underline/verbatim/code/strike); org's |
| |
| 328 | export-time text conversions (=--=/=---=/=...=, =x^2=, =a_{b}=, =\alpha=); timestamps |
| |
| 329 | (active/inactive, ranges); paragraphs and horizontal rules; images with |
| |
| 330 | =#+CAPTION=/=#+ATTR_HTML=, numbered =Figure N:=. |
| |
| 331 | |
33 | |
| 332 | *OUT — explicitly not v1 (parse-and-ignore or reject loudly):* Babel execution / |
34 | ** Your first site |
| 333 | =:results=; =#+TBLFM:= formulas; LaTeX / MathJax (passed through untouched, including past |
| |
| 334 | the text conversions); =#+INCLUDE:= (never expanded — reported as a diagnostic, so a page |
| |
| 335 | is never quietly short of content); citations; radio targets and macros; drawers other |
| |
| 336 | than PROPERTIES/LOGBOOK; column view / clocking / agenda semantics; non-HTML export |
| |
| 337 | blocks. |
| |
| 338 | |
35 | |
| 339 | *Scope guardrail:* every IN item gets a golden-file fixture; every OUT item gets a test |
| |
| 340 | asserting it degrades predictably (ignored, no crash). The IN/OUT line is enforced by |
| |
| 341 | =tests/constructs.rs=, defending against the project's #1 risk: scope creep back toward |
| |
| 342 | all-of-org. Phase 0 checked this line against a real 179-file corpus and found it sound |
| |
| 343 | (99.9% of construct uses in scope) — but also found one thing missing from it entirely: |
| |
| 344 | =#+SLUG:=. See [[#phase-0-the-corpus-audit-and-the-emacs-oracle][Phase 0]]. |
| |
| 345 | |
| |
| 346 | ** Phase plan |
| |
| 347 | | Phase | Scope | Status | |
| |
| 348 | |——--+——————————————————————————————————————————————————————————+——--| |
| |
| 349 | | *M0* | *Buildable skeleton: crate layout, module stubs, deps, test harness, fixtures* | *done* | |
| |
| 350 | | *v0.1* | *End-to-end core parse → render: =build= a single =.org= file to HTML* | *done* | |
| |
| 351 | | *v0.2* | *Multi-file SITE build: INDEX + RESOLVE internal links, minijinja templates, =build <src-dir> <out-dir>=, tables + footnotes* | *done* | |
| |
| 352 | | *v0.3* | *Incremental build layer: content/config/template hashing, dependency graph, per-page render keys, persisted cache manifest, invalidation* | *done* | |
| |
| 353 | | *v0.4* | *MVP: the full v1 construct scope — heading metadata, nested/description lists, block types, timestamps, images, syntect highlighting — with the IN/OUT line under test* | *done* | |
| |
| 354 | | *0* | *Corpus audit + =emacs --batch= ground-truth oracle* | *done* | |
| |
| 355 | | 1 | Line lexer + heading/section skeleton | done | |
| |
| 356 | | 2 | Block elements — lists, source blocks, tables, footnote defs, blocks by type, drawers | done | |
| |
| 357 | | 3 | Inline objects — emphasis, links, bare URLs, footnote refs, timestamps | done | |
| |
| 358 | | 4 | Rendering to HTML — tree walk, tables, footnote two-pass, minijinja templating, syntect highlighting | done | |
| |
| 359 | | 5 | Link resolution + symbol table (INDEX + RESOLVE, used-target list, broken-link reporting) | done | |
| |
| 360 | | 6 | Incremental build layer (hashing, dep graph, invalidation); =watch= on OS filesystem events | done | |
| |
| 361 | | *7* | *Hardening: rayon parallelism, error locations in parse diagnostics* | *done* | |
| |
| 362 | | *8* | *General use: config file, user templates, nav modes, =init= scaffold, safe discovery* | *done* | |
| |
| 363 | | *9* | *Generated listing pages: =[[collections]]=, sorted indexes, feeds via XML templates* | *done* | |
| |
| 364 | | *10* | *Grouped collections: one page per tag plus a tag index — full parity with the incumbent* | *done* | |
| |
| 365 | | *11* | *Pagination: numbered pages with a =paginator= context, composing with grouping* | *done* | |
| |
| 366 | | *12* | *=base_url=: =absolute=/=rfc822= filters, a valid RSS feed in the scaffold, canonical links* | *done* | |
| |
| 367 | | *13* | *=watch= on OS filesystem events, debounced, with the feedback loop closed* | *done* | |
| |
| 368 | | *14* | *Authoring: excerpts, word count, reading time, =truncate=, and draft pages* | *done* | |
| |
| 369 | | *15* | *Table of contents, section numbers, and org's =#+OPTIONS:= per-file switches* | *done* | |
| |
| 370 | | *16* | *=serve=: development server with long-poll live reload, loopback-bound* | *done* | |
| |
| 371 | | *17* | *Bundled TOML and Org syntaxes, a user syntax directory, and org's comma escape* | *done* | |
| |
| 372 | | *18* | *Per-page layouts: =[[pages]]= rules and =#+TEMPLATE:=* | *done* | |
| |
| 373 | | *19* | *Export parity: relative heading levels, special strings, sub/superscript, caption numbering, checkbox and counter markup, table marker columns, special blocks* | *done* | |
| |
| 374 | | *20* | *Correctness debt: org's entity table, table captions, a reported =#+INCLUDE:=, and an oracle that separates deliberate divergence from defects* | *done* | |
| |
| 375 | | *21* | *Extra asset roots; per-template hashing so one layout edit does not re-render the site* | *done* | |
| |
| 376 | | *22* | *Release engineering: CI on both platforms, a checked MSRV, release binaries, a changelog, and a written compatibility promise* | *done* | |
| |
| 377 | |
| |
| 378 | *** v0.2 in / out |
| |
| 379 | *Added in v0.2:* the INDEX stage (=SymbolTable= of =:ID:=/=:CUSTOM_ID:=/heading/=file:= |
| |
| 380 | targets across a directory); the RESOLVE stage — rewrites =[[#custom-id]]=, =[[id:...]]=, |
| |
| 381 | =[[*Heading]]= and =[[file:other.org]]= links to real relative output URLs, returns the |
| |
| 382 | =used_targets= list (the =uses= edges, spec §4.3/R2) and reports unresolved links as |
| |
| 383 | warnings rather than crashing; a minijinja base layout (title, nav, body) applied to every |
| |
| 384 | page; a =build <src-dir> <out-dir>= path that walks the tree, parses + resolves + renders + |
| |
| 385 | templates every =.org= into a linked static site and copies non-=.org= assets through; |
| |
| 386 | plus two new constructs — pipe *tables* (with header band from the rule row) and |
| |
| 387 | *footnotes* (block =[fn:1]= definitions, referenced =[fn:1]=, and inline =[fn:1:text]=, |
| |
| 388 | rendered as a numbered, back-linked notes section). |
| |
| 389 | |
| |
| 390 | *Left stubbed at v0.2, all closed in v0.4:* timestamps; TODO keywords and priorities; |
| |
| 391 | generic (non-PROPERTIES) drawers; real syntect tokenizing behind the =Highlighter= trait. |
| |
| 392 | |
| |
| 393 | *** v0.3 in / out |
| |
| 394 | *Added in v0.3 — the incremental build layer (spec §4, the flagship, non-retrofittable |
| |
| 395 | feature):* |
| |
| 396 | |
| |
| 397 | - *Three hash classes (spec §4.1)* in =src/incremental.rs=: a *content hash* (blake3 |
| |
| 398 | of a file's bytes), a *config hash* (blake3 of the resolved =BuildConfig=), and a |
| |
| 399 | *template hash* (blake3 of the template sources). A change in any one invalidates the |
| |
| 400 | pages it affects. |
| |
| 401 | - *Dependency graph (spec §4.3)* built from RESOLVE's =defines=/=uses= edges: a page |
| |
| 402 | depends on the targets it links to, so editing (or renaming a heading in) a file |
| |
| 403 | invalidates the pages that /link into/ it, not just the file itself — the load-bearing |
| |
| 404 | R2 invariant. On rebuild the graph is merged with the previous build's =defines= so a |
| |
| 405 | /removed/ target still pulls in its linkers. |
| |
| 406 | - *Per-page =render_key=* = =H(content ⊕ resolved-links ⊕ config ⊕ template)=. If a |
| |
| 407 | page's render key is unchanged, its on-disk output is already correct and it is skipped. |
| |
| 408 | The config component folds in a *site-structure hash* (every page's =(path, title)=), |
| |
| 409 | because the shared nav bar is global chrome — a title change or a page add/remove alters |
| |
| 410 | the nav on every page and so must re-render them all (otherwise byte-equivalence breaks). |
| |
| 411 | - *Persisted cache manifest* (=<out>/.orgo-cache.json=, JSON), carrying per-page |
| |
| 412 | records, the config/template hashes, and the serialized dependency graph, tagged with |
| |
| 413 | =CACHE_FORMAT_VERSION=. A version mismatch, a missing file, or a corrupt file all fall |
| |
| 414 | back to a clean full rebuild — the cache is an optimization, never a correctness |
| |
| 415 | dependency. |
| |
| 416 | - *Wired into =build_site=*: only pages whose render key changed (or that link into a |
| |
| 417 | changed file's targets) are re-rendered; unchanged outputs are left in place. =--no-cache= |
| |
| 418 | forces a full rebuild; =clean <out-dir>= removes the output directory (and its cache). |
| |
| 419 | =SiteReport= now reports =rendered= vs =skipped= counts. |
| |
| 420 | |
| |
| 421 | The hard gates are enforced by =tests/incremental.rs=: full-vs-incremental *byte |
| |
| 422 | equivalence* (and a second unchanged build re-rendering *zero* pages); *edit-one-file* |
| |
| 423 | re-renders exactly the changed page plus its linkers; *renamed-heading* invalidates the |
| |
| 424 | linking page and updates its emitted anchor; and cache *version-bump / missing / corrupt* |
| |
| 425 | all fall back to a full rebuild. |
| |
| 426 | |
| |
| 427 | *Out of scope in v0.3:* real syntect highlighting; timestamps and TODO keywords (all |
| |
| 428 | landed in v0.4). =watch= is a minimal mtime poll loop (=watch <src-dir> -o <out-dir>=), not |
| |
| 429 | an OS file-watcher — the fs-notify integration is deferred. The parse-tree cache (spec §4.5, |
| |
| 430 | "optionally") is not persisted: PARSE/INDEX/RESOLVE run for every file each build (cheap and |
| |
| 431 | pure); the incremental win is on RENDER + EMIT. |
| |
| 432 | |
| |
| 433 | *** v0.4 in / out — the MVP |
| |
| 434 | v0.4 closes the gap between the v1 scope above and what the code actually did, so every |
| |
| 435 | construct the IN list claims is now parsed, rendered, and pinned by a golden file: |
| |
| 436 | |
| |
| 437 | - *Heading metadata* — TODO keywords (the Emacs default =TODO=/=DONE= set, matched on a |
| |
| 438 | word boundary so =TODOs= is not one) and =[#A]= priority cookies, rendered with Emacs' |
| |
| 439 | own export classes so the output stays diffable against an =emacs --batch= oracle. |
| |
| 440 | - *Lists* — indentation-based nesting (a sub-list renders /inside/ its parent =<li>=), |
| |
| 441 | multi-paragraph item bodies, and =term :: definition= description lists as =<dl>=. |
| |
| 442 | - *Blocks by type* — =QUOTE=, =CENTER=, =EXAMPLE=, =EXPORT= and =SRC= are now distinct |
| |
| 443 | elements rather than all collapsing to a verbatim example block. Block matching is on the |
| |
| 444 | specific kind, so a source block can nest inside a quote. An =html= export block passes |
| |
| 445 | through; every other backend drops. |
| |
| 446 | - *Timestamps* — active =<...>= and inactive =[...]=, optional times, same-day time |
| |
| 447 | ranges and =--=-joined date ranges, rendered as =<time>= with a machine-readable |
| |
| 448 | =datetime=. Repeater/warning cookies are recognized and discarded. |
| |
| 449 | - *Images* — a description-less link to an image file renders as =<img>=; with an |
| |
| 450 | affiliated =#+CAPTION:=/=#+ATTR_HTML:= it is promoted to a =<figure>= with the caption as |
| |
| 451 | both =<figcaption>= and alt text. Links to non-=.org= files are now understood as asset |
| |
| 452 | links: neither resolved nor reported as broken. |
| |
| 453 | - *Syntax highlighting* — real syntect tokenizing to CSS classes (never inline styles, so |
| |
| 454 | themes live in the stylesheet). Every build emits the matching =syntax.css= and each page |
| |
| 455 | links it relative to its own depth. An unknown language degrades to escaped =<pre><code>=. |
| |
| 456 | - *Diagnostics* — broken links are reported as the org syntax the author wrote |
| |
| 457 | (=warning: b.org: unresolved link [[#setup]]=) rather than a Debug-printed enum. |
| |
| 458 | |
| |
| 459 | *The OUT line is now enforced, not just asserted.* =tests/constructs.rs= pins each |
| |
| 460 | excluded construct to a specific degradation: babel is never executed /and/ a checked-in |
| |
| 461 | =#+RESULTS:= block is dropped rather than published as if it were verified output; |
| |
| 462 | =#+TBLFM:= is inert; =#+INCLUDE:= is never expanded and says so; LaTeX, macros and radio targets survive |
| |
| 463 | as literal text; drawers other than PROPERTIES are captured and dropped; unmodelled block |
| |
| 464 | types keep their content verbatim. |
| |
| 465 | |
| |
| 466 | *Still out:* =#+TODO:= per-file keyword sequences; planning lines |
| |
| 467 | (=SCHEDULED:=/=DEADLINE:=), which render as ordinary paragraphs; and fixed-width =:= |
| |
| 468 | lines. |
| |
| 469 | |
| |
| 470 | ** Serving |
| |
| 471 | #+begin_src sh |
36 | #+begin_src sh |
| 472 | cargo run -- serve my-site -o _site # http://127.0.0.1:3000 |
37 | orgo init my-site |
| |
38 | orgo serve my-site -o _site |
| 473 | #+end_src |
39 | #+end_src |
| 474 | |
40 | |
| 475 | Builds, watches, serves, and reloads the browser when a rebuild lands — the loop =watch= |
41 | Open http://127.0.0.1:3000. Edit =my-site/index.org=, save, and the page reloads on its |
| 476 | leaves half-open. |
42 | own — that is the loop you will spend your time in. |
| 477 | |
43 | |
| 478 | - *Loopback by default.* A dev server serves unreviewed drafts off your laptop, so |
44 | =init= writes a starter post, a page layout you can edit, and a config file with every |
| 479 | reaching the local network is something you ask for with =--host 0.0.0.0=, never |
45 | setting explained in comments. It never overwrites a file you already have. |
| 480 | something you get. |
| |
| 481 | - *The reload script is injected on the way out*, never written to disk. What you |
| |
| 482 | deploy is the built site, and it must not carry a dev server's JavaScript. |
| |
| 483 | - *Long-polling, not WebSockets or SSE.* The browser asks "anything since generation |
| |
| 484 | N?" and the server holds the request until there is. Instant like a push, no protocol |
| |
| 485 | beyond ordinary HTTP, and no dependency. A streamed response would have been more |
| |
| 486 | elegant and does not work: tiny_http buffers a response until its body ends, so a body |
| |
| 487 | that never ends never reaches the client. |
| |
| 488 | - A reload only follows a *successful* rebuild. Reloading onto a stale page because the |
| |
| 489 | build just failed tells you nothing; the error is already on your terminal. |
| |
| 490 | |
46 | |
| 491 | URL resolution is the server's security boundary and is written as a pure function with |
47 | ** Or point it at writing you already have |
| 492 | its own tests: =..=, percent-encoded =..=, backslashes, absolute paths and embedded NULs |
| |
| 493 | all resolve to nothing rather than to somewhere outside the output directory. |
| |
| 494 | |
48 | |
| 495 | ** Watching |
| |
| 496 | #+begin_src sh |
49 | #+begin_src sh |
| 497 | cargo run -- watch my-site -o _site |
50 | orgo build ~/notes -o _site |
| 498 | #+end_src |
51 | #+end_src |
| 499 | |
52 | |
| 500 | Rebuilds on OS filesystem events rather than polling, so it costs nothing while nothing |
53 | No config file, no templates, no orgo-specific markup in your files. You get a real site: |
| 501 | happens. Write bursts are debounced — an editor saving a file writes a temp file, renames |
54 | every page, links between them resolved, navigation across the top, code highlighted. That |
| 502 | it over the original and touches the directory, which is one edit and several events. |
55 | is a supported way to use it rather than a demo — configuration changes what you get, it |
| 503 | |
56 | is never what makes it work. |
| 504 | Two rules decide what counts as a change, and they are not the same rules the build uses |
| |
| 505 | to find content: |
| |
| 506 | |
57 | |
| 507 | - *A build input is a change.* Editing =orgo.toml= or a template rebuilds, even |
58 | Nothing that should stay private is published: dot-directories like =.git=, your templates |
| 508 | though discovery skips both as non-content. The question is "would this change the |
59 | and the output folder itself are all skipped. |
| 509 | site?", not "is this a page?". |
| |
| 510 | - *Our own output is not.* =watch . -o _site= puts the output inside the source, so a |
| |
| 511 | rebuild's writes raise events that would trigger a rebuild, forever. Dot-directories go |
| |
| 512 | the same way — =.git= churns on every command — as do editor scratch files, including |
| |
| 513 | Emacs' =file.org~= backups, which do not start with a dot. |
| |
| 514 | |
60 | |
| 515 | Where native watching is unavailable (some network and container filesystems), it falls |
61 | Want to know what orgo will make of your files before trusting it with them? |
| 516 | back to polling and says so, rather than failing. |
62 | =orgo audit ~/notes= reports which org constructs you use and how each one lands, with |
| |
63 | counts and line numbers — never the text of your writing, so the report is safe to share. |
| 517 | |
64 | |
| 518 | ** Phase 0: the corpus audit and the Emacs oracle |
65 | ** What you can add when you want it |
| 519 | The v1 scope was, by its own admission, /recommended/ — a guess about which slice of org |
| |
| 520 | matters. Phase 0 replaces both halves of that guess with a measurement: an audit that asks |
| |
| 521 | what a real corpus actually uses, and an oracle that asks whether we render it the way |
| |
| 522 | Emacs does. |
| |
| 523 | |
66 | |
| 524 | The audit runs against any corpus — point it at your own notes before trusting this tool |
67 | Each of these is a few lines of config, and each has a page in the guide: |
| 525 | with them. The numbers below come from a 179-file site published today by weblorg, a |
| |
| 526 | wrapper around org's own HTML exporter, which makes it both a realistic workload and a |
| |
| 527 | directly comparable incumbent. With collections configured, orgo now reproduces |
| |
| 528 | *all 182 of that site's URLs*. |
| |
| 529 | |
68 | |
| 530 | #+begin_example |
69 | | A blog index, newest first | [[https://ccleberg.github.io/orgo/guide/03-collections.html][Collections]] | |
| 531 | cargo run -- audit <src-dir> # what does this corpus use, and is it in scope? |
70 | | Tag pages, and an index of tags | [[https://ccleberg.github.io/orgo/guide/03-collections.html][Collections]] | |
| 532 | cargo test --test oracle # how does our HTML differ from Emacs' own export? |
71 | | An RSS feed | [[https://ccleberg.github.io/orgo/guide/03-collections.html][Collections]] | |
| 533 | #+end_example |
72 | | Numbered pages when a list gets long | [[https://ccleberg.github.io/orgo/guide/03-collections.html][Collections]] | |
| |
73 | | Your own design, in ordinary HTML templates | [[https://ccleberg.github.io/orgo/guide/04-templates.html][Templates]] | |
| |
74 | | Drafts that stay unpublished until you say so | [[https://ccleberg.github.io/orgo/guide/06-authoring.html][Authoring]] | |
| |
75 | | A table of contents on long posts | [[https://ccleberg.github.io/orgo/guide/06-authoring.html][Authoring]] | |
| |
76 | | Clean URLs that survive a renamed file | [[https://ccleberg.github.io/orgo/guide/06-authoring.html][Authoring]] | |
| 534 | |
77 | |
| 535 | *** What the audit found |
78 | Rebuilds only touch the pages that actually changed, so saving a post on a site with |
| 536 | *The scope guess was sound.* 99.9% of construct uses in the corpus are in scope. The |
79 | hundreds of them stays instant. |
| 537 | whole out-of-scope tail is 8 uses: four =#+TBLFM:= in a post /about/ org-mode, three |
| |
| 538 | =\name= entities, and one =#+BEGIN_NOTE=. |
| |
| 539 | |
80 | |
| 540 | *=#+SLUG:= was a hole big enough to sink the project.* 178 of 179 files set it, and the |
81 | ** The documentation |
| 541 | published URL comes from it, not from the filename: =2018-11-28-aes-encryption.org= is |
| |
| 542 | served at =blog/aes-encryption.html=. orgo derived output paths from source filenames, |
| |
| 543 | so *169 of 179 pages would have been published at the wrong URL* — every inbound link and |
| |
| 544 | every search result, broken, by a tool that reported a clean build. Output paths now come |
| |
| 545 | from =#+SLUG:= when present ([[file:src/util.rs][=util::output_path=]]); slugs are sanitized so an |
| |
| 546 | author-supplied =../../etc/x= cannot escape the output directory, and two pages claiming one |
| |
| 547 | URL is a build error rather than a silently dropped page. Building the real corpus now |
| |
| 548 | reproduces all 179 of the live site's URLs exactly. |
| |
| 549 | |
82 | |
| 550 | *Some machinery is speculative.* The corpus contains no =id:=, =#custom-id= or =*Heading= |
83 | https://ccleberg.github.io/orgo/ |
| 551 | links at all — its cross-page links are hand-written relative URLs. The INDEX/RESOLVE |
| |
| 552 | symbol table that v0.2 was built around is, against this corpus, unexercised. |
| |
| 553 | |
84 | |
| 554 | *An audit can lie too.* The first run reported 23 uses of a custom TODO keyword sequence. |
85 | | [[https://ccleberg.github.io/orgo/quickstart.html][Quick start]] | A working site in two commands, then your own writing, then your own design. | |
| 555 | All 23 were false: the detector read the leading word of =* CSS Variables= as the keyword |
86 | | [[https://ccleberg.github.io/orgo/install.html][Install]] | Getting the binary, and running it without installing anything. | |
| 556 | =CSS=. The corpus defines no =#+TODO:= sequences at all, so the true count was zero. The |
87 | | [[https://ccleberg.github.io/orgo/guide/01-cli.html][Commands]] | Every command and flag, and what each is for. | |
| 557 | detector now matches conventional keyword names only — a tool that overstates a gap argues |
88 | | [[https://ccleberg.github.io/orgo/guide/02-configuration.html][Configuration]] | Every setting in =orgo.toml=, what it changes, and what it costs. | |
| 558 | for work nobody needs. |
89 | | [[https://ccleberg.github.io/orgo/guide/03-collections.html][Collections]] | Blog indexes, tag pages, pagination and RSS feeds. | |
| |
90 | | [[https://ccleberg.github.io/orgo/guide/04-templates.html][Templates]] | Layouts, and every variable a template can use. | |
| |
91 | | [[https://ccleberg.github.io/orgo/guide/05-org-support.html][Org support]] | Which org syntax is handled, which is not, and how the rest degrades. | |
| |
92 | | [[https://ccleberg.github.io/orgo/guide/06-authoring.html][Authoring]] | URLs, drafts, excerpts, tables of contents. | |
| |
93 | | [[https://ccleberg.github.io/orgo/guide/07-incremental.html][Incremental builds]] | How it decides what to rebuild. | |
| |
94 | | [[https://ccleberg.github.io/orgo/guide/08-workflow.html][Watching and serving]] | The write-save-see loop. | |
| |
95 | | [[https://ccleberg.github.io/orgo/guide/09-auditing.html][Auditing]] | Reading a corpus before trusting a tool with it. | |
| |
96 | | [[https://ccleberg.github.io/orgo/guide/10-deploying.html][Deploying]] | Producing a production build, and putting it somewhere. | |
| 559 | |
97 | |
| 560 | *** What the oracle found |
98 | ** Building from a checkout |
| 561 | =tests/oracle.rs= exports each fixture with org's own exporter via =emacs --batch=, reduces |
| |
| 562 | both sides to a semantic skeleton (element opens, closes and text, with layout =div=s, |
| |
| 563 | inline =span=s and all attributes but =href=/=src= dropped), and *snapshots the |
| |
| 564 | disagreement*. Snapshotting rather than asserting is deliberate: a checked-in divergence |
| |
| 565 | report gets reviewed and shows up as a diff, where a permanently red test gets ignored. |
| |
| 566 | Three invariants are asserted outright, and all three hold — heading structure, list |
| |
| 567 | nesting, and source-block text match Emacs exactly. |
| |
| 568 | |
99 | |
| 569 | *No bugs in orgo.* Every remaining divergence is a deliberate choice to emit better |
100 | #+begin_src sh |
| 570 | HTML than org does: |
101 | cargo test # includes a differential check against Emacs, when present |
| 571 | |
102 | cargo run -- serve docs -o docs/_site # read the documentation locally |
| 572 | | | orgo | Emacs | why | |
103 | #+end_src |
| 573 | |—————--+—————————+—————————-+—————————————| |
| |
| 574 | | emphasis | =<em>=/=<strong>= | =<i>=/=<b>= | semantic, not presentational | |
| |
| 575 | | captioned image | =<figure>=/=<figcaption>= | =<p>= + ="Figure 1: …"= | real figure semantics | |
| |
| 576 | | timestamp | =<time datetime="…">= | literal =<2024-01-15 Mon>= | machine-readable | |
| |
| 577 | | footnotes | =<section><ol>= | =<h2>Footnotes:</h2>= | a list of notes is a list | |
| |
| 578 | | heading anchor | slug of the text | =org1a2b3c4= | stable, and what the live site serves | |
| |
| 579 | | code | =<pre><code>= | =<pre>= | the HTML5 idiom | |
| |
| 580 | |
| |
| 581 | One genuine semantic difference: org treats a single blank line between a =1.= list and a |
| |
| 582 | =-= list as /one/ list and keeps the first item's bullet type, while we start a second list. |
| |
| 583 | We keep ours, on measurement rather than taste — the pattern occurs *zero* times in the |
| |
| 584 | corpus, so matching an org quirk would buy nothing and cost the more obvious reading. |
| |
| 585 | |
| |
| 586 | *The oracle's best catch was three bugs in itself.* Naive normalization reported code as |
| |
| 587 | corrupted (it trimmed each of syntect's per-token text runs, turning =def greet= into |
| |
| 588 | =defgreet=) and reported blocks at 36% agreement (syntect's spans flooded the diff). Both |
| |
| 589 | were measurement artifacts. A differential harness is a piece of software like any other, |
| |
| 590 | and the first divergences it reports are usually its own. |
| |
| 591 | |
| |
| 592 | ** Phase 7: hardening |
| |
| 593 | *** Parse diagnostics (=file:line: message=) |
| |
| 594 | The parser's contract is that it always returns a document — out-of-scope and malformed |
| |
| 595 | constructs degrade rather than crash. The gap was that they degraded /silently/, and in the |
| |
| 596 | worst cases the degradation is severe: an unterminated =#+BEGIN_SRC= reads the rest of the |
| |
| 597 | file as block content, and an unterminated drawer does the same but renders to nothing, so |
| |
| 598 | one missing line deletes most of a page from a build that reports success. |
| |
| 599 | |
| |
| 600 | =parse= now returns =Document::diagnostics=, each carrying a 1-based source line, and the |
| |
| 601 | build prints them as =file:line: message=. =--strict= turns them (and unresolved links) into |
| |
| 602 | a non-zero exit. Line numbers are threaded as an absolute offset through every nested parse, |
| |
| 603 | so a block inside a list item inside a section still reports its real file line — there is a |
| |
| 604 | test for exactly that, because reconstructed and re-indented nested slices are precisely |
| |
| 605 | where an off-by-N hides. The 179-file corpus produces zero diagnostics. |
| |
| 606 | |
| |
| 607 | *** Parallelism |
| |
| 608 | PARSE, RESOLVE and RENDER/EMIT run under rayon. PARSE is a pure function of one file's bytes |
| |
| 609 | and RESOLVE only reads the shared symbol table, which is what makes both safe to parallelize |
| |
| 610 | at all; INDEX stays sequential. |
| |
| 611 | |
| |
| 612 | | corpus | before | after | speedup | |
| |
| 613 | |————————+——--+——-+———| |
| |
| 614 | | 179 files (real) | 0.23s | 0.07s | 3.3× | |
| |
| 615 | | 1,790 files (10× copy) | 3.98s | 0.82s | 4.9× | |
| |
| 616 | |
| |
| 617 | Measured on 12 cores. =RAYON_NUM_THREADS=1= reproduces the old 3.98s exactly, so the gain is |
| |
| 618 | parallelism rather than incidental change, and the output is byte-identical to the sequential |
| |
| 619 | build across the whole corpus. |
| |
| 620 | |
| |
| 621 | *Parallelism must not be observable in the result.* =par_iter().collect()= preserves input |
| |
| 622 | order, so the emitted bytes are unaffected — but the build /report/ is the fragile half: |
| |
| 623 | pushing to =rendered=/=skipped= from inside the parallel pass would order them by thread |
| |
| 624 | scheduling, producing a non-deterministic report over a deterministic site. The parallel pass |
| |
| 625 | therefore returns only what was written, and the report is assembled sequentially afterwards. |
| |
| 626 | =parallel_builds_are_deterministic_in_output_and_report_order= holds that line, and it was |
| |
| 627 | verified by reintroducing the bug and watching it fail. |
| |
| 628 | |
| |
| 629 | *** The real scaling limit was not the CPU |
| |
| 630 | Going 10× on corpus size cost 17× in time, which parallelism improves without fixing: the |
| |
| 631 | cause was the nav bar listing *every* page, so an /n/-page site emitted /n/² nav links. At |
| |
| 632 | 1,790 pages each page carried 1,799 links and the output was 284 MB, against 5.5 MB for the |
| |
| 633 | 179-page corpus — 52× the bytes for 10× the input. |
| |
| 634 | |
| |
| 635 | The nav is now built from *top-level pages only* ([[file:src/site.rs][=is_top_level=]]): a nav is a |
| |
| 636 | map of the site's top level, not an index of its contents, and section pages reach their |
| |
| 637 | siblings through that section's landing page. Nav size becomes a function of the top level |
| |
| 638 | rather than of the corpus, and the quadratic disappears. |
| |
| 639 | |
| |
| 640 | | 1,790-page corpus (6 top-level pages) | before | after | |
| |
| 641 | |—————————————+——--+——-| |
| |
| 642 | | full build | 0.82s | 0.39s | |
| |
| 643 | | total output | 284 MB | 34 MB | |
| |
| 644 | | nav links per page | 1,799 | 6 | |
| |
| 645 | |
| |
| 646 | Scaling is now linear: 179 pages in 0.07s and 1,796 in 0.39s, where the small case is mostly |
| |
| 647 | the fixed cost of loading syntect's syntax definitions. |
| |
| 648 | |
| |
| 649 | The same rule sharpened the incremental build, which is the larger win. The site-structure |
| |
| 650 | hash — the thing that forces a global re-render — now covers only the pages that appear in |
| |
| 651 | the nav, because those are the only ones whose title or URL affects another page. *Adding a |
| |
| 652 | blog post used to re-render the entire site; now it renders one page.* A top-level page's |
| |
| 653 | title still invalidates everything, correctly, since every page displays it. |
| |
| 654 | |
| |
| 655 | *Trade-off worth knowing:* on a site whose sections live in subdirectories, only genuinely |
| |
| 656 | root-level pages appear — a site keeping its landing pages at =salary/index.org= and friends |
| |
| 657 | gets a one-entry nav. That is what =nav.mode = "explicit"= is for: list the pages you want, |
| |
| 658 | in the order you want them. |
| |
| 659 | |
| |
| 660 | *From v0.1 (core subset):* headings with nesting and anchors (every heading is now |
| |
| 661 | anchored — =:CUSTOM_ID:=/=:ID:= else a slug of its text) and trailing tags; paragraphs; |
| |
| 662 | plain lists (unordered + ordered) with checkboxes; source blocks; inline markup (=*bold*=, |
| |
| 663 | =/italic/=, =_underline_=, =+strike+=, ==verbatim==, =~code~=); links and bare URLs. |
| |
| 664 | |
| |
| 665 | ** Compatibility |
| |
| 666 | Versions mean something as of 1.0. The *stable surface* — changing incompatibly requires |
| |
| 667 | a major version — is what you actually build a site against: |
| |
| 668 | |
| |
| 669 | | Stable | Detail | |
| |
| 670 | |——————+——————————————————————————————————————————————-| |
| |
| 671 | | =orgo.toml= keys | Names, types and meaning. New keys are minor releases; removing one is major. | |
| |
| 672 | | Template context | =page=, =site=, =nav=, =root=, =pages=, =group=, =groups=, =paginator=, =stylesheet=, and the =absolute= / =rfc822= / =truncate= filters. | |
| |
| 673 | | CLI | Command names, flags, and exit codes. | |
| |
| 674 | | URLs | How a source path becomes an output path, including =#+SLUG:=. A generator that moves your URLs breaks every link anyone has to you. | |
| |
| 675 | |
| |
| 676 | Explicitly *not stable*, so that the above can be: |
| |
| 677 | |
| |
| 678 | - *The incremental cache.* Versioned, discarded on mismatch, never a correctness |
| |
| 679 | dependency. It changes whenever it needs to, in any release. |
| |
| 680 | - *Rendered HTML details.* orgo tracks what Emacs exports from the same file, and |
| |
| 681 | closing a gap changes markup. Changes that affect output are called out in |
| |
| 682 | [[file:CHANGELOG.org][CHANGELOG.org]] — the class names the documentation names (=post-list=, |
| |
| 683 | =figure-number=, =section-number-N=, =footnote-ref=) are the ones to write CSS against. |
| |
| 684 | - *The Rust API.* The crate is published so the binary can be installed with |
| |
| 685 | =cargo install=; the library exists to serve it, and its types move as the tool does. |
| |
| 686 | |
| |
| 687 | The *MSRV is 1.88*, checked in CI on every change. orgo's own code compiles on |
| |
| 688 | 1.82; the floor comes from dependencies. Raising it is a minor version, never a patch. |
| |
| 689 | |
| |
| 690 | ** Dependencies |
| |
| 691 | Parser is hand-written recursive descent (not =nom=/=chumsky=/=pest= — org is |
| |
| 692 | line-oriented and context-sensitive, not clean CFG). Key crates: =syntect= (syntax |
| |
| 693 | highlighting, behind a =Highlighter= trait so tree-sitter can be swapped in later), |
| |
| 694 | =minijinja= (runtime templates), =blake3= (content/cache hashing), =rayon= (parallel |
| |
| 695 | PARSE/RESOLVE/RENDER), =notify= (filesystem events for =watch=), =tiny_http= (the =serve= |
| |
| 696 | development server), =toml= (config), =chrono=, =camino=, =walkdir=, =clap=, =anyhow=/=thiserror=. |
| |
| 697 | =insta= for snapshot tests, and =emacs --batch= — optional, and only for the oracle. |
| |
| 698 | |
| |
| 699 | ** Build & test |
| |
| 700 | #+begin_example |
| |
| 701 | cargo build |
| |
| 702 | cargo test # 191 tests |
| |
| 703 | cargo run -- init my-site # scaffold a new site |
| |
| 704 | cargo run -- build fixtures/minimal.org -o minimal.html # single file |
| |
| 705 | cargo run -- build fixtures/site -o _site # whole site (incremental) |
| |
| 706 | cargo run -- audit fixtures/site # corpus audit (Phase 0) |
| |
| 707 | cargo run -- build fixtures/site -o _site --no-cache # force a full rebuild |
| |
| 708 | cargo run -- watch fixtures/site -o _site # rebuild on filesystem events |
| |
| 709 | cargo run -- serve fixtures/site -o _site # ... and serve with live reload |
| |
| 710 | cargo run -- clean _site # remove output + cache |
| |
| 711 | #+end_example |
| |
| 712 | |
| |
| 713 | A second =build= of an unchanged site re-renders nothing; editing a page re-renders only |
| |
| 714 | that page and the pages that link into it (watch the =rendered=/=cached= counts). |
| |
| 715 | |
104 | |
| 716 | A build emits =syntax.css= next to its output (the highlighter emits CSS classes, so the |
105 | ** Licence |
| 717 | stylesheet has to come with them) and every page links it. |
| |
| 718 | |
106 | |
| 719 | =fixtures/= holds tiny =.org= samples: the core ones (=minimal.org=, =core.org=, |
107 | [[file:LICENSE][0BSD]]. Do what you like with it. |
| 720 | =elements.org=, =table.org=, =footnote.org=), one per v1 construct group (=headings.org=, |
| |
| 721 | =lists.org=, =blocks.org=, =timestamps.org=, =images.org=), the scope guardrail |
| |
| 722 | (=outofscope.org=), and a linked multi-file site under =fixtures/site/= (=index.org=, |
| |
| 723 | =guide.org=, =about.org= + a =style.css= asset). The real corpus (golden files derived from |
| |
| 724 | actual documents) lands in Phase 0. =cargo test= runs =insta= snapshots of the element tree |
| |
| 725 | and rendered HTML for each fixture, the two templated site pages (proving cross-file link |
| |
| 726 | resolution), and the incremental gates. |
| |