Oregami
Repositories/oxedyne/fe2o3

oxedyne/fe2o3/fe2o3_steel/src/srv/publish/mod.rs

55.5 KiB, 416 runs

created by r1870400018:14356, which is this file's identity for as long as the history lasts, whatever it is later renamed to

download · who wrote it · its history

1//! Publishing the prose a site holds.
2//!
3//! A directory of Markdown becomes posts a site serves under a name of its own: real pages at real
4//! URLs, a feed, and a JSON list for a page that would rather render them itself.
5//!
6//! # Why real pages matter here
7//!
8//! A post that exists only inside a page's JavaScript cannot be linked to, cannot be found, and
9//! unfurls to nothing when it is pasted anywhere. Self-hosting prose in order to be read, and then
10//! serving it in a form only a browser running scripts can see, gives up the thing it was for. So the
11//! canonical form of a post is a page: a URL, HTML in the first response, and the tags a card is built
12//! from.
13//!
14//! [`json`] serves the same posts for a page that wants them inline. It is the convenience;
15//! [`page`] is the point.
16//!
17//! # What a file says
18//!
19//! A file names itself. `2026-07-17-on-rent.md` is the post `on-rent`, dated `2026-07-17`; a name
20//! without a leading date is a post without one. The title is the document's own most prominent
21//! heading, and the slug where it has no heading -- so a post says its title once, in the prose, and
22//! nowhere else.
23//!
24//! There is no front matter, deliberately. A metadata block is a second little language to learn, to
25//! parse and to get wrong, and everything above is already in the file or its name.
26//!
27//! # Where the posts live
28//!
29//! In a directory, for now, and in the vhost's database later. This module sits in the server process
30//! and is handed the database already, so the move is a store behind [`read_all`] rather than a
31//! rearrangement. A directory of Markdown is not a stand-in meanwhile: it is a real way to write, and
32//! the file is the source either way.
33//!
34//! [Written with AI entirely](https://need2know.ai/entirely-ai/code)\
35//! Anthropic Claude
36
37pub mod ai;
38pub mod comment;
39pub mod declare;
40pub mod dest;
41pub mod feed;
42pub mod json;
43pub mod page;
44pub mod send;
45pub mod store;
46pub mod subscribe;
47
48use oxedyne_fe2o3_core::prelude::*;
49use oxedyne_fe2o3_datime::time::{
50 CalClock,
51 CalClockZone,
52};
53use oxedyne_fe2o3_iop_crypto::enc::Encrypter;
54use oxedyne_fe2o3_iop_db::api::Database;
55use oxedyne_fe2o3_iop_hash::api::Hasher;
56use oxedyne_fe2o3_jdat::{
57 prelude::*,
58 id::NumIdDat,
59};
60use oxedyne_fe2o3_text::doc::{
61 Block,
62 Doc,
63 djot,
64 html,
65 markdown,
66 text_of,
67};
68
69use std::{
70 fs,
71 path::Path,
72 sync::{
73 Arc,
74 RwLock,
75 },
76};
77
78
79pub const DIR_DEFAULT: &str = "./www/public/content/posts";
80
81pub const PATH_DEFAULT: &str = "/posts";
82
83// The categories a site starts with where its config names none: a small, thematic taxonomy of
84// the kind a general blog keeps, the defined counterpart to the free-form tags. An operator
85// narrows, renames or empties this in one place. In config order, which is the order the filter
86// draws them.
87pub const CATEGORIES_DEFAULT: [&str; 6] =
88 ["Personal", "Technical", "Ideas", "Reviews", "Projects", "Announcements"];
89
90const EXT: &str = "md";
91
92const EXCERPT_LEN: usize = 200; // characters of a post's opening, for a card and a feed
93
94// Words a minute, for a post's reading time. Two hundred is the low end of the range measured for
95// silent reading of English prose, so the estimate errs towards telling a reader a piece is longer
96// than they will find it.
97pub const READ_WPM: usize = 200;
98
99/// How long a post of the given length takes to read, in whole minutes.
100///
101/// The one definition of reading time, so the badge a page shows and the slider a filter offers count
102/// the same way. Rounded up, and never below one: "0 min read" tells a reader nothing they wanted to
103/// know, and a slider whose floor is zero has a dead notch.
104pub fn read_mins(words: usize) -> usize {
105 words.div_ceil(READ_WPM).max(1)
106}
107
108
109/// Where a vhost's posts are kept.
110///
111/// Both produce the same [`Post`], so nothing downstream of the read knows which it was.
112#[derive(Clone, Copy, Debug, Default, Eq, PartialEq)]
113pub enum Source {
114 #[default]
115 Dir, // a directory of Markdown: what an editor writes, and where prose already is
116 // The vhost's database. What the composer writes, and the only source a draft can live in -- a
117 // directory on a server holds no drafts, since putting one there would publish it.
118 Store,
119}
120
121impl Source {
122
123 pub fn of(s: &str) -> Outcome<Self> {
124 match s {
125 "dir" => Ok(Self::Dir),
126 "store" => Ok(Self::Store),
127 // Not a lenient default: a site that meant `store` and typed `stor` would silently serve a
128 // directory instead, and discover it by noticing the wrong prose on its own front page.
129 _ => Err(err!(
130 "PublishConfig: 'source' must be 'dir' or 'store', not '{}'.", s;
131 Invalid, Input)),
132 }
133 }
134}
135
136/// A vhost's published prose: where the posts are, where they are served, and what the site calls
137/// them.
138///
139/// Absent from a vhost, the vhost publishes nothing and none of these paths are served. Every field
140/// has a default, so a config saying `"publish": {}` publishes an empty directory rather than failing
141/// to load -- the shape of a config is not the place to discover a typo in a path.
142#[derive(Clone, Debug, Default, Eq, PartialEq)]
143pub struct PublishConfig {
144 pub path: String, // URL prefix, without a trailing slash
145 // Directory holding the Markdown, absolute or relative to the app root. Read when the source is
146 // a directory, and the place an import reads from when it is the store.
147 pub dir: String,
148 pub source: Source,
149 pub title: String, // what the site calls its posts: index heading and feed name
150 pub site_name: String, // the site's own name, for the card a shared link makes
151 // The site's canonical origin, e.g. `https://example.com`, without a trailing slash. Needed
152 // rather than derived from the request, because a card's URLs and a feed's must be absolute,
153 // and a `Host` header is the client's word for where it thinks it is.
154 pub base_url: String,
155 // Stylesheets a page links, in order. A page carries no styling of its own: what prose should
156 // look like is the site's business, not the server's.
157 pub css: Vec<String>,
158 // The remotes this site is configured to post to, and the credentials to reach them. Empty
159 // where the site publishes only to its own pages, which is the default.
160 pub creds: send::DestCreds,
161 // The least seconds between two comments from one sender; `0` turns the interval off.
162 // Operational policy, not a constant, because the thing being counted is an address and an
163 // address is not a person: a household, an office and a university share one. A site whose
164 // readers are behind shared addresses wants this low or off; one being flooded wants it high.
165 // It also does nothing useful where the server sits behind a proxy that does not pass the
166 // client's address through, since then every reader is one address.
167 pub comment_rate_secs: u64,
168 pub comment_rate_hourly: u32, // comments one sender may leave in an hour; `0` is off
169 // The least seconds between two newsletter sign-ups from one address; `0` turns the interval
170 // off. A sign-up is not a comment, and is worth limiting harder: a comment costs a row, while
171 // a sign-up costs a piece of outbound mail to an address the sender chose, which is the whole
172 // of a mail-bombing tool and lands the cost on this host's sending reputation rather than on
173 // its disk. Double opt-in caps the damage at one message per address; it does not cap how many
174 // addresses a script may name.
175 pub subscribe_rate_secs: u64,
176 // How many sign-ups one address may make in an hour; `0` turns the count off. Deliberately
177 // small: a person signs up once, mistypes it, and signs up again, and a handful an hour covers
178 // that and nothing else. An office behind one address that genuinely needs more raises it, the
179 // same operational judgement `comment_rate_secs` documents.
180 pub subscribe_rate_hourly: u32,
181 // Whether this site takes comments on its posts. Off unless a site asks for it: a comment
182 // endpoint is an unauthenticated public write, and turning one on for every site that happens
183 // to publish prose -- which is what a default of `true` would do -- is not a decision this
184 // module gets to make on an operator's behalf. A `publish` block that names nothing takes no
185 // comments and serves no form.
186 pub comments: bool,
187 // The address the newsletter is sent from, e.g. `README <news@oxedyne.com>`. Empty falls back
188 // to the mail configuration's derived default (`news@<mail-domain>`), which is aligned with the
189 // DKIM signing domain so the signature authenticates. A `#[optional]` field: a `publish` block
190 // that names none still loads, and the newsletter takes the default.
191 pub newsletter_from: String,
192 // The categories a post may sit in: the site's defined taxonomy, the checkbox counterpart to
193 // the free-form tags. Drawn as the filter's category row and offered in the composer. An
194 // optional field: a `publish` block naming none takes CATEGORIES_DEFAULT, so a site gets a
195 // sensible set without configuring one, and an operator narrows or renames it in one place.
196 pub categories: Vec<String>,
197 // The author a post carries when its own source names none -- chiefly a directory post, which
198 // has no front matter to hold one. A member's site-login username; empty leaves such a post
199 // unattributed. An optional field.
200 pub default_author: String,
201 // The site's mark, drawn at the top of every page: a URL the served pages can reach, e.g.
202 // `/assets/logo.svg`. Empty draws the title as a word instead, which is what a site that
203 // configures no mark has always had. An optional field.
204 pub logo: String,
205 // Where the mark at the top of a page leads: the site's own front page, e.g.
206 // `https://example.com` or `/`. Empty leads back to the index of the posts, which is right for
207 // a site whose posts are all there is of it, and wrong for a blog that is one part of a larger
208 // site. An optional field.
209 pub home: String,
210 // What this site declares about its use of AI, and where the scheme it declares under lives. A
211 // site configuring none declares nothing and draws no marks, which is not the same as declaring
212 // that it used none. An optional field.
213 pub declare: declare::DeclareConfig,
214}
215
216impl PublishConfig {
217
218 pub fn from_datmap(m: &DaticleMap) -> Outcome<Self> {
219 let get_str = |key: &str, default: &str| -> Outcome<String> {
220 match m.get(&dat!(key)) {
221 Some(Dat::Str(s)) => Ok(s.clone()),
222 None => Ok(default.to_string()),
223 _ => Err(err!(
224 "PublishConfig: '{}' must be a string.", key;
225 Invalid, Input, Mismatch)),
226 }
227 };
228
229 // A count, however narrowly the grammar happened to type it. A bare `0` is not a `u64` to
230 // the decoder -- it is the smallest thing that holds it -- so a match on one variant refuses
231 // exactly the value an operator is most likely to write.
232 let get_count = |key: &str, default: u64| -> Outcome<u64> {
233 match m.get(&dat!(key)) {
234 None => Ok(default),
235 Some(Dat::U8(n)) => Ok(*n as u64),
236 Some(Dat::U16(n)) => Ok(*n as u64),
237 Some(Dat::U32(n)) => Ok(*n as u64),
238 Some(Dat::U64(n)) => Ok(*n),
239 Some(Dat::I8(n)) if *n >= 0 => Ok(*n as u64),
240 Some(Dat::I16(n)) if *n >= 0 => Ok(*n as u64),
241 Some(Dat::I32(n)) if *n >= 0 => Ok(*n as u64),
242 Some(Dat::I64(n)) if *n >= 0 => Ok(*n as u64),
243 _ => Err(err!(
244 "PublishConfig: '{}' must be a count of zero or more.", key;
245 Invalid, Input, Mismatch)),
246 }
247 };
248
249 let mut path = res!(get_str("path", PATH_DEFAULT));
250 // A trailing slash would make every route below double it, and a prefix that is not rooted
251 // would match nothing. Correct both rather than serve something subtly wrong.
252 while path.ends_with('/') {
253 path.pop();
254 }
255 if !path.starts_with('/') {
256 path.insert(0, '/');
257 }
258
259 let mut base_url = res!(get_str("base_url", ""));
260 while base_url.ends_with('/') {
261 base_url.pop();
262 }
263
264 // A list and a vek are both written as a list of strings and both mean one, so both are read.
265 // The rest of this config grammar accepts either, and a stylesheet list is no place to
266 // discover that it does not.
267 let strings = |items: &[Dat]| -> Outcome<Vec<String>> {
268 let mut out = Vec::new();
269 for item in items {
270 match item {
271 Dat::Str(s) => out.push(s.clone()),
272 _ => return Err(err!(
273 "PublishConfig: every 'css' entry must be a string.";
274 Invalid, Input, Mismatch)),
275 }
276 }
277 Ok(out)
278 };
279 let css = match m.get(&dat!("css")) {
280 Some(Dat::List(list)) => res!(strings(list)),
281 Some(Dat::Vek(vek)) => res!(strings(vek.as_slice())),
282 None => Vec::new(),
283 _ => return Err(err!(
284 "PublishConfig: 'css' must be a list of strings.";
285 Invalid, Input, Mismatch)),
286 };
287
288 let source = match m.get(&dat!("source")) {
289 Some(Dat::Str(s)) => res!(Source::of(s)),
290 None => Source::default(),
291 _ => return Err(err!(
292 "PublishConfig: 'source' must be a string.";
293 Invalid, Input, Mismatch)),
294 };
295
296 let creds = match m.get(&dat!("destinations")) {
297 Some(Dat::Map(dm)) => res!(send::DestCreds::from_datmap(dm)),
298 None => send::DestCreds::default(),
299 _ => return Err(err!(
300 "PublishConfig: 'destinations' must be a map.";
301 Invalid, Input, Mismatch)),
302 };
303
304 let declare = match m.get(&dat!("declare")) {
305 Some(Dat::Map(dm)) => res!(declare::DeclareConfig::from_datmap(dm)),
306 None => declare::DeclareConfig::default(),
307 _ => return Err(err!(
308 "PublishConfig: 'declare' must be a map.";
309 Invalid, Input, Mismatch)),
310 };
311
312 Ok(Self {
313 path,
314 dir: res!(get_str("dir", DIR_DEFAULT)),
315 source,
316 title: res!(get_str("title", "Posts")),
317 site_name: res!(get_str("site_name", "")),
318 base_url,
319 css,
320 creds,
321 // A site that names no From takes the mail default; the field is optional, so an existing
322 // `publish` block that predates the newsletter still loads.
323 comment_rate_secs: res!(get_count("comment_rate_secs", 30)),
324 comment_rate_hourly: res!(get_count("comment_rate_hourly", 10)) as u32,
325 // Optional, like the rest here: a `publish` block written before these fields existed
326 // loads and takes the defaults, which limit rather than not.
327 subscribe_rate_secs: res!(get_count("subscribe_rate_secs", 60)),
328 subscribe_rate_hourly: res!(get_count("subscribe_rate_hourly", 5)) as u32,
329 comments: match m.get(&dat!("comments")) {
330 Some(Dat::Bool(b)) => *b,
331 None => false,
332 _ => return Err(err!(
333 "PublishConfig: 'comments' must be true or false.";
334 Invalid, Input, Mismatch)),
335 },
336 newsletter_from: res!(get_str("newsletter_from", "")),
337 // The taxonomy, or the built-in set where a site names none. A site that wants no categories
338 // at all writes an empty list, which is distinct from naming none: the first is a deliberate
339 // nothing, the second takes the default.
340 categories: {
341 let cats = match m.get(&dat!("categories")) {
342 Some(Dat::List(list)) => res!(strings(list)),
343 Some(Dat::Vek(vek)) => res!(strings(vek.as_slice())),
344 None => CATEGORIES_DEFAULT.iter().map(|c| c.to_string()).collect(),
345 _ => return Err(err!(
346 "PublishConfig: 'categories' must be a list of strings.";
347 Invalid, Input, Mismatch)),
348 };
349 // A category is joined into a comma-separated field in the composer and in the filter's
350 // data attribute, so a comma inside a category name would split into two. A space is
351 // fine, and common ("Book reviews"); a comma is refused here rather than left to corrupt
352 // the field silently at a distance.
353 for c in &cats {
354 if c.contains(',') {
355 return Err(err!(
356 "PublishConfig: a category name may not contain a comma: '{}'.", c;
357 Invalid, Input));
358 }
359 }
360 cats
361 },
362 default_author: res!(get_str("default_author", "")),
363 logo: res!(get_str("logo", "")),
364 home: res!(get_str("home", "")),
365 declare,
366 })
367 }
368
369 /// Resolves every secret reference the config carries against the app root.
370 ///
371 /// A destination's token or app password is written as an `{env:}` or `{file:}` reference, never in
372 /// the clear, and is resolved once at startup -- the same treatment the SMTP submission password
373 /// gets, and for the same reason: a secret in the config file is a secret in every backup of it.
374 pub fn resolve_secrets(&mut self, root: &std::path::Path) -> Outcome<()> {
375 res!(self.creds.resolve_secrets(root));
376 Ok(())
377 }
378
379 /// Whether a request path belongs to the published prose.
380 ///
381 /// The prefix and what sits under it, and nothing that merely begins with the same letters: a
382 /// site publishing at `/asides` has not thereby claimed `/asides-are-great`.
383 pub fn owns(&self, path: &str) -> bool {
384 path == self.path
385 || (path.starts_with(&self.path)
386 && path.as_bytes().get(self.path.len()) == Some(&b'/'))
387 }
388
389 /// Where a reader who has finished with a subscription page should be sent.
390 ///
391 /// The site's own front door where it names one, and the index of the posts otherwise -- which
392 /// is right for a site whose posts are all there is of it, and is at least somewhere for one
393 /// where they are not.
394 pub fn home_or_index(&self) -> String {
395 if self.home.trim().is_empty() {
396 self.path.clone()
397 } else {
398 self.home.clone()
399 }
400 }
401
402 pub fn url_of(&self, path: &str) -> String {
403 let mut s = self.base_url.clone();
404 s.push_str(path);
405 s
406 }
407
408 pub fn path_of(&self, slug: &str) -> String {
409 let mut s = self.path.clone();
410 s.push('/');
411 s.push_str(slug);
412 s
413 }
414
415 pub fn feed_path(&self) -> String {
416 let mut s = self.path.clone();
417 s.push_str("/feed.xml");
418 s
419 }
420
421 pub fn json_path(&self) -> String {
422 let mut s = self.path.clone();
423 s.push_str("/index.json");
424 s
425 }
426
427 /// The URL path the site's declarations are served at.
428 ///
429 /// Public, and deliberately apart from [`json_path`](Self::json_path): a front page drawing a mark
430 /// beside a book wants a handful of levels, and would otherwise fetch every post and its rendered
431 /// prose to find them.
432 pub fn declare_path(&self) -> String {
433 let mut s = self.path.clone();
434 s.push_str("/declare.json");
435 s
436 }
437
438 pub fn subscribe_path(&self) -> String {
439 let mut s = self.path.clone();
440 s.push_str("/subscribe");
441 s
442 }
443
444 pub fn comment_edit_path(&self, slug: &str) -> String {
445 let mut s = self.path.clone();
446 s.push('/');
447 s.push_str(slug);
448 s.push_str("/comment/edit");
449 s
450 }
451
452 pub fn comment_edit_slug<'a>(&self, path: &'a str) -> Option<&'a str> {
453 let rest = path.strip_prefix(&self.path)?.strip_prefix('/')?;
454 let slug = rest.strip_suffix("/comment/edit")?;
455 if slug.is_empty() || !valid_slug(slug) {
456 return None;
457 }
458 Some(slug)
459 }
460
461 pub fn comment_preview_path(&self, slug: &str) -> String {
462 let mut s = self.path.clone();
463 s.push('/');
464 s.push_str(slug);
465 s.push_str("/comment/preview");
466 s
467 }
468
469 pub fn comment_preview_slug<'a>(&self, path: &'a str) -> Option<&'a str> {
470 let rest = path.strip_prefix(&self.path)?.strip_prefix('/')?;
471 let slug = rest.strip_suffix("/comment/preview")?;
472 if slug.is_empty() || !valid_slug(slug) {
473 return None;
474 }
475 Some(slug)
476 }
477
478 /// The URL path the comment form's script is served at.
479 ///
480 /// A file rather than an inline block, so a site can run a Content-Security-Policy without
481 /// `unsafe-inline`. An inline script forces every page that carries it to allow inline scripts,
482 /// which switches off the one layer that would contain a mistake in the render policy -- and the
483 /// same untrusted prose is rendered into the admin console, where a mistake would be worst.
484 pub fn comment_js_path(&self) -> String {
485 let mut s = self.path.clone();
486 s.push_str("/comments.js");
487 s
488 }
489
490 /// The URL path the index filter's script is served at. A file rather than an inline block, so a
491 /// site can run a Content-Security-Policy that forbids inline script and still get the filter.
492 pub fn filter_js_path(&self) -> String {
493 let mut s = self.path.clone();
494 s.push_str("/filter.js");
495 s
496 }
497
498 pub fn avatar_prefix(&self) -> String {
499 let mut s = self.path.clone();
500 s.push_str("/avatar/");
501 s
502 }
503
504 /// The URL a member's uploaded picture is served at.
505 ///
506 /// The path a profile stores once a member uploads one, so a byline points at this module rather
507 /// than at a file somewhere on disk that a deploy could take away.
508 pub fn avatar_path(&self, username: &str) -> String {
509 let mut s = self.avatar_prefix();
510 s.push_str(username);
511 s
512 }
513
514 /// The URL path a comment on a post is posted to.
515 ///
516 /// Under the post's own path rather than a shared endpoint, so which post is being commented on is
517 /// carried by the URL and cannot be swapped for another in the body.
518 pub fn comment_path(&self, slug: &str) -> String {
519 let mut s = self.path.clone();
520 s.push('/');
521 s.push_str(slug);
522 s.push_str("/comment");
523 s
524 }
525
526 pub fn comment_slug<'a>(&self, path: &'a str) -> Option<&'a str> {
527 let rest = path.strip_prefix(&self.path)?.strip_prefix('/')?;
528 let slug = rest.strip_suffix("/comment")?;
529 if slug.is_empty() || !valid_slug(slug) {
530 return None;
531 }
532 Some(slug)
533 }
534
535 /// The URL path a confirmation link points at, carrying the subscriber's token.
536 pub fn confirm_path(&self, token: &str) -> String {
537 let mut s = self.path.clone();
538 s.push_str("/confirm?token=");
539 s.push_str(token);
540 s
541 }
542
543 /// The URL path an unsubscribe link points at, carrying the subscriber's token.
544 pub fn unsubscribe_path(&self, token: &str) -> String {
545 let mut s = self.path.clone();
546 s.push_str("/unsubscribe?token=");
547 s.push_str(token);
548 s
549 }
550
551 /// Whether a request path is one of the subscription endpoints -- the sign-up, the confirm, or the
552 /// unsubscribe -- so the reader dispatch can hand it to [`subscribe`] before it reads the posts.
553 ///
554 /// The bare path only: the query, where the token rides, is matched apart. So `{path}/confirm` is
555 /// this whether or not it carries a token, and the handler answers a missing one with the same
556 /// bad-token page a wrong one gets.
557 pub fn subscription_of(&self, path: &str) -> Option<Subscription> {
558 if path == self.subscribe_path() {
559 Some(Subscription::Subscribe)
560 } else if path == self.confirm_bare_path() {
561 Some(Subscription::Confirm)
562 } else if path == self.unsubscribe_bare_path() {
563 Some(Subscription::Unsubscribe)
564 } else {
565 None
566 }
567 }
568
569 fn confirm_bare_path(&self) -> String {
570 let mut s = self.path.clone();
571 s.push_str("/confirm");
572 s
573 }
574
575 fn unsubscribe_bare_path(&self) -> String {
576 let mut s = self.path.clone();
577 s.push_str("/unsubscribe");
578 s
579 }
580
581}
582
583/// Which subscription endpoint a request named.
584///
585/// A small enum rather than three string comparisons at the call site, so the reader dispatch reads as
586/// a match and a new endpoint is a new arm.
587#[derive(Clone, Copy, Debug, Eq, PartialEq)]
588pub enum Subscription {
589 Subscribe, // `{path}/subscribe`: the sign-up form (GET) and where it posts (POST)
590 Confirm, // `{path}/confirm`: a confirmation link followed
591 Unsubscribe, // `{path}/unsubscribe`: an unsubscribe link followed
592}
593
594
595// The longest a slug may be. A key and a URL both hold one, and neither has a natural limit worth
596// relying on. The number is arbitrary; having one is not.
597pub const SLUG_MAX: usize = 128;
598
599/// Whether a word may be a post's name.
600///
601/// A slug is not decoration: it is pasted into a database key (`publish/post/<slug>`) and into a
602/// URL, so a form's idea of one cannot be taken at its word. A slug carrying a slash would reach
603/// past its own key and name a different post's, or a different thing entirely; one carrying a dot
604/// pair would do the same to a path; one carrying a space or a quote would arrive somewhere as
605/// something other than what was typed.
606///
607/// So the rule is a small alphabet rather than a list of what to reject: letters, digits, hyphen and
608/// underscore. Anything a list of forbidden characters missed would be allowed by default, and the
609/// thing about that mistake is that it does not announce itself.
610pub fn valid_slug(s: &str) -> bool {
611 !s.is_empty()
612 && s.len() <= SLUG_MAX
613 && s.bytes().all(|b| b.is_ascii_alphanumeric() || b == b'-' || b == b'_')
614}
615
616// The marks in a user-agent that name a thing which is not a reader. A substring list is a poor
617// way to identify a browser and an adequate way to discard the obvious machines, which is all this
618// is for. It will miss a crawler that lies, and that is tolerable: a tally is a rough shape, not a
619// headcount, and a bot pretending to be Chrome will not be caught by any list at all.
620const BOT_MARKS: &[&str] = &[
621 "bot", "crawl", "spider", "slurp", "archiver", "curl", "wget", "python-requests",
622 "headlesschrome", "facebookexternalhit", "embedly", "preview", "monitor", "uptime",
623 "scrapy", "feedfetcher", "pingdom", "lighthouse", "http-client",
624];
625
626/// Whether a user-agent names something that is not a person reading.
627///
628/// Lowercased before the comparison, because a user-agent's capitalisation is the sender's choice.
629/// An absent or empty user-agent counts as a bot: every real browser sends one, and a request with
630/// none is a script that did not bother.
631pub fn looks_automated(ua: Option<&str>) -> bool {
632 let ua = match ua {
633 Some(s) if !s.trim().is_empty() => s.to_lowercase(),
634 _ => return true,
635 };
636 BOT_MARKS.iter().any(|m| ua.contains(m))
637}
638
639/// Whether a request for a post should add one to its tally.
640///
641/// Three exclusions, and each is here for a reason worth keeping:
642///
643/// - **A request carrying a management session is the author.** A count that climbs while its author
644/// re-reads their own draft measures the author's attention, not a reader's, and is worse than no
645/// count because it looks like one.
646/// - **A request from an obvious machine is not a read.** See [`looks_automated`].
647/// - **A `HEAD` asked for no prose.** It is how a monitor checks the site is up and how a chat client
648/// fetches a link preview, several times an hour and forever. Counted, the tally would measure the
649/// monitor.
650///
651/// There is deliberately nothing here about *who* the reader is: no identifier is derived, stored or
652/// compared, so two reads by one person count twice and the site never learns they were one person.
653/// That is the trade this counter makes on purpose -- it is a tally of readings, not of readers.
654pub fn counts_as_read(has_manage_session: bool, user_agent: Option<&str>, head_only: bool) -> bool {
655 !head_only && !has_manage_session && !looks_automated(user_agent)
656}
657
658// The longest a tag may be, once normalised. A tag is a facet in a URL and a word on a card,
659// neither with a natural limit worth relying on. The number is arbitrary; having one is not.
660pub const TAG_MAX: usize = 32;
661
662/// Whether a word may be a tag, once normalised.
663///
664/// A small alphabet rather than a reject-list, on the same reasoning as [`valid_slug`]: lowercase
665/// letters, digits and the hyphen, and nothing else. A tag is pasted into a query (`?tag=rust`) and
666/// shown to a reader, so a space, a slash or a capital in one would be a tag that reaches somewhere
667/// as something other than it looks.
668///
669/// The check is against the normalised form, so `valid_tag` normalises first and a caller may pass
670/// what a person typed: `Rust` normalises to `rust` and passes, `a b` normalises to `a b` and does
671/// not -- a space is dropped, not guessed into a hyphen.
672pub fn valid_tag(s: &str) -> bool {
673 let t = normalise_tag(s);
674 !t.is_empty()
675 && t.len() <= TAG_MAX
676 && t.bytes().all(|b| b.is_ascii_lowercase() || b.is_ascii_digit() || b == b'-')
677}
678
679/// A tag as the store keeps it, from a tag as a person typed it.
680///
681/// Trimmed and lowercased, so `Rust` and ` rust ` are the one tag `rust`. A space is left as a
682/// space, deliberately: [`valid_tag`] then drops such a tag rather than this guessing that a space
683/// meant a hyphen and minting a tag the author did not type.
684pub fn normalise_tag(s: &str) -> String {
685 s.trim().to_lowercase()
686}
687
688/// The tags a record keeps, from the comma-separated field a form said.
689///
690/// Split on commas, each normalised and validated, the invalid dropped in silence -- a small
691/// alphabet, not a reject-list, as [`valid_tag`] is -- and deduped keeping first appearance. Empty
692/// or whitespace gives no tags.
693pub fn parse_tags(s: &str) -> Vec<String> {
694 let mut out: Vec<String> = Vec::new();
695 for part in s.split(',') {
696 let t = normalise_tag(part);
697 if !valid_tag(&t) {
698 continue;
699 }
700 if !out.iter().any(|x| x == &t) {
701 out.push(t);
702 }
703 }
704 out
705}
706
707pub const DATE_LEN: usize = 10; // `YYYY-MM-DD`
708pub const STAMP_LEN: usize = 16; // `YYYY-MM-DDTHH:MM`
709
710/// Whether a word may be a post's date.
711///
712/// `YYYY-MM-DD`, the shape [`split_date`] reads out of a filename -- or `YYYY-MM-DDTHH:MM`, which is
713/// the same day with a minute on it. Both are ISO 8601, which is what the feed and `<time>` need:
714/// Atom's dates are ISO, and a date that is not one reaches a reader's feed reader as a malformed
715/// entry rather than as an error anyone here would see. Empty is allowed -- a post without a date is
716/// a post, and says so by carrying none.
717///
718/// # Why a minute, when a filename only ever said a day
719///
720/// Because a day is not an order. Posts sort by date, and two posts of one day fall back to sorting
721/// by slug -- alphabetically, which is to say arbitrarily, and not at all by which was written
722/// first. A directory could not say more than the day, since the date was in the filename and there
723/// is no front matter to put a time in. A record can: its date is a field.
724///
725/// It matters most where most of the writing is. A note is the thing an author writes most, and
726/// several notes in a day is the ordinary case for the form, so the day-only date was weakest
727/// exactly where the module expects the traffic.
728///
729/// A space is accepted where the `T` goes, because that is how a person writes a date;
730/// [`normalise_date`] takes it at the door so one shape reaches the store.
731///
732/// # What is not checked
733///
734/// The calendar. `2026-02-31` passes, and so does `2026-07-17T99:99`. Refusing either means owning a
735/// calendar and a clock, which is the dependency this module does not have and the reason the feed
736/// is Atom rather than RSS. A date that is shaped right and means nothing is the author's typo to
737/// see, and it is visible -- it is printed on the post.
738pub fn valid_date(s: &str) -> bool {
739 if s.is_empty() {
740 return true;
741 }
742 let b = s.as_bytes();
743 if b.len() != DATE_LEN && b.len() != STAMP_LEN {
744 return false;
745 }
746 let day = b[..DATE_LEN].iter().enumerate().all(|(i, c)| {
747 match i {
748 4 | 7 => *c == b'-',
749 _ => c.is_ascii_digit(),
750 }
751 });
752 if !day || b.len() == DATE_LEN {
753 return day;
754 }
755 // `T14:30`, or ` 14:30` from a person who wrote it the way people do.
756 (b[10] == b'T' || b[10] == b' ')
757 && b[11].is_ascii_digit()
758 && b[12].is_ascii_digit()
759 && b[13] == b':'
760 && b[14].is_ascii_digit()
761 && b[15].is_ascii_digit()
762}
763
764/// Today, as a post writes a date: `2026-07-22`, in UTC.
765///
766/// What an author who gave no date meant. A post reaches a feed, and Atom requires every entry to
767/// say when it was updated -- so an undated post is not a post without a date on the page, it is a
768/// post the feed has to invent one for, and the invention was the epoch. That sorts the piece below
769/// everything written since 1970 in every reader in the world, silently. Dating it on the way in
770/// costs the author nothing (the field is filled in for them, and they may change it) and is the
771/// only version of the story where nothing downstream has to guess.
772///
773/// [`CalClock`] does the civil arithmetic, per the fe2o3 calendar. A clock that will not read gives
774/// nothing rather than a wrong day: the caller then keeps the post undated, which is the behaviour
775/// that has always been there.
776pub fn today() -> Option<String> {
777 let secs = match std::time::SystemTime::now()
778 .duration_since(std::time::UNIX_EPOCH)
779 {
780 Ok(d) => d.as_secs() as i64,
781 Err(_) => return None,
782 };
783 let cc = match CalClock::from_unix_timestamp_seconds(secs, CalClockZone::utc()) {
784 Ok(cc) => cc,
785 Err(_) => return None,
786 };
787 Some(fmt!("{:04}-{:02}-{:02}", cc.year(), cc.month(), cc.day()))
788}
789
790/// The date a record keeps, from the date a form said.
791///
792/// One shape in the store, so nothing downstream has to know there were two. ISO puts a `T` between
793/// the day and the hour and people put a space, so the space is taken here rather than handled
794/// everywhere after here.
795pub fn normalise_date(s: &str) -> String {
796 let s = s.trim();
797 if s.len() == STAMP_LEN && s.as_bytes()[10] == b' ' {
798 let mut out = s.to_string();
799 out.replace_range(10..11, "T");
800 return out;
801 }
802 s.to_string()
803}
804
805/// The date a person reads, from the date a record keeps.
806///
807/// The stored form is ISO, so a post dated to the minute carries a `T` in the middle of it. That is
808/// for a machine. A reader gets a space, and the `T` form stays in the `datetime` attribute beside
809/// it, which is the whole point of `<time>` having both.
810pub fn date_text(date: &str) -> String {
811 date.replacen('T', " ", 1)
812}
813
814
815/// The syntax a post is written in.
816///
817/// Both read into the same document tree, so a post's kind of markup is a fact about how it was
818/// typed and nothing a reader ever sees: the page, the feed and the excerpt are made from the tree,
819/// which knows neither. What Djot buys an author over Markdown is the power to name a box (`:::`) and
820/// a style (`{.class}`) in the prose itself, which Markdown has no syntax for. A post says which it
821/// is so the two can sit side by side in one store, each read by its own front-end.
822#[derive(Clone, Copy, Debug, Default, Eq, PartialEq)]
823pub enum Markup {
824 #[default]
825 Markdown, // the form most prose is already written in, and the default
826 Djot, // for prose that wants to name a box or a style
827}
828
829impl Markup {
830
831 /// The word a record stores.
832 pub fn as_str(&self) -> &'static str {
833 match self {
834 Self::Markdown => "markdown",
835 Self::Djot => "djot",
836 }
837 }
838
839 /// The markup a word names. An unknown word is Markdown, the safe default: it is what nearly every
840 /// post is, and a record this version cannot place should read as the ordinary thing rather than
841 /// the exception.
842 pub fn of(s: &str) -> Self {
843 match s {
844 "djot" => Self::Djot,
845 _ => Self::Markdown,
846 }
847 }
848}
849
850/// Whether a post is anybody's business but its author's.
851#[derive(Clone, Copy, Debug, Default, Eq, PartialEq)]
852pub enum PostState {
853 #[default]
854 Draft, // written, not published; served to nobody
855 Live, // published
856}
857
858impl PostState {
859
860 /// The word a record stores.
861 pub fn as_str(&self) -> &'static str {
862 match self {
863 Self::Draft => "draft",
864 Self::Live => "live",
865 }
866 }
867
868 /// The state a word names. **An unknown word is a draft**, deliberately: a record this version
869 /// cannot make sense of should not thereby become published. The safe reading of a state nobody
870 /// understands is that it is not ready.
871 pub fn of(s: &str) -> Self {
872 match s {
873 "live" => Self::Live,
874 _ => Self::Draft,
875 }
876 }
877}
878
879/// An author, as a reader is shown one: the login username a post stores, resolved to a display name,
880/// an avatar and a public handle through the member's profile.
881///
882/// What the filter draws a face from, and what a post's byline reads. Built by
883/// [`store::resolve_authors`] from a username and its profile, so the page layer never touches the
884/// database to draw an author.
885///
886/// # The username never reaches a page
887///
888/// It is the SHA-256 of the member's passphrase. Anything public derived from it -- the whole hash, a
889/// prefix of it, a hash of it -- is a verifier a guess can be tested against, offline and unwatched,
890/// and a page is read by anyone. So [`handle`](Author::handle) is what is drawn and matched on, and
891/// `username` stays on the server side of the resolution: it is here only to pair an author with the
892/// posts that name them.
893#[derive(Clone, Debug, Default, Eq, PartialEq)]
894pub struct Author {
895 // The login username a post stores. Matched against a post's `author` on the server, and never
896 // rendered, emitted or put in a path. See the note above.
897 pub username: String,
898 // The name this author wears in public: the profile's handle, or a name made for the page where
899 // the member has no profile yet. What a page matches a post to an author by.
900 pub handle: String,
901 // The name a reader sees: the profile's name, or `Anonymous` where the profile named none --
902 // never the username, which is not a name and is not for showing.
903 pub name: String,
904 pub avatar: String, // path or URL; empty draws an initial from the name instead
905 // What the author says they write about. Shown above the posts and under a byline; empty shows
906 // nothing, rather than an empty box where a description would be.
907 pub bio: String,
908}
909
910impl Author {
911
912 /// An author from a username and the profile it resolved to, applying the fallbacks: an unnamed
913 /// member is `Anonymous`, an avatarless one lets the reader draw an initial, and one who has never
914 /// saved a profile takes the handle the caller made for the page.
915 ///
916 /// `spare_handle` is used only where the profile holds none -- a member who has written a post but
917 /// never opened their profile. It must not be derived from the username; callers pass a position
918 /// on the page, which identifies an author within one rendering and says nothing anywhere else.
919 pub fn from_profile(username: &str, profile: &store::Profile, spare_handle: &str) -> Self {
920 Self {
921 username: username.to_string(),
922 handle: if profile.handle.is_empty() {
923 spare_handle.to_string()
924 } else {
925 profile.handle.clone()
926 },
927 name: if profile.name.is_empty() {
928 fmt!("Anonymous")
929 } else {
930 profile.name.clone()
931 },
932 avatar: profile.avatar.clone(),
933 bio: profile.bio.clone(),
934 }
935 }
936
937 /// The first character of the display name, upper-cased, or `?` where even that is empty.
938 pub fn initial(&self) -> String {
939 self.name.chars().next()
940 .map(|c| c.to_uppercase().to_string())
941 .unwrap_or_else(|| "?".to_string())
942 }
943}
944
945
946/// One post, as a reader gets it.
947#[derive(Clone, Debug)]
948pub struct Post {
949 pub slug: String, // the post's name in a URL
950 pub title: String, // its own most prominent heading, or the slug where it has none
951 // The member who wrote it, by their site-login username. Empty where none is named. Resolved to
952 // a display name and avatar through the member's profile at the point it is drawn.
953 pub author: String,
954 // The categories the post sits in, from the site's configured set. Empty for an uncategorised
955 // post. The free-form counterpart is `tags`.
956 pub categories: Vec<String>,
957 pub date: Option<String>,
958 pub excerpt: String, // the opening of the prose, flattened, for a card and a feed
959 pub html: String, // the prose, rendered
960 // How many words the prose runs to, for the reading time shown above it. Counted from the tree
961 // rather than the rendered HTML, so no tag or attribute is mistaken for a word.
962 pub words: usize,
963 // Where else the post has been published, as a destination and the permalink it landed at.
964 // Drawn on the page as "also on …", so a reader can follow the post to where the conversation
965 // is. Empty for a post read from a directory, which records no deliveries, and for one sent
966 // nowhere.
967 pub also_on: Vec<(dest::Destination, String)>,
968 // The tags the post carries, normalised. Drawn on the page and the card as tag links, one per
969 // entry in the feed. Empty for a post with none, and for one read from a directory.
970 pub tags: Vec<String>,
971 // How much the writing of this post needed AI, where its author said. Nothing where they said
972 // nothing, which every post written before the field existed reads as, and every post read from
973 // a directory -- a file on disk carries no field to hold one. Undeclared draws no mark: a
974 // declaration nobody made is not this module's to invent.
975 pub ai_level: Option<declare::Level>,
976}
977
978
979/// Reads every post in a directory, newest first.
980///
981/// A file that will not read or will not parse is passed over with a complaint in the log rather than
982/// failing the lot: one broken post should not take the others off the page, and the log is where its
983/// author will look. The directory itself failing is a different thing, and is an error -- an empty
984/// shelf looks like the truth and is not.
985pub fn read_all(dir: &str, id: &str) -> Outcome<Vec<Post>> {
986 let sources = res!(read_sources(dir, id));
987
988 let mut posts = Vec::new();
989 for (stem, source) in sources {
990 let (date, slug) = split_date(&stem);
991 // A file on disk is not a draft, or it would not be on the disk of a server: everything in a
992 // directory is live. A directory of files is Markdown: it is the form prose already exists in,
993 // and a file on disk carries no field to name an author, a category or another markup.
994 match render_source(&source, slug, date, Markup::Markdown) {
995 Ok(p) => posts.push(p),
996 Err(e) => warn!(
997 "{}: posts: skipping '{}', which will not read as Markdown: {}", id, stem, e),
998 }
999 }
1000
1001 // Newest first, and among posts of one date, or of none, by slug. The date descending and the slug
1002 // ascending are compared in opposite directions, so they are compared apart.
1003 posts.sort_by(|a, b| b.date.cmp(&a.date).then_with(|| a.slug.cmp(&b.slug)));
1004 Ok(posts)
1005}
1006
1007/// Every post a vhost publishes, newest first, from wherever it keeps them.
1008///
1009/// The one place the source is chosen, and the only part of this module that touches the database. So
1010/// the genericity the database drags along -- five type parameters, threaded from the web handler --
1011/// stops here, and everything downstream is a plain function over a slice of posts.
1012///
1013/// A store-backed vhost with no database is a misconfiguration rather than an empty site, and says so:
1014/// silently serving nothing would look exactly like a site that has published nothing.
1015pub fn read<
1016 const UIDL: usize,
1017 UID: NumIdDat<UIDL>,
1018 ENC: Encrypter,
1019 KH: Hasher,
1020 DB: Database<UIDL, UID, ENC, KH>,
1021>(
1022 cfg: &PublishConfig,
1023 db: Option<&(Arc<RwLock<DB>>, UID)>,
1024 id: &str,
1025)
1026 -> Outcome<Vec<Post>>
1027{
1028 match cfg.source {
1029 Source::Dir => read_all(&cfg.dir, id),
1030 Source::Store => match db {
1031 Some(db) => store::list(db, id),
1032 None => Err(err!(
1033 "publish: this vhost keeps its posts in the store and has no database \
1034 configured; give the vhost a 'db_dir_rel' or set 'source' to 'dir'.";
1035 Invalid, Input, Missing)),
1036 },
1037 }
1038}
1039
1040/// Every Markdown file in a directory, as its stem and its text.
1041///
1042/// A file that will not read is passed over with a complaint in the log rather than failing the lot:
1043/// one broken post should not take the others off the page, and the log is where its author will look.
1044/// The directory itself failing is a different thing, and is an error -- an empty shelf looks like the
1045/// truth and is not.
1046pub fn read_sources(dir: &str, id: &str) -> Outcome<Vec<(String, String)>> {
1047 let entries = res!(fs::read_dir(Path::new(dir)), IO, File);
1048
1049 let mut out = Vec::new();
1050 for entry in entries {
1051 let entry = res!(entry, IO, File);
1052 let path = entry.path();
1053 if path.extension().map(|e| e != EXT).unwrap_or(true) {
1054 continue;
1055 }
1056 let stem = match path.file_stem().and_then(|s| s.to_str()) {
1057 Some(s) => s.to_string(),
1058 None => {
1059 warn!("{}: posts: skipping a file whose name is not text: {:?}", id, path);
1060 continue;
1061 }
1062 };
1063 match fs::read_to_string(&path) {
1064 Ok(src) => out.push((stem, src)),
1065 Err(e) => warn!("{}: posts: skipping '{}': {}", id, stem, e),
1066 }
1067 }
1068 Ok(out)
1069}
1070
1071/// Markdown, as a reader gets it.
1072///
1073/// The one place prose becomes a [`Post`], so a post from a directory and a post from the store are
1074/// the same post, made the same way. The title is the document's own most prominent heading, and the
1075/// slug where it has none.
1076pub fn render_source(
1077 source: &str,
1078 slug: String,
1079 date: Option<String>,
1080 markup: Markup,
1081)
1082 -> Outcome<Post>
1083{
1084 let doc = res!(parse_markup(source, markup));
1085 let title = doc.top_heading().unwrap_or_else(|| slug.clone());
1086 Ok(Post {
1087 slug,
1088 title,
1089 // A post made from source alone names no author or categories: those are a record's fields,
1090 // which the store threads in where it has one, as it does the tags and deliveries below.
1091 author: String::new(),
1092 categories: Vec::new(),
1093 date,
1094 excerpt: excerpt_of(&doc),
1095 words: doc.word_count(),
1096 html: html::render(&doc),
1097 also_on: Vec::new(),
1098 tags: Vec::new(),
1099 // Nor any declaration: a file on disk has no field to hold one, and a post the store threads
1100 // its record into takes the level from there.
1101 ai_level: None,
1102 })
1103}
1104
1105/// Reads source in the syntax a post names, into the tree.
1106///
1107/// The one place either front-end is chosen. Both produce the same tree, so every caller past this
1108/// knows nothing of which was read.
1109pub fn parse_markup(source: &str, markup: Markup) -> Outcome<Doc> {
1110 match markup {
1111 Markup::Markdown => markdown::parse(source),
1112 Markup::Djot => djot::parse(source),
1113 }
1114}
1115
1116/// The HTML of a run of source, for a preview of prose not yet saved.
1117///
1118/// The same parse and the same render a published post goes through, over source straight from an
1119/// editor rather than from the store -- which is the whole of what a live preview is: the page a
1120/// reader would get, shown to the author as they type, so the box a Djot `:::` makes is seen where it
1121/// will land and not guessed at.
1122pub fn render_html(source: &str, markup: Markup) -> Outcome<String> {
1123 Ok(html::render(&res!(parse_markup(source, markup))))
1124}
1125
1126/// Reads one post by slug from a directory, where it exists.
1127///
1128/// The slug is what a reader put in a URL, so it is checked before it is allowed near a path: a name
1129/// is letters, digits, a dash or an underscore, which leaves nothing to climb out of the directory
1130/// with. A post is found by its slug whatever date its file wears, since the date is not in the URL.
1131pub fn read_one(dir: &str, slug: &str, id: &str) -> Outcome<Option<Post>> {
1132 if slug.is_empty()
1133 || !slug.chars().all(|c| c.is_ascii_alphanumeric() || c == '-' || c == '_')
1134 {
1135 return Ok(None);
1136 }
1137 let posts = res!(read_all(dir, id));
1138 Ok(posts.into_iter().find(|p| p.slug == slug))
1139}
1140
1141/// The opening of a document, flattened to its words, as a card and a feed want it.
1142///
1143/// The first paragraph, cut at a word boundary. Not the first heading: that is the title, and a card
1144/// saying its own title twice says nothing.
1145fn excerpt_of(doc: &Doc) -> String {
1146 let mut s = String::new();
1147 for block in &doc.blocks {
1148 if let Block::Para(content) = block {
1149 s = text_of(content);
1150 break;
1151 }
1152 }
1153 if s.chars().count() <= EXCERPT_LEN {
1154 return s;
1155 }
1156 // Cut at the last space before the limit, so a card ends on a word rather than mid-syllable.
1157 let cut = s.char_indices()
1158 .take(EXCERPT_LEN)
1159 .filter(|(_, c)| c.is_whitespace())
1160 .map(|(i, _)| i)
1161 .last()
1162 .unwrap_or_else(|| s.char_indices().nth(EXCERPT_LEN).map(|(i, _)| i).unwrap_or(s.len()));
1163 let mut out = s[..cut].trim_end().to_string();
1164 out.push('…');
1165 out
1166}
1167
1168/// Splits a leading `YYYY-MM-DD-` from a file's stem, giving the date it names and the slug that is
1169/// left. A stem that does not begin with a date is all slug.
1170///
1171/// The shape is checked rather than the value: a date is ten characters, digits where digits belong
1172/// and dashes where dashes belong, followed by a dash. `2026-13-45` passes, and is a date this does not
1173/// have to understand -- it sorts, which is all that is asked of it here.
1174pub fn split_date(stem: &str) -> (Option<String>, String) {
1175 let b = stem.as_bytes();
1176 if b.len() < 11 || b[10] != b'-' {
1177 return (None, stem.to_string());
1178 }
1179 let shaped = b[..10].iter().enumerate().all(|(i, c)| {
1180 match i {
1181 4 | 7 => *c == b'-',
1182 _ => c.is_ascii_digit(),
1183 }
1184 });
1185 if !shaped {
1186 return (None, stem.to_string());
1187 }
1188 (Some(stem[..10].to_string()), stem[11..].to_string())
1189}
1190
1191#[cfg(test)]
1192mod tests {
1193 use super::*;
1194
1195 #[test]
1196 fn test_a_dated_name_splits_into_a_date_and_a_slug_00() -> Outcome<()> {
1197 assert_eq!(
1198 split_date("2026-07-17-on-rent"),
1199 (Some("2026-07-17".to_string()), "on-rent".to_string()),
1200 );
1201 Ok(())
1202 }
1203
1204 /// A name that does not begin with a date is all slug, and says nothing about when it was written.
1205 #[test]
1206 fn test_an_undated_name_is_all_slug_01() -> Outcome<()> {
1207 assert_eq!(split_date("on-rent"), (None, "on-rent".to_string()));
1208 // Shaped like a date but not punctuated like one.
1209 assert_eq!(split_date("2026_07_17-on-rent"), (None, "2026_07_17-on-rent".to_string()));
1210 // A date with nothing after it is a name, not a date and an empty slug.
1211 assert_eq!(split_date("2026-07-17"), (None, "2026-07-17".to_string()));
1212 Ok(())
1213 }
1214
1215 /// The shape is what is checked. A date this cannot make sense of still sorts, and sorting is all
1216 /// that is asked of it.
1217 #[test]
1218 fn test_a_date_is_checked_for_shape_not_sense_02() -> Outcome<()> {
1219 assert_eq!(
1220 split_date("2026-13-45-impossible"),
1221 (Some("2026-13-45".to_string()), "impossible".to_string()),
1222 );
1223 Ok(())
1224 }
1225
1226 /// The prefix and what sits under it, and nothing that merely starts the same way.
1227 #[test]
1228 fn test_a_vhost_owns_its_prefix_and_no_more_03() -> Outcome<()> {
1229 let cfg = PublishConfig { path: fmt!("/asides"), ..Default::default() };
1230 assert!(cfg.owns("/asides"));
1231 assert!(cfg.owns("/asides/on-rent"));
1232 assert!(cfg.owns("/asides/feed.xml"));
1233 assert!(!cfg.owns("/asides-are-great"));
1234 assert!(!cfg.owns("/aside"));
1235 assert!(!cfg.owns("/"));
1236 Ok(())
1237 }
1238
1239 /// A path is rooted and has no trailing slash, whatever the config said, so every route built from
1240 /// it is the route it looks like.
1241 #[test]
1242 fn test_a_path_is_tidied_04() -> Outcome<()> {
1243 let m = mapdat!{ "path" => dat!("asides/") }.get_map().unwrap();
1244 let cfg = res!(PublishConfig::from_datmap(&m));
1245 assert_eq!(cfg.path, "/asides");
1246 assert_eq!(cfg.feed_path(), "/asides/feed.xml");
1247 assert_eq!(cfg.path_of("on-rent"), "/asides/on-rent");
1248 Ok(())
1249 }
1250
1251 /// A tag is validated against its normalised form, so what a person types is judged by what the
1252 /// store would keep: case and surrounding space do not decide it, a space inside does.
1253 #[test]
1254 fn test_a_tag_is_a_small_alphabet_06() -> Outcome<()> {
1255 assert!(valid_tag("rust"));
1256 assert!(valid_tag("Rust")); // normalised to `rust`
1257 assert!(valid_tag(" web ")); // trimmed
1258 assert!(valid_tag("web-dev"));
1259 assert!(valid_tag("c99"));
1260 assert!(!valid_tag(""));
1261 assert!(!valid_tag(" "));
1262 assert!(!valid_tag("a b")); // a space is dropped, not hyphenated
1263 assert!(!valid_tag("a/b"));
1264 assert!(!valid_tag("café"));
1265 assert!(!valid_tag(&"x".repeat(TAG_MAX + 1)));
1266 Ok(())
1267 }
1268
1269 /// The comma field splits, normalises, drops the invalid in silence, and dedupes keeping first
1270 /// appearance.
1271 #[test]
1272 fn test_tags_parse_from_a_comma_field_07() -> Outcome<()> {
1273 assert_eq!(parse_tags("rust, web"), vec![fmt!("rust"), fmt!("web")]);
1274 assert_eq!(parse_tags("Rust, RUST, rust"), vec![fmt!("rust")]);
1275 assert_eq!(parse_tags(" web , , rust "), vec![fmt!("web"), fmt!("rust")]);
1276 // The invalid drop out and the valid stay.
1277 assert_eq!(parse_tags("rust, a b, web"), vec![fmt!("rust"), fmt!("web")]);
1278 assert!(parse_tags("").is_empty());
1279 assert!(parse_tags(" , ").is_empty());
1280 Ok(())
1281 }
1282
1283 /// A slug that could climb out of the directory finds nothing, and finds it before it touches a
1284 /// path.
1285 #[test]
1286 fn test_a_slug_cannot_climb_out_05() -> Outcome<()> {
1287 for bad in ["../../etc/passwd", "..", "a/b", "a.b", ""] {
1288 assert!(
1289 res!(read_one("/nonexistent-directory", bad, "test")).is_none(),
1290 "'{}' was not refused", bad,
1291 );
1292 }
1293 Ok(())
1294 }
1295
1296 /// The obvious machines are discarded, whatever case they announce themselves in, and a request
1297 /// with no user-agent at all is one of them.
1298 #[test]
1299 fn test_a_machine_is_not_a_reader_08() -> Outcome<()> {
1300 for ua in [
1301 "Googlebot/2.1 (+http://www.google.com/bot.html)",
1302 "Mozilla/5.0 (compatible; bingbot/2.0)",
1303 "curl/8.5.0",
1304 "python-requests/2.31.0",
1305 "facebookexternalhit/1.1",
1306 "Mozilla/5.0 HeadlessChrome/120.0.0.0",
1307 ] {
1308 assert!(looks_automated(Some(ua)), "'{}' was taken for a reader", ua);
1309 }
1310 // No user-agent, and an empty one, are scripts that did not bother.
1311 assert!(looks_automated(None));
1312 assert!(looks_automated(Some(" ")));
1313
1314 // A real browser is a reader.
1315 for ua in [
1316 "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) \
1317 Chrome/120.0.0.0 Safari/537.36",
1318 "Mozilla/5.0 (iPhone; CPU iPhone OS 17_0 like Mac OS X) AppleWebKit/605.1.15",
1319 ] {
1320 assert!(!looks_automated(Some(ua)), "'{}' was taken for a machine", ua);
1321 }
1322 Ok(())
1323 }
1324
1325 /// The author reading their own post is not a read, whatever they are browsing with.
1326 #[test]
1327 fn test_the_author_is_not_a_reader_09() -> Outcome<()> {
1328 let browser = "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 Chrome/120.0.0.0";
1329 // A management session is the author, so it never counts.
1330 assert!(!counts_as_read(true, Some(browser), false));
1331 // The same request without one is a reader.
1332 assert!(counts_as_read(false, Some(browser), false));
1333 // And a machine is not a reader either way.
1334 assert!(!counts_as_read(false, Some("Googlebot/2.1"), false));
1335 assert!(!counts_as_read(true, Some("Googlebot/2.1"), false));
1336 // A HEAD asked for no prose, so it read none -- however browser-like it looks. This is
1337 // the uptime monitor that would otherwise be the site's most devoted reader.
1338 assert!(!counts_as_read(false, Some(browser), true));
1339 Ok(())
1340 }
1341
1342 /// Comments are off unless a site asks, and a config that says nothing still loads.
1343 #[test]
1344 fn test_comments_are_off_unless_asked_10() -> Outcome<()> {
1345 // A publish block naming nothing: loads, and takes no comments.
1346 let m = DaticleMap::new();
1347 let cfg = res!(PublishConfig::from_datmap(&m));
1348 assert!(!cfg.comments, "a site that said nothing was given a public write endpoint");
1349
1350 // One that asks gets them.
1351 let mut m = DaticleMap::new();
1352 m.insert(dat!("comments"), Dat::Bool(true));
1353 assert!(res!(PublishConfig::from_datmap(&m)).comments);
1354
1355 // And one that says something else is refused rather than guessed at.
1356 let mut m = DaticleMap::new();
1357 m.insert(dat!("comments"), dat!("yes".to_string()));
1358 assert!(PublishConfig::from_datmap(&m).is_err(), "a non-boolean 'comments' was accepted");
1359 Ok(())
1360 }
1361
1362 /// The categories default where none are named, take a named list, and refuse a comma inside a
1363 /// name -- the character that joins them in the composer and the filter, which a name may not hold.
1364 #[test]
1365 fn test_categories_default_and_reject_commas_12() -> Outcome<()> {
1366 // Named none: the built-in set.
1367 let cfg = res!(PublishConfig::from_datmap(&DaticleMap::new()));
1368 assert_eq!(cfg.categories, CATEGORIES_DEFAULT.iter().map(|c| c.to_string()).collect::<Vec<_>>());
1369
1370 // A multi-word name is fine: a space is not the delimiter.
1371 let mut m = DaticleMap::new();
1372 m.insert(dat!("categories"), Dat::List(vec![dat!("Book reviews".to_string()),
1373 dat!("Personal".to_string())]));
1374 let cfg = res!(PublishConfig::from_datmap(&m));
1375 assert_eq!(cfg.categories, vec![fmt!("Book reviews"), fmt!("Personal")]);
1376
1377 // A comma in a name is refused, since it would split the joined field in two.
1378 let mut m = DaticleMap::new();
1379 m.insert(dat!("categories"), Dat::List(vec![dat!("Reviews, essays".to_string())]));
1380 assert!(PublishConfig::from_datmap(&m).is_err(), "a comma in a category name was accepted");
1381 Ok(())
1382 }
1383
1384 /// The comment path names its post, and only a real slug.
1385 #[test]
1386 fn test_a_comment_path_names_its_post_11() -> Outcome<()> {
1387 let mut m = DaticleMap::new();
1388 m.insert(dat!("path"), dat!("/posts".to_string()));
1389 let cfg = res!(PublishConfig::from_datmap(&m));
1390
1391 assert_eq!(cfg.comment_path("a-post"), "/posts/a-post/comment");
1392 assert_eq!(cfg.comment_slug("/posts/a-post/comment"), Some("a-post"));
1393
1394 // Not comment paths at all.
1395 assert_eq!(cfg.comment_slug("/posts/a-post"), None);
1396 assert_eq!(cfg.comment_slug("/posts/comment"), None);
1397 assert_eq!(cfg.comment_slug("/elsewhere/a/comment"), None);
1398 // A slug that is not a name a post may wear reaches no lookup.
1399 assert_eq!(cfg.comment_slug("/posts/../../etc/comment"), None);
1400 assert_eq!(cfg.comment_slug("/posts//comment"), None);
1401 Ok(())
1402 }
1403
1404 /// Today is a date the store will take and the feed will date an entry by. A day the composer
1405 /// offers and the save falls back to is worth nothing if it fails the module's own check.
1406 #[test]
1407 fn test_today_is_a_date_a_post_may_wear_13() -> Outcome<()> {
1408 let d = res!(today().ok_or_else(|| err!("The clock would not read."; Missing)));
1409 assert_eq!(d.len(), DATE_LEN, "today is not a day-only date: {}", d);
1410 assert!(valid_date(&d), "the store would refuse today: {}", d);
1411 // Not the epoch, which is the thing the fallback exists to stop the feed emitting.
1412 assert!(!d.starts_with("1970-"), "the clock read as the epoch: {}", d);
1413 // Shaped as the store keeps it, so it round-trips without normalisation.
1414 assert_eq!(normalise_date(&d), d);
1415 Ok(())
1416 }
1417}