This commit is contained in:
shoney.arickathil 2026-04-05 00:45:02 +02:00
commit c32e482a58
271 changed files with 15357 additions and 0 deletions

3
.gitignore vendored Normal file
View file

@ -0,0 +1,3 @@
/target
/data
/content

6
.gitmodules vendored Normal file
View file

@ -0,0 +1,6 @@
[submodule "reference/writeonce-app"]
path = reference/writeonce-app
url = https://github.com/shoneyJ/writeonce-app
[submodule "reference/writeonce-api"]
path = reference/writeonce-api
url = https://github.com/shoneyJ/writeonce-api

579
Cargo.lock generated Normal file
View file

@ -0,0 +1,579 @@
# This file is automatically @generated by Cargo.
# It is not intended for manual editing.
version = 4
[[package]]
name = "anyhow"
version = "1.0.102"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "7f202df86484c868dbad7eaa557ef785d5c66295e41b460ef922eca0723b842c"
[[package]]
name = "bincode"
version = "1.3.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b1f45e9417d87227c7a56d22e471c6206462cba514c7590c09aff4cf6d1ddcad"
dependencies = [
"serde",
]
[[package]]
name = "bitflags"
version = "2.11.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "843867be96c8daad0d758b57df9392b6d8d271134fce549de6ce169ff98a92af"
[[package]]
name = "cfg-if"
version = "1.0.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801"
[[package]]
name = "equivalent"
version = "1.0.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "877a4ace8713b0bcf2a4e7eec82529c029f1d0619886d18145fea96c3ffe5c0f"
[[package]]
name = "errno"
version = "0.3.14"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "39cab71617ae0d63f51a36d69f866391735b51691dbda63cf6f96d042b63efeb"
dependencies = [
"libc",
"windows-sys",
]
[[package]]
name = "fastrand"
version = "2.3.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "37909eebbb50d72f9059c3b6d82c0463f2ff062c9e95845c43a6c9c0355411be"
[[package]]
name = "foldhash"
version = "0.1.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d9c4f5dac5e15c24eb999c26181a6ca40b39fe946cbe4c263c7209467bc83af2"
[[package]]
name = "getrandom"
version = "0.4.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0de51e6874e94e7bf76d726fc5d13ba782deca734ff60d5bb2fb2607c7406555"
dependencies = [
"cfg-if",
"libc",
"r-efi",
"wasip2",
"wasip3",
]
[[package]]
name = "hashbrown"
version = "0.15.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9229cfe53dfd69f0609a49f65461bd93001ea1ef889cd5529dd176593f5338a1"
dependencies = [
"foldhash",
]
[[package]]
name = "hashbrown"
version = "0.16.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "841d1cc9bed7f9236f321df977030373f4a4163ae1a7dbfe1a51a2c1a51d9100"
[[package]]
name = "heck"
version = "0.5.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "2304e00983f87ffb38b55b444b5e3b60a884b5d30c0fca7d82fe33449bbe55ea"
[[package]]
name = "id-arena"
version = "2.3.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "3d3067d79b975e8844ca9eb072e16b31c3c1c36928edf9c6789548c524d0d954"
[[package]]
name = "indexmap"
version = "2.13.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "7714e70437a7dc3ac8eb7e6f8df75fd8eb422675fc7678aff7364301092b1017"
dependencies = [
"equivalent",
"hashbrown 0.16.1",
"serde",
"serde_core",
]
[[package]]
name = "itoa"
version = "1.0.18"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682"
[[package]]
name = "leb128fmt"
version = "0.1.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "09edd9e8b54e49e587e4f6295a7d29c3ea94d469cb40ab8ca70b288248a81db2"
[[package]]
name = "libc"
version = "0.2.183"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b5b646652bf6661599e1da8901b3b9522896f01e736bad5f723fe7a3a27f899d"
[[package]]
name = "linux-raw-sys"
version = "0.12.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "32a66949e030da00e8c7d4434b251670a91556f4144941d37452769c25d58a53"
[[package]]
name = "log"
version = "0.4.29"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "5e5032e24019045c762d3c0f28f5b6b8bbf38563a65908389bf7978758920897"
[[package]]
name = "memchr"
version = "2.8.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f8ca58f447f06ed17d5fc4043ce1b10dd205e060fb3ce5b979b8ed8e59ff3f79"
[[package]]
name = "once_cell"
version = "1.21.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50"
[[package]]
name = "prettyplease"
version = "0.2.37"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "479ca8adacdd7ce8f1fb39ce9ecccbfe93a3f1344b3d0d97f20bc0196208f62b"
dependencies = [
"proc-macro2",
"syn",
]
[[package]]
name = "proc-macro2"
version = "1.0.106"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8fd00f0bb2e90d81d1044c2b32617f68fcb9fa3bb7640c23e9c748e53fb30934"
dependencies = [
"unicode-ident",
]
[[package]]
name = "quote"
version = "1.0.45"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "41f2619966050689382d2b44f664f4bc593e129785a36d6ee376ddf37259b924"
dependencies = [
"proc-macro2",
]
[[package]]
name = "r-efi"
version = "6.0.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf"
[[package]]
name = "rustix"
version = "1.1.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b6fe4565b9518b83ef4f91bb47ce29620ca828bd32cb7e408f0062e9930ba190"
dependencies = [
"bitflags",
"errno",
"libc",
"linux-raw-sys",
"windows-sys",
]
[[package]]
name = "semver"
version = "1.0.27"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d767eb0aabc880b29956c35734170f26ed551a859dbd361d140cdbeca61ab1e2"
[[package]]
name = "serde"
version = "1.0.228"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9a8e94ea7f378bd32cbbd37198a4a91436180c5bb472411e48b5ec2e2124ae9e"
dependencies = [
"serde_core",
"serde_derive",
]
[[package]]
name = "serde_core"
version = "1.0.228"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "41d385c7d4ca58e59fc732af25c3983b67ac852c1a25000afe1175de458b67ad"
dependencies = [
"serde_derive",
]
[[package]]
name = "serde_derive"
version = "1.0.228"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d540f220d3187173da220f885ab66608367b6574e925011a9353e4badda91d79"
dependencies = [
"proc-macro2",
"quote",
"syn",
]
[[package]]
name = "serde_json"
version = "1.0.149"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "83fc039473c5595ace860d8c4fafa220ff474b3fc6bfdb4293327f1a37e94d86"
dependencies = [
"itoa",
"memchr",
"serde",
"serde_core",
"zmij",
]
[[package]]
name = "syn"
version = "2.0.117"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e665b8803e7b1d2a727f4023456bbbbe74da67099c585258af0ad9c5013b9b99"
dependencies = [
"proc-macro2",
"quote",
"unicode-ident",
]
[[package]]
name = "tempfile"
version = "3.27.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd"
dependencies = [
"fastrand",
"getrandom",
"once_cell",
"rustix",
"windows-sys",
]
[[package]]
name = "unicode-ident"
version = "1.0.24"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e6e4313cd5fcd3dad5cafa179702e2b244f760991f45397d14d4ebf38247da75"
[[package]]
name = "unicode-xid"
version = "0.2.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ebc1c04c71510c7f702b52b7c350734c9ff1295c464a03335b00bb84fc54f853"
[[package]]
name = "wasip2"
version = "1.0.2+wasi-0.2.9"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9517f9239f02c069db75e65f174b3da828fe5f5b945c4dd26bd25d89c03ebcf5"
dependencies = [
"wit-bindgen",
]
[[package]]
name = "wasip3"
version = "0.4.0+wasi-0.3.0-rc-2026-01-06"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "5428f8bf88ea5ddc08faddef2ac4a67e390b88186c703ce6dbd955e1c145aca5"
dependencies = [
"wit-bindgen",
]
[[package]]
name = "wasm-encoder"
version = "0.244.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "990065f2fe63003fe337b932cfb5e3b80e0b4d0f5ff650e6985b1048f62c8319"
dependencies = [
"leb128fmt",
"wasmparser",
]
[[package]]
name = "wasm-metadata"
version = "0.244.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "bb0e353e6a2fbdc176932bbaab493762eb1255a7900fe0fea1a2f96c296cc909"
dependencies = [
"anyhow",
"indexmap",
"wasm-encoder",
"wasmparser",
]
[[package]]
name = "wasmparser"
version = "0.244.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "47b807c72e1bac69382b3a6fb3dbe8ea4c0ed87ff5629b8685ae6b9a611028fe"
dependencies = [
"bitflags",
"hashbrown 0.15.5",
"indexmap",
"semver",
]
[[package]]
name = "windows-link"
version = "0.2.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f0805222e57f7521d6a62e36fa9163bc891acd422f971defe97d64e70d0a4fe5"
[[package]]
name = "windows-sys"
version = "0.61.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ae137229bcbd6cdf0f7b80a31df61766145077ddf49416a728b02cb3921ff3fc"
dependencies = [
"windows-link",
]
[[package]]
name = "wit-bindgen"
version = "0.51.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d7249219f66ced02969388cf2bb044a09756a083d0fab1e566056b04d9fbcaa5"
dependencies = [
"wit-bindgen-rust-macro",
]
[[package]]
name = "wit-bindgen-core"
version = "0.51.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ea61de684c3ea68cb082b7a88508a8b27fcc8b797d738bfc99a82facf1d752dc"
dependencies = [
"anyhow",
"heck",
"wit-parser",
]
[[package]]
name = "wit-bindgen-rust"
version = "0.51.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b7c566e0f4b284dd6561c786d9cb0142da491f46a9fbed79ea69cdad5db17f21"
dependencies = [
"anyhow",
"heck",
"indexmap",
"prettyplease",
"syn",
"wasm-metadata",
"wit-bindgen-core",
"wit-component",
]
[[package]]
name = "wit-bindgen-rust-macro"
version = "0.51.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0c0f9bfd77e6a48eccf51359e3ae77140a7f50b1e2ebfe62422d8afdaffab17a"
dependencies = [
"anyhow",
"prettyplease",
"proc-macro2",
"quote",
"syn",
"wit-bindgen-core",
"wit-bindgen-rust",
]
[[package]]
name = "wit-component"
version = "0.244.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9d66ea20e9553b30172b5e831994e35fbde2d165325bec84fc43dbf6f4eb9cb2"
dependencies = [
"anyhow",
"bitflags",
"indexmap",
"log",
"serde",
"serde_derive",
"serde_json",
"wasm-encoder",
"wasm-metadata",
"wasmparser",
"wit-parser",
]
[[package]]
name = "wit-parser"
version = "0.244.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ecc8ac4bc1dc3381b7f59c34f00b67e18f910c2c0f50015669dde7def656a736"
dependencies = [
"anyhow",
"id-arena",
"indexmap",
"log",
"semver",
"serde",
"serde_derive",
"serde_json",
"unicode-xid",
"wasmparser",
]
[[package]]
name = "wo-event"
version = "0.1.0"
dependencies = [
"libc",
]
[[package]]
name = "wo-htmlx"
version = "0.1.0"
dependencies = [
"serde",
"serde_json",
"tempfile",
"wo-md",
"wo-model",
]
[[package]]
name = "wo-http"
version = "0.1.0"
dependencies = [
"libc",
"tempfile",
]
[[package]]
name = "wo-index"
version = "0.1.0"
dependencies = [
"bincode",
"libc",
"serde",
"tempfile",
"wo-model",
]
[[package]]
name = "wo-md"
version = "0.1.0"
[[package]]
name = "wo-model"
version = "0.1.0"
dependencies = [
"bincode",
"serde",
"serde_json",
"tempfile",
"wo-md",
]
[[package]]
name = "wo-route"
version = "0.1.0"
dependencies = [
"wo-http",
]
[[package]]
name = "wo-rt"
version = "0.1.0"
dependencies = [
"bincode",
"libc",
"serde",
"serde_json",
"tempfile",
"wo-event",
"wo-htmlx",
"wo-http",
"wo-md",
"wo-model",
"wo-route",
"wo-serve",
"wo-store",
"wo-sub",
"wo-watch",
]
[[package]]
name = "wo-seg"
version = "0.1.0"
dependencies = [
"bincode",
"libc",
"serde",
"tempfile",
"wo-model",
]
[[package]]
name = "wo-serve"
version = "0.1.0"
dependencies = [
"libc",
"tempfile",
"wo-http",
]
[[package]]
name = "wo-store"
version = "0.1.0"
dependencies = [
"serde_json",
"tempfile",
"wo-index",
"wo-model",
"wo-seg",
]
[[package]]
name = "wo-sub"
version = "0.1.0"
dependencies = [
"bincode",
"libc",
"serde",
"serde_json",
"tempfile",
"wo-event",
"wo-model",
"wo-store",
]
[[package]]
name = "wo-watch"
version = "0.1.0"
dependencies = [
"libc",
"tempfile",
"wo-store",
]
[[package]]
name = "zmij"
version = "1.0.21"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b8848ee67ecc8aedbaf3e4122217aff892639231befc6a1b58d29fff4c2cabaa"

23
Cargo.toml Normal file
View file

@ -0,0 +1,23 @@
[workspace]
resolver = "2"
members = [
"crates/wo-model",
"crates/wo-seg",
"crates/wo-index",
"crates/wo-store",
"crates/wo-watch",
"crates/wo-event",
"crates/wo-sub",
"crates/wo-rt",
"crates/wo-http",
"crates/wo-route",
"crates/wo-htmlx",
"crates/wo-md",
"crates/wo-serve",
]
[workspace.dependencies]
serde = { version = "1", features = ["derive"] }
serde_json = "1"
bincode = "1"
libc = "0.2"

64
README.md Normal file
View file

@ -0,0 +1,64 @@
# writeonce
A single self-contained binary that serves a content platform — no external database, no cloud pipeline, no JavaScript framework. Built in Rust on raw Linux kernel primitives.
## Why
The original writeonce system spread across five repositories, four languages, AWS infrastructure (S3, Lambda, API Gateway), PostgreSQL, and an Angular frontend. All of that to serve articles from local files. This project collapses everything into one process that owns storage, serves content, and pushes real-time updates.
## Architecture
- **Single process** — one binary replaces S3 + Lambda + Rust API + PostgreSQL + Angular
- **Embedded storage** — custom `.seg` segment files with positional indexing, no external database
- **Real-time subscriptions** — route-based SSE streams push content diffs to connected clients
- **Server-rendered HTML** — `.htmlx` templates with data bindings, minimal client-side JS
- **Markdown-first content** — `.md` files are the source of truth, JSON holds only metadata
- **Linux kernel I/O** — `epoll`, `inotify`, `eventfd`, `timerfd`, `sendfile` — no tokio, no async runtime
## Workspace Crates
| Crate | Purpose |
|-------|---------|
| `wo-model` | Article and metadata types |
| `wo-seg` | Segment file reader/writer (.seg format) |
| `wo-index` | Title hash map, date sorted array, tags inverted index |
| `wo-store` | Query API over segments and indexes |
| `wo-watch` | `inotify`-based content directory watcher |
| `wo-event` | `epoll` event loop, `eventfd`, `timerfd`, `signalfd` |
| `wo-sub` | Subscription manager and diff delivery |
| `wo-rt` | Single-threaded runtime tying I/O sources together |
| `wo-http` | HTTP request parsing and response writing |
| `wo-route` | URL routing and handler dispatch |
| `wo-htmlx` | Template engine for `.htmlx` files |
| `wo-md` | Markdown to HTML rendering |
| `wo-serve` | Binary entry point — wires everything together |
## Build
```sh
cargo build --release
```
## Deploy
The binary runs behind nginx with Let's Encrypt SSL. See `infra/setup.sh` for first-time server setup and `docs/07-ssl.md` for the full deployment walkthrough.
```sh
# Build, copy binary, sync content, restart service
./infra/deploy.sh
```
## Documentation
Design documents live in `docs/`:
- `00-linux.md` — Linux kernel primitives used
- `01-problem.md` — Problem statement and motivation
- `02-recovery.md` — Target architecture
- `03-data.md` — Embedded storage and subscription model
- `04-ui.md` — Server-rendered HTMLX templates
- `05-datalayer.md` — Data layer implementation status
- `06-markdown-render.md` — Markdown-first content model
- `07-ssl.md` — SSL, nginx, and deployment
- `runtime/` — Deep dives on async runtimes, fibers, and Rust's ownership model
- `future-scope/` — Planned features including AI agent content management

View file

@ -0,0 +1,7 @@
[package]
name = "wo-event"
version = "0.1.0"
edition = "2021"
[dependencies]
libc = { workspace = true }

View file

@ -0,0 +1,183 @@
use std::io;
use std::os::unix::io::RawFd;
use std::time::Duration;
/// Caller-assigned identifier for a registered file descriptor.
pub type Token = u64;
/// Interest flags for epoll registration.
#[derive(Debug, Clone, Copy)]
pub enum Interest {
Readable,
Writable,
ReadWrite,
}
impl Interest {
fn to_epoll_flags(self) -> u32 {
match self {
Interest::Readable => libc::EPOLLIN as u32,
Interest::Writable => libc::EPOLLOUT as u32,
Interest::ReadWrite => (libc::EPOLLIN | libc::EPOLLOUT) as u32,
}
}
}
/// An event delivered by the event loop.
#[derive(Debug, Clone)]
pub struct Event {
pub token: Token,
pub readable: bool,
pub writable: bool,
pub error: bool,
pub hangup: bool,
}
/// Single-threaded event loop built on epoll.
pub struct EventLoop {
epoll_fd: RawFd,
}
impl EventLoop {
/// Create a new event loop.
pub fn new() -> io::Result<Self> {
let fd = unsafe { libc::epoll_create1(libc::EPOLL_CLOEXEC) };
if fd < 0 {
return Err(io::Error::last_os_error());
}
Ok(Self { epoll_fd: fd })
}
/// Register a file descriptor for the given interest.
pub fn register(&self, fd: RawFd, interest: Interest, token: Token) -> io::Result<()> {
let mut event = libc::epoll_event {
events: interest.to_epoll_flags() | libc::EPOLLRDHUP as u32,
u64: token,
};
let ret = unsafe { libc::epoll_ctl(self.epoll_fd, libc::EPOLL_CTL_ADD, fd, &mut event) };
if ret < 0 {
Err(io::Error::last_os_error())
} else {
Ok(())
}
}
/// Modify interest for an already-registered fd.
pub fn modify(&self, fd: RawFd, interest: Interest, token: Token) -> io::Result<()> {
let mut event = libc::epoll_event {
events: interest.to_epoll_flags() | libc::EPOLLRDHUP as u32,
u64: token,
};
let ret = unsafe { libc::epoll_ctl(self.epoll_fd, libc::EPOLL_CTL_MOD, fd, &mut event) };
if ret < 0 {
Err(io::Error::last_os_error())
} else {
Ok(())
}
}
/// Remove a file descriptor from the event loop.
pub fn deregister(&self, fd: RawFd) -> io::Result<()> {
let ret = unsafe {
libc::epoll_ctl(self.epoll_fd, libc::EPOLL_CTL_DEL, fd, std::ptr::null_mut())
};
if ret < 0 {
Err(io::Error::last_os_error())
} else {
Ok(())
}
}
/// Wait for events. Returns when at least one event is ready or timeout expires.
///
/// `timeout`: `None` blocks indefinitely, `Some(duration)` sets a timeout.
/// Returns up to 64 events per call.
pub fn poll(&self, timeout: Option<Duration>) -> io::Result<Vec<Event>> {
let timeout_ms = match timeout {
None => -1i32,
Some(d) => d.as_millis() as i32,
};
let mut events = [libc::epoll_event { events: 0, u64: 0 }; 64];
let n = unsafe {
libc::epoll_wait(self.epoll_fd, events.as_mut_ptr(), events.len() as i32, timeout_ms)
};
if n < 0 {
let err = io::Error::last_os_error();
// EINTR is not an error — just return empty.
if err.raw_os_error() == Some(libc::EINTR) {
return Ok(vec![]);
}
return Err(err);
}
let result = (0..n as usize)
.map(|i| {
let e = events[i].events;
Event {
token: events[i].u64,
readable: (e & libc::EPOLLIN as u32) != 0,
writable: (e & libc::EPOLLOUT as u32) != 0,
error: (e & libc::EPOLLERR as u32) != 0,
hangup: (e & (libc::EPOLLHUP | libc::EPOLLRDHUP) as u32) != 0,
}
})
.collect();
Ok(result)
}
/// Get the raw epoll file descriptor.
pub fn fd(&self) -> RawFd {
self.epoll_fd
}
}
impl Drop for EventLoop {
fn drop(&mut self) {
unsafe { libc::close(self.epoll_fd) };
}
}
#[cfg(test)]
mod tests {
use super::*;
use crate::EventFd;
#[test]
fn register_and_poll_eventfd() {
let eloop = EventLoop::new().unwrap();
let efd = EventFd::new().unwrap();
eloop.register(efd.fd(), Interest::Readable, 42).unwrap();
// Write to eventfd from the same thread.
efd.write(1).unwrap();
let events = eloop.poll(Some(Duration::from_millis(100))).unwrap();
assert_eq!(events.len(), 1);
assert_eq!(events[0].token, 42);
assert!(events[0].readable);
}
#[test]
fn poll_timeout_no_events() {
let eloop = EventLoop::new().unwrap();
let events = eloop.poll(Some(Duration::from_millis(10))).unwrap();
assert!(events.is_empty());
}
#[test]
fn deregister() {
let eloop = EventLoop::new().unwrap();
let efd = EventFd::new().unwrap();
eloop.register(efd.fd(), Interest::Readable, 1).unwrap();
eloop.deregister(efd.fd()).unwrap();
efd.write(1).unwrap();
let events = eloop.poll(Some(Duration::from_millis(10))).unwrap();
assert!(events.is_empty());
}
}

View file

@ -0,0 +1,66 @@
use std::io;
use std::os::unix::io::RawFd;
/// Wrapper around Linux eventfd(2) for lightweight signaling.
pub struct EventFd {
fd: RawFd,
}
impl EventFd {
/// Create a new eventfd with initial value 0.
pub fn new() -> io::Result<Self> {
let fd = unsafe { libc::eventfd(0, libc::EFD_NONBLOCK | libc::EFD_CLOEXEC) };
if fd < 0 {
return Err(io::Error::last_os_error());
}
Ok(Self { fd })
}
/// Write a value to the eventfd (signals waiters).
pub fn write(&self, val: u64) -> io::Result<()> {
let buf = val.to_ne_bytes();
let ret = unsafe { libc::write(self.fd, buf.as_ptr() as *const libc::c_void, 8) };
if ret < 0 {
Err(io::Error::last_os_error())
} else {
Ok(())
}
}
/// Read the current value (resets to 0).
pub fn read(&self) -> io::Result<u64> {
let mut buf = [0u8; 8];
let ret = unsafe { libc::read(self.fd, buf.as_mut_ptr() as *mut libc::c_void, 8) };
if ret < 0 {
Err(io::Error::last_os_error())
} else {
Ok(u64::from_ne_bytes(buf))
}
}
/// Get the raw file descriptor for epoll registration.
pub fn fd(&self) -> RawFd {
self.fd
}
}
impl Drop for EventFd {
fn drop(&mut self) {
unsafe { libc::close(self.fd) };
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn write_and_read() {
let efd = EventFd::new().unwrap();
efd.write(5).unwrap();
efd.write(3).unwrap();
// eventfd accumulates writes.
let val = efd.read().unwrap();
assert_eq!(val, 8);
}
}

View file

@ -0,0 +1,9 @@
mod epoll;
mod eventfd;
mod timerfd;
mod signalfd;
pub use epoll::{EventLoop, Event, Interest, Token};
pub use eventfd::EventFd;
pub use timerfd::TimerFd;
pub use signalfd::SignalFd;

View file

@ -0,0 +1,62 @@
use std::io;
use std::os::unix::io::RawFd;
/// Wrapper around Linux signalfd(2) for handling signals via fd.
///
/// Blocks SIGINT and SIGTERM in the process signal mask and delivers
/// them as readable events on the fd instead.
pub struct SignalFd {
fd: RawFd,
}
impl SignalFd {
/// Create a signalfd that catches SIGINT and SIGTERM.
///
/// Also blocks these signals in the process signal mask so they
/// don't trigger default handlers.
pub fn new() -> io::Result<Self> {
let mut mask: libc::sigset_t = unsafe { std::mem::zeroed() };
unsafe {
libc::sigemptyset(&mut mask);
libc::sigaddset(&mut mask, libc::SIGINT);
libc::sigaddset(&mut mask, libc::SIGTERM);
// Block these signals so they go to signalfd instead.
let ret = libc::pthread_sigmask(libc::SIG_BLOCK, &mask, std::ptr::null_mut());
if ret != 0 {
return Err(io::Error::from_raw_os_error(ret));
}
}
let fd = unsafe { libc::signalfd(-1, &mask, libc::SFD_NONBLOCK | libc::SFD_CLOEXEC) };
if fd < 0 {
return Err(io::Error::last_os_error());
}
Ok(Self { fd })
}
/// Read a pending signal. Returns the signal number (e.g., SIGINT = 2).
pub fn read(&self) -> io::Result<i32> {
let mut info: libc::signalfd_siginfo = unsafe { std::mem::zeroed() };
let size = std::mem::size_of::<libc::signalfd_siginfo>();
let ret = unsafe {
libc::read(self.fd, &mut info as *mut _ as *mut libc::c_void, size)
};
if ret < 0 {
Err(io::Error::last_os_error())
} else {
Ok(info.ssi_signo as i32)
}
}
pub fn fd(&self) -> RawFd {
self.fd
}
}
impl Drop for SignalFd {
fn drop(&mut self) {
unsafe { libc::close(self.fd) };
}
}

View file

@ -0,0 +1,91 @@
use std::io;
use std::os::unix::io::RawFd;
use std::time::Duration;
/// Wrapper around Linux timerfd_create(2) for timer-as-fd.
pub struct TimerFd {
fd: RawFd,
}
impl TimerFd {
/// Create a new monotonic timer fd.
pub fn new() -> io::Result<Self> {
let fd = unsafe {
libc::timerfd_create(
libc::CLOCK_MONOTONIC,
libc::TFD_NONBLOCK | libc::TFD_CLOEXEC,
)
};
if fd < 0 {
return Err(io::Error::last_os_error());
}
Ok(Self { fd })
}
/// Arm the timer to fire once after `initial` and then repeat every `interval`.
///
/// Pass `Duration::ZERO` for `interval` for a one-shot timer.
pub fn set(&self, initial: Duration, interval: Duration) -> io::Result<()> {
let spec = libc::itimerspec {
it_interval: duration_to_timespec(interval),
it_value: duration_to_timespec(initial),
};
let ret =
unsafe { libc::timerfd_settime(self.fd, 0, &spec, std::ptr::null_mut()) };
if ret < 0 {
Err(io::Error::last_os_error())
} else {
Ok(())
}
}
/// Read the number of expirations since last read.
pub fn read(&self) -> io::Result<u64> {
let mut buf = [0u8; 8];
let ret = unsafe { libc::read(self.fd, buf.as_mut_ptr() as *mut libc::c_void, 8) };
if ret < 0 {
Err(io::Error::last_os_error())
} else {
Ok(u64::from_ne_bytes(buf))
}
}
pub fn fd(&self) -> RawFd {
self.fd
}
}
impl Drop for TimerFd {
fn drop(&mut self) {
unsafe { libc::close(self.fd) };
}
}
fn duration_to_timespec(d: Duration) -> libc::timespec {
libc::timespec {
tv_sec: d.as_secs() as libc::time_t,
tv_nsec: d.subsec_nanos() as libc::c_long,
}
}
#[cfg(test)]
mod tests {
use super::*;
use crate::{EventLoop, Interest};
#[test]
fn timer_fires() {
let eloop = EventLoop::new().unwrap();
let timer = TimerFd::new().unwrap();
timer.set(Duration::from_millis(20), Duration::ZERO).unwrap();
eloop.register(timer.fd(), Interest::Readable, 99).unwrap();
let events = eloop.poll(Some(Duration::from_millis(100))).unwrap();
assert!(!events.is_empty());
assert_eq!(events[0].token, 99);
let expirations = timer.read().unwrap();
assert!(expirations >= 1);
}
}

View file

@ -0,0 +1,13 @@
[package]
name = "wo-htmlx"
version = "0.1.0"
edition = "2021"
[dependencies]
wo-model = { path = "../wo-model" }
wo-md = { path = "../wo-md" }
serde = { workspace = true }
serde_json = { workspace = true }
[dev-dependencies]
tempfile = "3"

View file

@ -0,0 +1,18 @@
/// A parsed template node.
#[derive(Debug, Clone, PartialEq)]
pub enum Node {
/// Raw HTML passthrough.
Literal(String),
/// Variable binding: `{{path.to.value}}`
Binding(Vec<String>),
/// Iteration: `{{#each path}} ... {{/each}}`
Each {
path: Vec<String>,
body: Vec<Node>,
},
/// Partial inclusion: `{{> name arg=value}}`
Partial {
name: String,
args: Vec<(String, Vec<String>)>,
},
}

View file

@ -0,0 +1,11 @@
mod ast;
mod parser;
mod render;
mod registry;
mod value;
pub use ast::Node;
pub use parser::parse;
pub use render::render;
pub use registry::TemplateRegistry;
pub use value::Value;

View file

@ -0,0 +1,173 @@
use crate::ast::Node;
/// Parse a template string into a list of AST nodes.
pub fn parse(template: &str) -> Vec<Node> {
let mut nodes = Vec::new();
let mut rest = template;
while !rest.is_empty() {
if let Some(pos) = rest.find("{{") {
// Literal before the opening `{{`.
if pos > 0 {
nodes.push(Node::Literal(rest[..pos].to_string()));
}
let after_open = &rest[pos + 2..];
if let Some(close) = after_open.find("}}") {
let expr = after_open[..close].trim();
rest = &after_open[close + 2..];
if let Some(each_path) = expr.strip_prefix("#each ") {
// Block: {{#each path}} ... {{/each}}
let path = parse_path(each_path.trim());
let (body, remaining) = parse_until_end_each(rest);
nodes.push(Node::Each { path, body });
rest = remaining;
} else if let Some(partial_expr) = expr.strip_prefix("> ") {
// Partial: {{> name arg=value}}
let (name, args) = parse_partial_expr(partial_expr.trim());
nodes.push(Node::Partial { name, args });
} else if !expr.starts_with('/') {
// Binding: {{path.to.value}}
let path = parse_path(expr);
nodes.push(Node::Binding(path));
}
// {{/each}} handled by parse_until_end_each
} else {
// No closing `}}` — treat rest as literal.
nodes.push(Node::Literal(rest.to_string()));
break;
}
} else {
// No more `{{` — rest is literal.
nodes.push(Node::Literal(rest.to_string()));
break;
}
}
nodes
}
fn parse_path(s: &str) -> Vec<String> {
s.split('.').map(|p| p.trim().to_string()).collect()
}
fn parse_until_end_each(input: &str) -> (Vec<Node>, &str) {
// Find the matching {{/each}}.
let mut depth = 1;
let mut pos = 0;
let bytes = input.as_bytes();
while pos < bytes.len() {
if let Some(open) = input[pos..].find("{{") {
let abs = pos + open;
let after = &input[abs + 2..];
if let Some(close) = after.find("}}") {
let expr = after[..close].trim();
if expr.starts_with("#each ") {
depth += 1;
} else if expr == "/each" {
depth -= 1;
if depth == 0 {
let body_str = &input[..abs];
let rest = &after[close + 2..];
return (parse(body_str), rest);
}
}
pos = abs + 2 + close + 2;
} else {
break;
}
} else {
break;
}
}
// Unmatched {{#each}} — return rest as literal body.
(vec![Node::Literal(input.to_string())], "")
}
fn parse_partial_expr(expr: &str) -> (String, Vec<(String, Vec<String>)>) {
let parts: Vec<&str> = expr.splitn(2, ' ').collect();
let name = parts[0].to_string();
let mut args = Vec::new();
if parts.len() > 1 {
for arg in parts[1].split_whitespace() {
if let Some((key, val)) = arg.split_once('=') {
args.push((key.to_string(), parse_path(val)));
}
}
}
(name, args)
}
#[cfg(test)]
mod tests {
use super::*;
use crate::ast::Node;
#[test]
fn parse_literal() {
let nodes = parse("<h1>Hello</h1>");
assert_eq!(nodes, vec![Node::Literal("<h1>Hello</h1>".into())]);
}
#[test]
fn parse_binding() {
let nodes = parse("{{article.title}}");
assert_eq!(nodes, vec![Node::Binding(vec!["article".into(), "title".into()])]);
}
#[test]
fn parse_mixed() {
let nodes = parse("<h1>{{title}}</h1>");
assert_eq!(nodes.len(), 3);
assert_eq!(nodes[0], Node::Literal("<h1>".into()));
assert_eq!(nodes[1], Node::Binding(vec!["title".into()]));
assert_eq!(nodes[2], Node::Literal("</h1>".into()));
}
#[test]
fn parse_each() {
let nodes = parse("{{#each items}}<li>{{this}}</li>{{/each}}");
assert_eq!(nodes.len(), 1);
match &nodes[0] {
Node::Each { path, body } => {
assert_eq!(path, &vec!["items".to_string()]);
assert_eq!(body.len(), 3);
}
_ => panic!("expected Each"),
}
}
#[test]
fn parse_partial() {
let nodes = parse("{{> article-card article=this}}");
assert_eq!(nodes.len(), 1);
match &nodes[0] {
Node::Partial { name, args } => {
assert_eq!(name, "article-card");
assert_eq!(args.len(), 1);
assert_eq!(args[0].0, "article");
assert_eq!(args[0].1, vec!["this".to_string()]);
}
_ => panic!("expected Partial"),
}
}
#[test]
fn parse_nested_each() {
let nodes = parse("{{#each sections}}{{#each paragraphs}}{{this}}{{/each}}{{/each}}");
assert_eq!(nodes.len(), 1);
match &nodes[0] {
Node::Each { body, .. } => {
assert_eq!(body.len(), 1);
assert!(matches!(&body[0], Node::Each { .. }));
}
_ => panic!("expected nested Each"),
}
}
}

View file

@ -0,0 +1,121 @@
use std::collections::HashMap;
use std::fs;
use std::io;
use std::path::Path;
use crate::ast::Node;
use crate::parser;
/// Holds parsed templates and partials, loaded from a templates directory.
pub struct TemplateRegistry {
/// Page templates (e.g., "home" -> parsed nodes from home.htmlx).
templates: HashMap<String, Vec<Node>>,
/// Component partials (e.g., "article-card" -> parsed nodes).
partials: HashMap<String, Vec<Node>>,
}
impl TemplateRegistry {
/// Load all .htmlx files from the given templates directory.
///
/// Templates in the root are page templates.
/// Templates in `components/` are partials.
pub fn load(templates_dir: &Path) -> io::Result<Self> {
let mut templates = HashMap::new();
let mut partials = HashMap::new();
// Load page templates from root.
if templates_dir.exists() {
for entry in fs::read_dir(templates_dir)? {
let entry = entry?;
let path = entry.path();
if path.is_file() && has_htmlx_ext(&path) {
let name = stem(&path);
let content = fs::read_to_string(&path)?;
templates.insert(name, parser::parse(&content));
}
}
}
// Load partials from components/.
let components_dir = templates_dir.join("components");
if components_dir.exists() {
for entry in fs::read_dir(&components_dir)? {
let entry = entry?;
let path = entry.path();
if path.is_file() && has_htmlx_ext(&path) {
let name = stem(&path);
let content = fs::read_to_string(&path)?;
partials.insert(name, parser::parse(&content));
}
}
}
Ok(Self { templates, partials })
}
/// Get a page template by name.
pub fn get(&self, name: &str) -> Option<&Vec<Node>> {
self.templates.get(name)
}
/// Get the partials map (for passing to render()).
pub fn partials(&self) -> &HashMap<String, Vec<Node>> {
&self.partials
}
/// Number of loaded templates.
pub fn template_count(&self) -> usize {
self.templates.len()
}
/// Number of loaded partials.
pub fn partial_count(&self) -> usize {
self.partials.len()
}
}
fn has_htmlx_ext(path: &Path) -> bool {
path.extension().and_then(|e| e.to_str()) == Some("htmlx")
}
fn stem(path: &Path) -> String {
path.file_stem()
.and_then(|s| s.to_str())
.unwrap_or("")
.to_string()
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn load_templates() {
let tmp = tempfile::tempdir().unwrap();
let dir = tmp.path();
fs::write(dir.join("home.htmlx"), "<h1>{{title}}</h1>").unwrap();
fs::write(dir.join("article.htmlx"), "<article>{{body}}</article>").unwrap();
fs::create_dir_all(dir.join("components")).unwrap();
fs::write(
dir.join("components/card.htmlx"),
"<div>{{name}}</div>",
)
.unwrap();
let reg = TemplateRegistry::load(dir).unwrap();
assert_eq!(reg.template_count(), 2);
assert_eq!(reg.partial_count(), 1);
assert!(reg.get("home").is_some());
assert!(reg.get("article").is_some());
assert!(reg.partials().contains_key("card"));
}
#[test]
fn empty_directory() {
let tmp = tempfile::tempdir().unwrap();
let reg = TemplateRegistry::load(tmp.path()).unwrap();
assert_eq!(reg.template_count(), 0);
}
}

View file

@ -0,0 +1,140 @@
use std::collections::BTreeMap;
use crate::ast::Node;
use crate::value::Value;
/// Render a list of AST nodes against a context value.
///
/// `partials` maps partial names to their parsed node lists.
pub fn render(
nodes: &[Node],
context: &Value,
partials: &std::collections::HashMap<String, Vec<Node>>,
) -> String {
let mut output = String::new();
for node in nodes {
match node {
Node::Literal(text) => {
output.push_str(text);
}
Node::Binding(path) => {
if path.len() == 1 && path[0] == "this" {
output.push_str(&context.to_display());
} else {
let val = context.resolve(path);
output.push_str(&val.to_display());
}
}
Node::Each { path, body } => {
let list_val = if path.len() == 1 && path[0] == "this" {
context
} else {
context.resolve(path)
};
for item in list_val.as_list() {
output.push_str(&render(body, item, partials));
}
}
Node::Partial { name, args } => {
if let Some(partial_nodes) = partials.get(name.as_str()) {
// Build the partial context from args.
let partial_ctx = if args.is_empty() {
context.clone()
} else {
let mut map = BTreeMap::new();
for (key, path) in args {
let val = if path.len() == 1 && path[0] == "this" {
context.clone()
} else {
context.resolve(path).clone()
};
map.insert(key.clone(), val);
}
Value::Object(map)
};
output.push_str(&render(partial_nodes, &partial_ctx, partials));
}
}
}
}
output
}
#[cfg(test)]
mod tests {
use super::*;
use crate::parser;
use std::collections::HashMap;
fn ctx(json: serde_json::Value) -> Value {
Value::from_json(&json)
}
#[test]
fn render_literal() {
let nodes = parser::parse("<h1>Hello</h1>");
let result = render(&nodes, &Value::Null, &HashMap::new());
assert_eq!(result, "<h1>Hello</h1>");
}
#[test]
fn render_binding() {
let nodes = parser::parse("<h1>{{title}}</h1>");
let context = ctx(serde_json::json!({"title": "My Article"}));
let result = render(&nodes, &context, &HashMap::new());
assert_eq!(result, "<h1>My Article</h1>");
}
#[test]
fn render_nested_binding() {
let nodes = parser::parse("by {{article.author}}");
let context = ctx(serde_json::json!({"article": {"author": "Shoney"}}));
let result = render(&nodes, &context, &HashMap::new());
assert_eq!(result, "by Shoney");
}
#[test]
fn render_each() {
let nodes = parser::parse("<ul>{{#each tags}}<li>{{this}}</li>{{/each}}</ul>");
let context = ctx(serde_json::json!({"tags": ["rust", "linux"]}));
let result = render(&nodes, &context, &HashMap::new());
assert_eq!(result, "<ul><li>rust</li><li>linux</li></ul>");
}
#[test]
fn render_each_objects() {
let nodes = parser::parse("{{#each sections}}<h2>{{heading}}</h2>{{/each}}");
let context = ctx(serde_json::json!({
"sections": [
{"heading": "Intro"},
{"heading": "Details"}
]
}));
let result = render(&nodes, &context, &HashMap::new());
assert_eq!(result, "<h2>Intro</h2><h2>Details</h2>");
}
#[test]
fn render_partial() {
let nodes = parser::parse("{{> card item=this}}");
let mut partials = HashMap::new();
partials.insert(
"card".to_string(),
parser::parse("<div>{{item.name}}</div>"),
);
let context = ctx(serde_json::json!({"name": "Test"}));
let result = render(&nodes, &context, &partials);
assert_eq!(result, "<div>Test</div>");
}
#[test]
fn render_missing_binding() {
let nodes = parser::parse("{{nonexistent}}");
let context = ctx(serde_json::json!({}));
let result = render(&nodes, &context, &HashMap::new());
assert_eq!(result, "");
}
}

View file

@ -0,0 +1,122 @@
use std::collections::BTreeMap;
/// Template context value — the data that template bindings resolve against.
#[derive(Debug, Clone, PartialEq)]
pub enum Value {
String(String),
Number(f64),
Bool(bool),
List(Vec<Value>),
Object(BTreeMap<String, Value>),
Null,
}
impl Value {
/// Resolve a dotted path like ["article", "title"] against this value.
pub fn resolve(&self, path: &[String]) -> &Value {
let mut current = self;
for key in path {
match current {
Value::Object(map) => {
current = map.get(key.as_str()).unwrap_or(&Value::Null);
}
_ => return &Value::Null,
}
}
current
}
/// Convert to a display string for template output.
pub fn to_display(&self) -> String {
match self {
Value::String(s) => s.clone(),
Value::Number(n) => {
if *n == (*n as i64) as f64 {
format!("{}", *n as i64)
} else {
format!("{}", n)
}
}
Value::Bool(b) => b.to_string(),
Value::Null => String::new(),
Value::List(items) => {
let parts: Vec<String> = items.iter().map(|v| v.to_display()).collect();
parts.join(", ")
}
Value::Object(_) => "[object]".to_string(),
}
}
/// Convert to a list for iteration (returns empty vec if not a list).
pub fn as_list(&self) -> &[Value] {
match self {
Value::List(items) => items,
_ => &[],
}
}
/// Convert a serde_json::Value to a template Value.
pub fn from_json(json: &serde_json::Value) -> Self {
match json {
serde_json::Value::Null => Value::Null,
serde_json::Value::Bool(b) => Value::Bool(*b),
serde_json::Value::Number(n) => Value::Number(n.as_f64().unwrap_or(0.0)),
serde_json::Value::String(s) => Value::String(s.clone()),
serde_json::Value::Array(arr) => {
Value::List(arr.iter().map(Value::from_json).collect())
}
serde_json::Value::Object(map) => {
let btree = map
.iter()
.map(|(k, v)| (k.clone(), Value::from_json(v)))
.collect();
Value::Object(btree)
}
}
}
}
/// Convert an Article to a template Value using serde_json as intermediary.
pub fn article_to_value(article: &wo_model::Article) -> Value {
let json = serde_json::to_value(article).unwrap_or(serde_json::Value::Null);
Value::from_json(&json)
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn resolve_path() {
let mut inner = BTreeMap::new();
inner.insert("title".into(), Value::String("Hello".into()));
let mut root = BTreeMap::new();
root.insert("article".into(), Value::Object(inner));
let val = Value::Object(root);
let result = val.resolve(&["article".into(), "title".into()]);
assert_eq!(result, &Value::String("Hello".into()));
}
#[test]
fn resolve_missing() {
let val = Value::Object(BTreeMap::new());
assert_eq!(val.resolve(&["nope".into()]), &Value::Null);
}
#[test]
fn from_json() {
let json: serde_json::Value = serde_json::json!({
"name": "test",
"count": 42,
"active": true,
"tags": ["a", "b"]
});
let val = Value::from_json(&json);
assert_eq!(
val.resolve(&["name".into()]),
&Value::String("test".into())
);
assert_eq!(val.resolve(&["count".into()]), &Value::Number(42.0));
}
}

10
crates/wo-http/Cargo.toml Normal file
View file

@ -0,0 +1,10 @@
[package]
name = "wo-http"
version = "0.1.0"
edition = "2021"
[dependencies]
libc = { workspace = true }
[dev-dependencies]
tempfile = "3"

View file

@ -0,0 +1,202 @@
use std::io;
use std::os::unix::io::RawFd;
use crate::request::{self, ParseResult, Request};
use crate::response::Response;
/// Connection state machine.
#[derive(Debug, Clone, Copy, PartialEq)]
pub enum State {
/// Accumulating request bytes.
ReadingRequest,
/// Writing response bytes.
WritingResponse,
/// Initial response sent, fd handed to subscription manager.
/// Connection stays open for diff payloads.
Subscribed,
/// Ready to close.
Done,
}
/// A single HTTP connection.
pub struct Connection {
pub fd: RawFd,
pub state: State,
read_buf: Vec<u8>,
write_buf: Vec<u8>,
write_offset: usize,
parsed_request: Option<Request>,
}
impl Connection {
pub fn new(fd: RawFd) -> Self {
Self {
fd,
state: State::ReadingRequest,
read_buf: Vec::with_capacity(4096),
write_buf: Vec::new(),
write_offset: 0,
parsed_request: None,
}
}
/// Read available data from the socket into the read buffer.
/// Returns Ok(true) if data was read, Ok(false) if connection closed.
pub fn read(&mut self) -> io::Result<bool> {
let mut buf = [0u8; 4096];
let n = unsafe {
libc::read(self.fd, buf.as_mut_ptr() as *mut libc::c_void, buf.len())
};
if n < 0 {
let err = io::Error::last_os_error();
if err.raw_os_error() == Some(libc::EAGAIN) {
return Ok(true); // no data yet, but connection still open
}
return Err(err);
}
if n == 0 {
return Ok(false); // connection closed
}
self.read_buf.extend_from_slice(&buf[..n as usize]);
Ok(true)
}
/// Try to parse the request from the read buffer.
/// Returns Some(Request) if a full request was parsed.
pub fn try_parse(&mut self) -> Option<Request> {
if self.parsed_request.is_some() {
return self.parsed_request.clone();
}
match request::parse(&self.read_buf) {
ParseResult::Complete(req, _consumed) => {
self.parsed_request = Some(req.clone());
Some(req)
}
ParseResult::Incomplete => None,
ParseResult::Error(_) => {
self.state = State::Done;
None
}
}
}
/// Set the response to write back to the client.
pub fn set_response(&mut self, response: &Response) {
self.write_buf = response.to_bytes();
self.write_offset = 0;
self.state = State::WritingResponse;
}
/// Write pending response bytes to the socket.
/// Returns Ok(true) if all bytes written, Ok(false) if partial (need EPOLLOUT).
pub fn write(&mut self) -> io::Result<bool> {
let remaining = &self.write_buf[self.write_offset..];
if remaining.is_empty() {
return Ok(true);
}
let n = unsafe {
libc::write(
self.fd,
remaining.as_ptr() as *const libc::c_void,
remaining.len(),
)
};
if n < 0 {
let err = io::Error::last_os_error();
if err.raw_os_error() == Some(libc::EAGAIN) {
return Ok(false);
}
return Err(err);
}
self.write_offset += n as usize;
if self.write_offset >= self.write_buf.len() {
Ok(true) // all written
} else {
Ok(false) // partial write
}
}
/// Transition to subscribed state (connection stays open).
pub fn set_subscribed(&mut self) {
self.state = State::Subscribed;
}
/// Mark connection as done (will be closed).
pub fn set_done(&mut self) {
self.state = State::Done;
}
/// Take the parsed request out of the connection.
pub fn take_request(&mut self) -> Option<Request> {
self.parsed_request.take()
}
}
impl Drop for Connection {
fn drop(&mut self) {
unsafe { libc::close(self.fd) };
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn connection_lifecycle() {
// Use a socketpair to simulate a connection.
let mut fds = [0i32; 2];
assert_eq!(
unsafe { libc::socketpair(libc::AF_UNIX, libc::SOCK_STREAM | libc::SOCK_NONBLOCK, 0, fds.as_mut_ptr()) },
0
);
let (server_fd, client_fd) = (fds[0], fds[1]);
// Write a request from the "client" side.
let request_bytes = b"GET /blog/test HTTP/1.1\r\nHost: localhost\r\n\r\n";
unsafe {
libc::write(
client_fd,
request_bytes.as_ptr() as *const libc::c_void,
request_bytes.len(),
);
}
// Create connection on the "server" side.
// We need to manage the fd manually since Connection will close it on drop.
let dup_fd = unsafe { libc::dup(server_fd) };
let mut conn = Connection::new(dup_fd);
assert_eq!(conn.state, State::ReadingRequest);
// Read and parse.
assert!(conn.read().unwrap());
let req = conn.try_parse().unwrap();
assert_eq!(req.path, "/blog/test");
// Set response.
let response = Response::html("<h1>Test</h1>".into());
conn.set_response(&response);
assert_eq!(conn.state, State::WritingResponse);
// Write response.
let complete = conn.write().unwrap();
assert!(complete);
// Read from client side to verify.
let mut buf = [0u8; 4096];
let n = unsafe {
libc::read(client_fd, buf.as_mut_ptr() as *mut libc::c_void, buf.len())
};
assert!(n > 0);
let response_str = std::str::from_utf8(&buf[..n as usize]).unwrap();
assert!(response_str.contains("HTTP/1.1 200 OK"));
assert!(response_str.contains("<h1>Test</h1>"));
unsafe {
libc::close(server_fd);
libc::close(client_fd);
}
}
}

View file

@ -0,0 +1,4 @@
pub mod listener;
pub mod request;
pub mod response;
pub mod connection;

View file

@ -0,0 +1,202 @@
use std::io;
use std::os::unix::io::RawFd;
/// A non-blocking TCP listener wrapping raw socket syscalls.
pub struct TcpListener {
fd: RawFd,
}
impl TcpListener {
/// Bind to the given address (e.g., "0.0.0.0:3000") and start listening.
pub fn bind(addr: &str) -> io::Result<Self> {
let (ip, port) = parse_addr(addr)?;
let fd = unsafe {
libc::socket(
libc::AF_INET,
libc::SOCK_STREAM | libc::SOCK_NONBLOCK | libc::SOCK_CLOEXEC,
0,
)
};
if fd < 0 {
return Err(io::Error::last_os_error());
}
// SO_REUSEADDR to allow quick restart.
let optval: libc::c_int = 1;
let ret = unsafe {
libc::setsockopt(
fd,
libc::SOL_SOCKET,
libc::SO_REUSEADDR,
&optval as *const _ as *const libc::c_void,
std::mem::size_of::<libc::c_int>() as libc::socklen_t,
)
};
if ret < 0 {
unsafe { libc::close(fd) };
return Err(io::Error::last_os_error());
}
let sockaddr = libc::sockaddr_in {
sin_family: libc::AF_INET as libc::sa_family_t,
sin_port: port.to_be(),
sin_addr: libc::in_addr { s_addr: ip },
sin_zero: [0; 8],
};
let ret = unsafe {
libc::bind(
fd,
&sockaddr as *const _ as *const libc::sockaddr,
std::mem::size_of::<libc::sockaddr_in>() as libc::socklen_t,
)
};
if ret < 0 {
unsafe { libc::close(fd) };
return Err(io::Error::last_os_error());
}
let ret = unsafe { libc::listen(fd, 128) };
if ret < 0 {
unsafe { libc::close(fd) };
return Err(io::Error::last_os_error());
}
Ok(Self { fd })
}
/// Accept a new connection. Returns the client fd or EAGAIN if none pending.
pub fn accept(&self) -> io::Result<Option<RawFd>> {
let client_fd = unsafe {
libc::accept4(
self.fd,
std::ptr::null_mut(),
std::ptr::null_mut(),
libc::SOCK_NONBLOCK | libc::SOCK_CLOEXEC,
)
};
if client_fd < 0 {
let err = io::Error::last_os_error();
if err.raw_os_error() == Some(libc::EAGAIN)
|| err.raw_os_error() == Some(libc::EWOULDBLOCK)
{
return Ok(None);
}
return Err(err);
}
Ok(Some(client_fd))
}
/// The listener file descriptor for epoll registration.
pub fn fd(&self) -> RawFd {
self.fd
}
/// Get the actual bound address (useful when binding to port 0).
pub fn local_addr(&self) -> io::Result<(String, u16)> {
let mut addr: libc::sockaddr_in = unsafe { std::mem::zeroed() };
let mut len = std::mem::size_of::<libc::sockaddr_in>() as libc::socklen_t;
let ret = unsafe {
libc::getsockname(
self.fd,
&mut addr as *mut _ as *mut libc::sockaddr,
&mut len,
)
};
if ret < 0 {
return Err(io::Error::last_os_error());
}
let ip = u32::from_be(addr.sin_addr.s_addr);
let port = u16::from_be(addr.sin_port);
let ip_str = format!(
"{}.{}.{}.{}",
(ip >> 24) & 0xFF,
(ip >> 16) & 0xFF,
(ip >> 8) & 0xFF,
ip & 0xFF,
);
Ok((ip_str, port))
}
}
impl Drop for TcpListener {
fn drop(&mut self) {
unsafe { libc::close(self.fd) };
}
}
fn parse_addr(addr: &str) -> io::Result<(u32, u16)> {
let parts: Vec<&str> = addr.rsplitn(2, ':').collect();
if parts.len() != 2 {
return Err(io::Error::new(io::ErrorKind::InvalidInput, "expected host:port"));
}
let port: u16 = parts[0]
.parse()
.map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "invalid port"))?;
let ip_str = parts[1];
let ip = if ip_str == "0.0.0.0" {
0u32
} else if ip_str == "127.0.0.1" {
0x7F000001u32.to_be()
} else {
let octets: Vec<u8> = ip_str
.split('.')
.map(|s| s.parse().unwrap_or(0))
.collect();
if octets.len() != 4 {
return Err(io::Error::new(io::ErrorKind::InvalidInput, "invalid IP"));
}
u32::from_ne_bytes([octets[0], octets[1], octets[2], octets[3]])
};
Ok((ip, port))
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn bind_and_accept() {
let listener = TcpListener::bind("127.0.0.1:0").unwrap();
let (_, port) = listener.local_addr().unwrap();
assert!(port > 0);
// No pending connections → None.
assert!(listener.accept().unwrap().is_none());
// Connect a client.
let client_fd = unsafe {
libc::socket(libc::AF_INET, libc::SOCK_STREAM, 0)
};
assert!(client_fd >= 0);
let addr = libc::sockaddr_in {
sin_family: libc::AF_INET as libc::sa_family_t,
sin_port: port.to_be(),
sin_addr: libc::in_addr {
s_addr: 0x7F000001u32.to_be(),
},
sin_zero: [0; 8],
};
let ret = unsafe {
libc::connect(
client_fd,
&addr as *const _ as *const libc::sockaddr,
std::mem::size_of::<libc::sockaddr_in>() as libc::socklen_t,
)
};
assert_eq!(ret, 0);
// Now accept should return a fd.
let accepted = listener.accept().unwrap();
assert!(accepted.is_some());
unsafe {
libc::close(client_fd);
libc::close(accepted.unwrap());
}
}
}

View file

@ -0,0 +1,158 @@
use std::collections::HashMap;
/// HTTP method.
#[derive(Debug, Clone, Copy, PartialEq)]
pub enum Method {
Get,
Post,
Head,
Unknown,
}
/// A parsed HTTP/1.1 request.
#[derive(Debug, Clone)]
pub struct Request {
pub method: Method,
pub path: String,
pub query: Option<String>,
pub headers: HashMap<String, String>,
}
/// Result of attempting to parse a request from a byte buffer.
pub enum ParseResult {
/// Request fully parsed; returns the request and number of bytes consumed.
Complete(Request, usize),
/// Need more data.
Incomplete,
/// Malformed request.
Error(String),
}
/// Parse an HTTP/1.1 request from a byte buffer.
///
/// Returns `ParseResult::Complete` when the full header has been received
/// (delimited by `\r\n\r\n`).
pub fn parse(buf: &[u8]) -> ParseResult {
// Find the end of headers.
let header_end = match find_header_end(buf) {
Some(pos) => pos,
None => {
if buf.len() > 8192 {
return ParseResult::Error("request too large".into());
}
return ParseResult::Incomplete;
}
};
let header_str = match std::str::from_utf8(&buf[..header_end]) {
Ok(s) => s,
Err(_) => return ParseResult::Error("invalid UTF-8 in headers".into()),
};
let mut lines = header_str.lines();
// Request line: "GET /path HTTP/1.1"
let request_line = match lines.next() {
Some(l) => l,
None => return ParseResult::Error("empty request".into()),
};
let mut parts = request_line.split_whitespace();
let method = match parts.next() {
Some("GET") => Method::Get,
Some("POST") => Method::Post,
Some("HEAD") => Method::Head,
Some(_) => Method::Unknown,
None => return ParseResult::Error("missing method".into()),
};
let raw_path = match parts.next() {
Some(p) => p,
None => return ParseResult::Error("missing path".into()),
};
// Split path and query string.
let (path, query) = match raw_path.split_once('?') {
Some((p, q)) => (p.to_string(), Some(q.to_string())),
None => (raw_path.to_string(), None),
};
// Parse headers.
let mut headers = HashMap::new();
for line in lines {
if line.is_empty() {
break;
}
if let Some((key, value)) = line.split_once(':') {
headers.insert(
key.trim().to_lowercase(),
value.trim().to_string(),
);
}
}
// Bytes consumed: header + \r\n\r\n delimiter.
let consumed = header_end + 4;
ParseResult::Complete(Request { method, path, query, headers }, consumed)
}
fn find_header_end(buf: &[u8]) -> Option<usize> {
for i in 0..buf.len().saturating_sub(3) {
if &buf[i..i + 4] == b"\r\n\r\n" {
return Some(i);
}
}
None
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn parse_simple_get() {
let raw = b"GET /blog/linux-misc HTTP/1.1\r\nHost: localhost\r\n\r\n";
match parse(raw) {
ParseResult::Complete(req, consumed) => {
assert_eq!(req.method, Method::Get);
assert_eq!(req.path, "/blog/linux-misc");
assert!(req.query.is_none());
assert_eq!(req.headers.get("host").unwrap(), "localhost");
assert_eq!(consumed, raw.len());
}
_ => panic!("expected Complete"),
}
}
#[test]
fn parse_with_query_string() {
let raw = b"GET /tag/rust?page=2 HTTP/1.1\r\n\r\n";
match parse(raw) {
ParseResult::Complete(req, _) => {
assert_eq!(req.path, "/tag/rust");
assert_eq!(req.query.as_deref(), Some("page=2"));
}
_ => panic!("expected Complete"),
}
}
#[test]
fn parse_incomplete() {
let raw = b"GET / HTTP/1.1\r\nHost: local";
assert!(matches!(parse(raw), ParseResult::Incomplete));
}
#[test]
fn parse_multiple_headers() {
let raw = b"GET / HTTP/1.1\r\nHost: localhost\r\nAccept: text/html\r\nConnection: keep-alive\r\n\r\n";
match parse(raw) {
ParseResult::Complete(req, _) => {
assert_eq!(req.headers.len(), 3);
assert_eq!(req.headers.get("accept").unwrap(), "text/html");
assert_eq!(req.headers.get("connection").unwrap(), "keep-alive");
}
_ => panic!("expected Complete"),
}
}
}

View file

@ -0,0 +1,110 @@
/// HTTP response builder.
#[derive(Debug, Clone)]
pub struct Response {
pub status: u16,
pub status_text: String,
pub headers: Vec<(String, String)>,
pub body: Vec<u8>,
}
impl Response {
pub fn new(status: u16, status_text: &str) -> Self {
Self {
status,
status_text: status_text.to_string(),
headers: vec![],
body: vec![],
}
}
/// 200 OK with HTML body.
pub fn html(body: String) -> Self {
let mut r = Self::new(200, "OK");
r.header("Content-Type", "text/html; charset=utf-8");
r.header("Content-Length", &body.len().to_string());
r.body = body.into_bytes();
r
}
/// 200 OK with a raw body and content type.
pub fn ok(body: Vec<u8>, content_type: &str) -> Self {
let mut r = Self::new(200, "OK");
r.header("Content-Type", content_type);
r.header("Content-Length", &body.len().to_string());
r.body = body;
r
}
/// 404 Not Found.
pub fn not_found() -> Self {
let body = "<h1>404 Not Found</h1>";
let mut r = Self::new(404, "Not Found");
r.header("Content-Type", "text/html; charset=utf-8");
r.header("Content-Length", &body.len().to_string());
r.body = body.as_bytes().to_vec();
r
}
/// 500 Internal Server Error.
pub fn internal_error(msg: &str) -> Self {
let body = format!("<h1>500 Internal Server Error</h1><p>{}</p>", msg);
let mut r = Self::new(500, "Internal Server Error");
r.header("Content-Type", "text/html; charset=utf-8");
r.header("Content-Length", &body.len().to_string());
r.body = body.into_bytes();
r
}
/// Add a header.
pub fn header(&mut self, key: &str, value: &str) -> &mut Self {
self.headers.push((key.to_string(), value.to_string()));
self
}
/// Serialize the response to bytes for writing to a socket.
pub fn to_bytes(&self) -> Vec<u8> {
let mut buf = Vec::with_capacity(256 + self.body.len());
// Status line.
buf.extend_from_slice(
format!("HTTP/1.1 {} {}\r\n", self.status, self.status_text).as_bytes(),
);
// Headers.
for (key, value) in &self.headers {
buf.extend_from_slice(format!("{}: {}\r\n", key, value).as_bytes());
}
// End of headers.
buf.extend_from_slice(b"\r\n");
// Body.
buf.extend_from_slice(&self.body);
buf
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn html_response() {
let r = Response::html("<h1>Hello</h1>".into());
let bytes = r.to_bytes();
let s = String::from_utf8(bytes).unwrap();
assert!(s.starts_with("HTTP/1.1 200 OK\r\n"));
assert!(s.contains("Content-Type: text/html"));
assert!(s.contains("<h1>Hello</h1>"));
}
#[test]
fn not_found_response() {
let r = Response::not_found();
assert_eq!(r.status, 404);
let bytes = r.to_bytes();
let s = String::from_utf8(bytes).unwrap();
assert!(s.contains("404 Not Found"));
}
}

View file

@ -0,0 +1,13 @@
[package]
name = "wo-index"
version = "0.1.0"
edition = "2021"
[dependencies]
wo-model = { path = "../wo-model" }
libc = { workspace = true }
bincode = { workspace = true }
serde = { workspace = true }
[dev-dependencies]
tempfile = "3"

194
crates/wo-index/src/date.rs Normal file
View file

@ -0,0 +1,194 @@
use std::io;
use std::path::Path;
/// Sorted index mapping timestamps to byte offsets in a .seg file.
///
/// On-disk format:
/// ```text
/// [8 bytes] entry_count: u64 LE
/// For each entry (sorted by timestamp ascending):
/// [8 bytes] timestamp: i64 LE
/// [8 bytes] offset: u64 LE
/// ```
const ENTRY_SIZE: usize = 16; // i64 + u64
const HEADER_SIZE: usize = 8;
pub struct DateIndex {
data: Vec<u8>,
count: usize,
}
impl DateIndex {
/// Build a date index. Entries are sorted by timestamp before writing.
pub fn build(path: &Path, entries: &mut [(i64, u64)]) -> io::Result<()> {
entries.sort_by_key(|e| e.0);
let file_size = HEADER_SIZE + entries.len() * ENTRY_SIZE;
let mut data = vec![0u8; file_size];
data[0..8].copy_from_slice(&(entries.len() as u64).to_le_bytes());
for (i, &(ts, offset)) in entries.iter().enumerate() {
let base = HEADER_SIZE + i * ENTRY_SIZE;
data[base..base + 8].copy_from_slice(&ts.to_le_bytes());
data[base + 8..base + 16].copy_from_slice(&offset.to_le_bytes());
}
std::fs::write(path, &data)?;
Ok(())
}
/// Open a date index from disk.
pub fn open(path: &Path) -> io::Result<Self> {
let data = std::fs::read(path)?;
if data.len() < HEADER_SIZE {
return Err(io::Error::new(io::ErrorKind::InvalidData, "date.idx too small"));
}
let count = u64::from_le_bytes(data[0..8].try_into().unwrap()) as usize;
Ok(Self { data, count })
}
/// Get offsets for all entries with timestamps in [start, end].
pub fn range(&self, start: i64, end: i64) -> Vec<u64> {
let lo = self.lower_bound(start);
let hi = self.upper_bound(end);
(lo..hi).map(|i| self.offset_at(i)).collect()
}
/// Get offsets for the N most recent entries (highest timestamps).
pub fn latest(&self, n: usize) -> Vec<u64> {
let start = self.count.saturating_sub(n);
(start..self.count)
.rev()
.map(|i| self.offset_at(i))
.collect()
}
/// Number of entries.
pub fn len(&self) -> usize {
self.count
}
pub fn is_empty(&self) -> bool {
self.count == 0
}
fn timestamp_at(&self, i: usize) -> i64 {
let base = HEADER_SIZE + i * ENTRY_SIZE;
i64::from_le_bytes(self.data[base..base + 8].try_into().unwrap())
}
fn offset_at(&self, i: usize) -> u64 {
let base = HEADER_SIZE + i * ENTRY_SIZE;
u64::from_le_bytes(self.data[base + 8..base + 16].try_into().unwrap())
}
/// Binary search: first index where timestamp >= target.
fn lower_bound(&self, target: i64) -> usize {
let (mut lo, mut hi) = (0usize, self.count);
while lo < hi {
let mid = lo + (hi - lo) / 2;
if self.timestamp_at(mid) < target {
lo = mid + 1;
} else {
hi = mid;
}
}
lo
}
/// Binary search: first index where timestamp > target.
fn upper_bound(&self, target: i64) -> usize {
let (mut lo, mut hi) = (0usize, self.count);
while lo < hi {
let mid = lo + (hi - lo) / 2;
if self.timestamp_at(mid) <= target {
lo = mid + 1;
} else {
hi = mid;
}
}
lo
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn build_and_range_query() {
let tmp = tempfile::tempdir().unwrap();
let path = tmp.path().join("date.idx");
let mut entries = vec![
(1000i64, 100u64),
(2000, 200),
(3000, 300),
(4000, 400),
(5000, 500),
];
DateIndex::build(&path, &mut entries).unwrap();
let idx = DateIndex::open(&path).unwrap();
assert_eq!(idx.len(), 5);
// Full range.
let all = idx.range(0, 9999);
assert_eq!(all, vec![100, 200, 300, 400, 500]);
// Partial range.
let mid = idx.range(2000, 4000);
assert_eq!(mid, vec![200, 300, 400]);
// Single.
let one = idx.range(3000, 3000);
assert_eq!(one, vec![300]);
// Empty range.
let none = idx.range(6000, 9000);
assert!(none.is_empty());
}
#[test]
fn latest_entries() {
let tmp = tempfile::tempdir().unwrap();
let path = tmp.path().join("date.idx");
let mut entries = vec![
(1000i64, 100u64),
(2000, 200),
(3000, 300),
];
DateIndex::build(&path, &mut entries).unwrap();
let idx = DateIndex::open(&path).unwrap();
let top2 = idx.latest(2);
assert_eq!(top2, vec![300, 200]);
let top10 = idx.latest(10);
assert_eq!(top10, vec![300, 200, 100]);
}
#[test]
fn unsorted_input() {
let tmp = tempfile::tempdir().unwrap();
let path = tmp.path().join("date.idx");
let mut entries = vec![
(5000i64, 500u64),
(1000, 100),
(3000, 300),
];
DateIndex::build(&path, &mut entries).unwrap();
let idx = DateIndex::open(&path).unwrap();
// Should be sorted on disk.
let all = idx.range(0, 9999);
assert_eq!(all, vec![100, 300, 500]);
}
}

View file

@ -0,0 +1,40 @@
mod title;
mod date;
mod tags;
pub use title::TitleIndex;
pub use date::DateIndex;
pub use tags::TagIndex;
use std::io;
use std::path::Path;
/// Rebuild all three index files from an iterator of article metadata.
///
/// Each entry is `(sys_title, published_on_timestamp, tags, seg_offset)`.
pub fn rebuild_all(
index_dir: &Path,
entries: &[(String, i64, Vec<String>, u64)],
) -> io::Result<()> {
std::fs::create_dir_all(index_dir)?;
// Build title index.
let title_entries: Vec<(&str, u64)> = entries.iter().map(|e| (e.0.as_str(), e.3)).collect();
TitleIndex::build(&index_dir.join("title.idx"), &title_entries)?;
// Build date index.
let mut date_entries: Vec<(i64, u64)> = entries.iter().map(|e| (e.1, e.3)).collect();
DateIndex::build(&index_dir.join("date.idx"), &mut date_entries)?;
// Build tags index.
let mut tag_map: std::collections::HashMap<String, Vec<u64>> = std::collections::HashMap::new();
for entry in entries {
for tag in &entry.2 {
tag_map.entry(tag.clone()).or_default().push(entry.3);
}
}
let tag_entries: Vec<(String, Vec<u64>)> = tag_map.into_iter().collect();
TagIndex::build(&index_dir.join("tags.idx"), &tag_entries)?;
Ok(())
}

View file

@ -0,0 +1,97 @@
use std::collections::HashMap;
use std::io;
use std::path::Path;
use serde::{Deserialize, Serialize};
/// Inverted index mapping tag strings to lists of .seg byte offsets.
///
/// Serialized to disk using bincode for simplicity (article count is
/// small enough that loading the full index into memory is fine).
#[derive(Debug, Serialize, Deserialize)]
struct TagIndexData {
tags: HashMap<String, Vec<u64>>,
}
pub struct TagIndex {
tags: HashMap<String, Vec<u64>>,
}
impl TagIndex {
/// Build a tags index from a list of (tag, offsets) pairs.
pub fn build(path: &Path, entries: &[(String, Vec<u64>)]) -> io::Result<()> {
let tags: HashMap<String, Vec<u64>> = entries.iter().cloned().collect();
let data = TagIndexData { tags };
let bytes = bincode::serialize(&data).map_err(|e| {
io::Error::new(io::ErrorKind::InvalidData, format!("bincode: {}", e))
})?;
std::fs::write(path, bytes)?;
Ok(())
}
/// Open a tags index from disk.
pub fn open(path: &Path) -> io::Result<Self> {
let bytes = std::fs::read(path)?;
let data: TagIndexData = bincode::deserialize(&bytes).map_err(|e| {
io::Error::new(io::ErrorKind::InvalidData, format!("bincode: {}", e))
})?;
Ok(Self { tags: data.tags })
}
/// Get all offsets for articles tagged with the given tag.
pub fn get(&self, tag: &str) -> Option<&[u64]> {
self.tags.get(tag).map(|v| v.as_slice())
}
/// All tag names in the index.
pub fn tags(&self) -> Vec<&str> {
self.tags.keys().map(|s| s.as_str()).collect()
}
/// Number of distinct tags.
pub fn len(&self) -> usize {
self.tags.len()
}
pub fn is_empty(&self) -> bool {
self.tags.is_empty()
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn build_and_lookup() {
let tmp = tempfile::tempdir().unwrap();
let path = tmp.path().join("tags.idx");
let entries = vec![
("rust".to_string(), vec![100u64, 200]),
("linux".to_string(), vec![300]),
("tutorial".to_string(), vec![100, 300, 400]),
];
TagIndex::build(&path, &entries).unwrap();
let idx = TagIndex::open(&path).unwrap();
assert_eq!(idx.len(), 3);
assert_eq!(idx.get("rust"), Some(vec![100u64, 200].as_slice()));
assert_eq!(idx.get("linux"), Some(vec![300u64].as_slice()));
assert_eq!(idx.get("tutorial"), Some(vec![100u64, 300, 400].as_slice()));
assert_eq!(idx.get("nonexistent"), None);
}
#[test]
fn empty_index() {
let tmp = tempfile::tempdir().unwrap();
let path = tmp.path().join("tags.idx");
TagIndex::build(&path, &[]).unwrap();
let idx = TagIndex::open(&path).unwrap();
assert!(idx.is_empty());
assert_eq!(idx.get("anything"), None);
}
}

View file

@ -0,0 +1,183 @@
use std::collections::hash_map::DefaultHasher;
use std::hash::{Hash, Hasher};
use std::io;
use std::path::Path;
/// On-disk hash index mapping `sys_title` (string) to byte offset in a .seg file.
///
/// Uses open addressing with linear probing. The table is sized at 2x the entry
/// count (load factor 0.5) to keep collision chains short.
///
/// On-disk format:
/// ```text
/// [8 bytes] table_size: u64 LE (number of slots)
/// [8 bytes] entry_count: u64 LE
/// For each slot (table_size slots):
/// [8 bytes] offset: u64 LE (0 = empty, otherwise seg offset + 1)
/// [2 bytes] key_len: u16 LE
/// [128 bytes] key_data: zero-padded sys_title
/// Total slot size: 138 bytes
/// ```
const SLOT_SIZE: usize = 8 + 2 + 128; // offset + key_len + key_data
const HEADER_SIZE: usize = 16; // table_size + entry_count
const MAX_KEY_LEN: usize = 128;
pub struct TitleIndex {
data: Vec<u8>,
table_size: u64,
}
impl TitleIndex {
/// Build a title index file from a list of (sys_title, seg_offset) pairs.
pub fn build(path: &Path, entries: &[(&str, u64)]) -> io::Result<()> {
let table_size = (entries.len() * 2).max(16) as u64;
let file_size = HEADER_SIZE + (table_size as usize) * SLOT_SIZE;
let mut data = vec![0u8; file_size];
// Write header.
data[0..8].copy_from_slice(&table_size.to_le_bytes());
data[8..16].copy_from_slice(&(entries.len() as u64).to_le_bytes());
// Insert entries.
for &(key, offset) in entries {
let slot = Self::find_empty_slot(&data, table_size, key);
Self::write_slot(&mut data, slot, key, offset);
}
std::fs::write(path, &data)?;
Ok(())
}
/// Open a title index from disk (loads into memory).
pub fn open(path: &Path) -> io::Result<Self> {
let data = std::fs::read(path)?;
if data.len() < HEADER_SIZE {
return Err(io::Error::new(io::ErrorKind::InvalidData, "title.idx too small"));
}
let table_size = u64::from_le_bytes(data[0..8].try_into().unwrap());
Ok(Self { data, table_size })
}
/// Look up a sys_title, returning the seg file byte offset if found.
pub fn get(&self, sys_title: &str) -> Option<u64> {
let mut slot = Self::hash_key(sys_title) % self.table_size;
for _ in 0..self.table_size {
let (stored_offset, stored_key) = self.read_slot(slot as usize);
if stored_offset == 0 {
return None; // empty slot, key not found
}
if stored_key == sys_title {
return Some(stored_offset - 1); // stored as offset + 1
}
slot = (slot + 1) % self.table_size;
}
None
}
/// Number of entries in the index.
pub fn len(&self) -> usize {
u64::from_le_bytes(self.data[8..16].try_into().unwrap()) as usize
}
pub fn is_empty(&self) -> bool {
self.len() == 0
}
fn hash_key(key: &str) -> u64 {
let mut hasher = DefaultHasher::new();
key.hash(&mut hasher);
hasher.finish()
}
fn find_empty_slot(data: &[u8], table_size: u64, key: &str) -> usize {
let mut slot = (Self::hash_key(key) % table_size) as usize;
loop {
let base = HEADER_SIZE + slot * SLOT_SIZE;
let stored_offset = u64::from_le_bytes(data[base..base + 8].try_into().unwrap());
if stored_offset == 0 {
return slot;
}
slot = (slot + 1) % table_size as usize;
}
}
fn write_slot(data: &mut [u8], slot: usize, key: &str, offset: u64) {
let base = HEADER_SIZE + slot * SLOT_SIZE;
let stored_offset = offset + 1; // +1 so that 0 means empty
data[base..base + 8].copy_from_slice(&stored_offset.to_le_bytes());
let key_bytes = key.as_bytes();
let key_len = key_bytes.len().min(MAX_KEY_LEN) as u16;
data[base + 8..base + 10].copy_from_slice(&key_len.to_le_bytes());
data[base + 10..base + 10 + key_len as usize].copy_from_slice(&key_bytes[..key_len as usize]);
}
fn read_slot(&self, slot: usize) -> (u64, String) {
let base = HEADER_SIZE + slot * SLOT_SIZE;
let stored_offset = u64::from_le_bytes(self.data[base..base + 8].try_into().unwrap());
let key_len = u16::from_le_bytes(self.data[base + 8..base + 10].try_into().unwrap()) as usize;
let key = String::from_utf8_lossy(&self.data[base + 10..base + 10 + key_len]).to_string();
(stored_offset, key)
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn build_and_lookup() {
let tmp = tempfile::tempdir().unwrap();
let path = tmp.path().join("title.idx");
let entries = vec![
("linux-misc", 32u64),
("aws-lambda-pulumi", 500),
("rust-patterns", 1200),
];
TitleIndex::build(&path, &entries).unwrap();
let idx = TitleIndex::open(&path).unwrap();
assert_eq!(idx.len(), 3);
assert_eq!(idx.get("linux-misc"), Some(32));
assert_eq!(idx.get("aws-lambda-pulumi"), Some(500));
assert_eq!(idx.get("rust-patterns"), Some(1200));
assert_eq!(idx.get("nonexistent"), None);
}
#[test]
fn empty_index() {
let tmp = tempfile::tempdir().unwrap();
let path = tmp.path().join("title.idx");
TitleIndex::build(&path, &[]).unwrap();
let idx = TitleIndex::open(&path).unwrap();
assert!(idx.is_empty());
assert_eq!(idx.get("anything"), None);
}
#[test]
fn many_entries() {
let tmp = tempfile::tempdir().unwrap();
let path = tmp.path().join("title.idx");
let keys: Vec<String> = (0..100).map(|i| format!("article-{}", i)).collect();
let entries: Vec<(&str, u64)> = keys.iter().enumerate().map(|(i, k)| (k.as_str(), i as u64 * 100)).collect();
TitleIndex::build(&path, &entries).unwrap();
let idx = TitleIndex::open(&path).unwrap();
assert_eq!(idx.len(), 100);
for (i, key) in keys.iter().enumerate() {
assert_eq!(idx.get(key), Some(i as u64 * 100));
}
}
}

6
crates/wo-md/Cargo.toml Normal file
View file

@ -0,0 +1,6 @@
[package]
name = "wo-md"
version = "0.1.0"
edition = "2021"
[dependencies]

324
crates/wo-md/src/block.rs Normal file
View file

@ -0,0 +1,324 @@
use crate::highlight::highlight;
use crate::inline::markdown_to_html;
/// Parse a full markdown document into HTML.
///
/// Handles block-level elements (headings, paragraphs, code fences, lists,
/// blockquotes, images) and delegates inline formatting to `markdown_to_html`.
pub fn markdown_to_html_block(md: &str) -> String {
let lines: Vec<&str> = md.lines().collect();
let len = lines.len();
let mut html = String::with_capacity(md.len() * 2);
let mut i = 0;
while i < len {
let line = lines[i];
let trimmed = line.trim();
// Skip blank lines.
if trimmed.is_empty() {
i += 1;
continue;
}
// Fenced code block: ```lang ... ```
if trimmed.starts_with("```") {
let lang = trimmed[3..].trim();
i += 1;
let mut code = String::new();
while i < len && !lines[i].trim().starts_with("```") {
if !code.is_empty() {
code.push('\n');
}
code.push_str(lines[i]);
i += 1;
}
if i < len {
i += 1; // skip closing ```
}
if lang.is_empty() {
html.push_str("<pre><code>");
html.push_str(&escape_html(&code));
html.push_str("</code></pre>\n");
} else {
html.push_str(&format!("<pre><code class=\"language-{}\">", escape_html(lang)));
html.push_str(&highlight(&code, lang));
html.push_str("</code></pre>\n");
}
continue;
}
// Heading: # ... ######
if trimmed.starts_with('#') {
let level = trimmed.chars().take_while(|&c| c == '#').count().min(6);
let text = trimmed[level..].trim();
html.push_str(&format!(
"<h{0}>{1}</h{0}>\n",
level,
markdown_to_html(text)
));
i += 1;
continue;
}
// Blockquote: > text
if trimmed.starts_with('>') {
let mut quote_lines = Vec::new();
while i < len && lines[i].trim().starts_with('>') {
let content = lines[i].trim().strip_prefix('>').unwrap_or("").trim();
quote_lines.push(content);
i += 1;
}
html.push_str("<blockquote>");
html.push_str(&markdown_to_html(&quote_lines.join(" ")));
html.push_str("</blockquote>\n");
continue;
}
// Unordered list: - item or * item
if (trimmed.starts_with("- ") || trimmed.starts_with("* "))
&& !trimmed.starts_with("---")
{
html.push_str("<ul>\n");
while i < len {
let lt = lines[i].trim();
if lt.starts_with("- ") || lt.starts_with("* ") {
let text = &lt[2..];
html.push_str(&format!("<li>{}</li>\n", markdown_to_html(text)));
i += 1;
} else if lt.is_empty() {
i += 1;
break;
} else {
break;
}
}
html.push_str("</ul>\n");
continue;
}
// Ordered list: 1. item
if trimmed.len() > 2 && trimmed.as_bytes()[0].is_ascii_digit() {
if let Some(rest) = strip_ordered_prefix(trimmed) {
html.push_str("<ol>\n");
html.push_str(&format!("<li>{}</li>\n", markdown_to_html(rest)));
i += 1;
while i < len {
let lt = lines[i].trim();
if let Some(rest) = strip_ordered_prefix(lt) {
html.push_str(&format!("<li>{}</li>\n", markdown_to_html(rest)));
i += 1;
} else if lt.is_empty() {
i += 1;
break;
} else {
break;
}
}
html.push_str("</ol>\n");
continue;
}
}
// Image: ![alt](url)
if trimmed.starts_with("![") {
if let Some((alt, url)) = parse_image(trimmed) {
html.push_str(&format!(
"<img src=\"{}\" alt=\"{}\">\n",
escape_html(url),
escape_html(alt)
));
i += 1;
continue;
}
}
// Horizontal rule: --- or ***
if trimmed == "---" || trimmed == "***" || trimmed == "___" {
html.push_str("<hr>\n");
i += 1;
continue;
}
// Paragraph: collect consecutive non-blank, non-block lines.
let mut para_lines = Vec::new();
while i < len {
let lt = lines[i].trim();
if lt.is_empty()
|| lt.starts_with('#')
|| lt.starts_with("```")
|| lt.starts_with('>')
|| lt == "---"
|| lt == "***"
|| lt == "___"
{
break;
}
// Check if next line starts a list
if (lt.starts_with("- ") || lt.starts_with("* ")) && !lt.starts_with("---") {
break;
}
if strip_ordered_prefix(lt).is_some() && para_lines.is_empty() {
break;
}
para_lines.push(lt);
i += 1;
}
if !para_lines.is_empty() {
html.push_str("<p>");
html.push_str(&markdown_to_html(&para_lines.join(" ")));
html.push_str("</p>\n");
}
}
html
}
fn escape_html(s: &str) -> String {
s.replace('&', "&amp;")
.replace('<', "&lt;")
.replace('>', "&gt;")
.replace('"', "&quot;")
}
fn strip_ordered_prefix(s: &str) -> Option<&str> {
let dot_pos = s.find(". ")?;
if dot_pos > 0 && s[..dot_pos].chars().all(|c| c.is_ascii_digit()) {
Some(&s[dot_pos + 2..])
} else {
None
}
}
fn parse_image(s: &str) -> Option<(&str, &str)> {
// ![alt](url)
let alt_start = s.find("![")? + 2;
let alt_end = s[alt_start..].find(']')? + alt_start;
let url_start = s[alt_end..].find('(')? + alt_end + 1;
let url_end = s[url_start..].find(')')? + url_start;
Some((&s[alt_start..alt_end], &s[url_start..url_end]))
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn headings() {
let md = "# Title\n\n## Subtitle\n\n### Third";
let html = markdown_to_html_block(md);
assert!(html.contains("<h1>Title</h1>"));
assert!(html.contains("<h2>Subtitle</h2>"));
assert!(html.contains("<h3>Third</h3>"));
}
#[test]
fn paragraphs() {
let md = "First paragraph.\n\nSecond paragraph.";
let html = markdown_to_html_block(md);
assert!(html.contains("<p>First paragraph.</p>"));
assert!(html.contains("<p>Second paragraph.</p>"));
}
#[test]
fn code_fence_with_language() {
let md = "```rust\nfn main() {\n let x = 42;\n}\n```";
let html = markdown_to_html_block(md);
assert!(html.contains("<pre><code class=\"language-rust\">"));
assert!(html.contains("<span class=\"kw\">fn</span>"));
assert!(html.contains("<span class=\"num\">42</span>"));
assert!(html.contains("</code></pre>"));
}
#[test]
fn code_fence_no_language() {
let md = "```\nplain code\n```";
let html = markdown_to_html_block(md);
assert!(html.contains("<pre><code>plain code</code></pre>"));
}
#[test]
fn unordered_list() {
let md = "- Item one\n- Item two\n- Item three";
let html = markdown_to_html_block(md);
assert!(html.contains("<ul>"));
assert!(html.contains("<li>Item one</li>"));
assert!(html.contains("<li>Item two</li>"));
assert!(html.contains("</ul>"));
}
#[test]
fn ordered_list() {
let md = "1. First\n2. Second\n3. Third";
let html = markdown_to_html_block(md);
assert!(html.contains("<ol>"));
assert!(html.contains("<li>First</li>"));
assert!(html.contains("<li>Third</li>"));
assert!(html.contains("</ol>"));
}
#[test]
fn blockquote() {
let md = "> This is a quote\n> spanning two lines";
let html = markdown_to_html_block(md);
assert!(html.contains("<blockquote>This is a quote spanning two lines</blockquote>"));
}
#[test]
fn image() {
let md = "![Alt text](https://example.com/img.png)";
let html = markdown_to_html_block(md);
assert!(html.contains("<img src=\"https://example.com/img.png\" alt=\"Alt text\">"));
}
#[test]
fn horizontal_rule() {
let md = "Before\n\n---\n\nAfter";
let html = markdown_to_html_block(md);
assert!(html.contains("<hr>"));
}
#[test]
fn inline_formatting_in_paragraphs() {
let md = "Use **bold** and `code` in a paragraph.";
let html = markdown_to_html_block(md);
assert!(html.contains("<strong>bold</strong>"));
assert!(html.contains("<code>code</code>"));
}
#[test]
fn full_article() {
let md = "\
# Getting Started
Welcome to the guide.
## Installation
Install with cargo:
```bash
cargo install writeonce
```
## Features
- Fast rendering
- Zero dependencies
- **Bold** feature
> Note: this is a quote.
![Logo](logo.png)
";
let html = markdown_to_html_block(md);
assert!(html.contains("<h1>Getting Started</h1>"));
assert!(html.contains("<h2>Installation</h2>"));
assert!(html.contains("<p>Install with cargo:</p>"));
assert!(html.contains("<pre><code class=\"language-bash\">"));
assert!(html.contains("<li>Fast rendering</li>"));
assert!(html.contains("<blockquote>"));
assert!(html.contains("<img src=\"logo.png\""));
}
}

View file

@ -0,0 +1,154 @@
/// Server-side syntax highlighting for code blocks.
///
/// Returns HTML with `<span class="...">` tokens. Pair with code-theme.css.
pub fn highlight(code: &str, language: &str) -> String {
let keywords = keywords_for(language);
let mut result = String::with_capacity(code.len() * 2);
let chars: Vec<char> = code.chars().collect();
let len = chars.len();
let mut i = 0;
while i < len {
// String literals (double or single quotes).
if chars[i] == '"' || chars[i] == '\'' {
let quote = chars[i];
let start = i;
i += 1;
while i < len && chars[i] != quote {
if chars[i] == '\\' { i += 1; } // skip escaped char
i += 1;
}
if i < len { i += 1; } // closing quote
let s: String = chars[start..i].iter().collect();
result.push_str(&format!("<span class=\"str\">{}</span>", escape(&s)));
continue;
}
// Line comments.
if i + 1 < len && chars[i] == '/' && chars[i + 1] == '/' {
let start = i;
while i < len && chars[i] != '\n' { i += 1; }
let s: String = chars[start..i].iter().collect();
result.push_str(&format!("<span class=\"cm\">{}</span>", escape(&s)));
continue;
}
// Hash comments (bash, yaml, etc.).
if chars[i] == '#' && (language == "bash" || language == "yaml" || language == "python") {
let start = i;
while i < len && chars[i] != '\n' { i += 1; }
let s: String = chars[start..i].iter().collect();
result.push_str(&format!("<span class=\"cm\">{}</span>", escape(&s)));
continue;
}
// Numbers.
if chars[i].is_ascii_digit() {
let start = i;
while i < len && (chars[i].is_ascii_alphanumeric() || chars[i] == '.') { i += 1; }
let s: String = chars[start..i].iter().collect();
result.push_str(&format!("<span class=\"num\">{}</span>", escape(&s)));
continue;
}
// Identifiers / keywords.
if chars[i].is_ascii_alphabetic() || chars[i] == '_' {
let start = i;
while i < len && (chars[i].is_ascii_alphanumeric() || chars[i] == '_') { i += 1; }
let word: String = chars[start..i].iter().collect();
if keywords.contains(&word.as_str()) {
result.push_str(&format!("<span class=\"kw\">{}</span>", escape(&word)));
} else {
result.push_str(&escape(&word));
}
continue;
}
// Everything else.
result.push_str(&escape(&chars[i].to_string()));
i += 1;
}
result
}
fn escape(s: &str) -> String {
s.replace('&', "&amp;")
.replace('<', "&lt;")
.replace('>', "&gt;")
}
fn keywords_for(language: &str) -> &'static [&'static str] {
match language {
"rust" => &[
"fn", "let", "mut", "const", "struct", "enum", "impl", "trait", "pub", "use",
"mod", "crate", "self", "super", "return", "if", "else", "match", "for", "while",
"loop", "break", "continue", "where", "type", "as", "in", "ref", "move",
"async", "await", "unsafe", "extern", "dyn", "true", "false",
],
"go" => &[
"func", "var", "const", "type", "struct", "interface", "map", "chan",
"package", "import", "return", "if", "else", "for", "range", "switch",
"case", "default", "go", "defer", "select", "true", "false", "nil",
],
"bash" | "sh" => &[
"if", "then", "else", "elif", "fi", "for", "while", "do", "done",
"case", "esac", "function", "return", "export", "local", "echo",
"in", "true", "false",
],
"yaml" => &["true", "false", "null", "yes", "no"],
"json" => &["true", "false", "null"],
"python" => &[
"def", "class", "if", "elif", "else", "for", "while", "return",
"import", "from", "as", "with", "try", "except", "finally",
"raise", "pass", "lambda", "yield", "True", "False", "None", "in",
"not", "and", "or", "is", "async", "await",
],
_ => &[],
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn highlight_rust() {
let code = "fn main() {\n let x = 42;\n}";
let html = highlight(code, "rust");
assert!(html.contains("<span class=\"kw\">fn</span>"));
assert!(html.contains("<span class=\"kw\">let</span>"));
assert!(html.contains("<span class=\"num\">42</span>"));
}
#[test]
fn highlight_string() {
let code = r#"let s = "hello";"#;
let html = highlight(code, "rust");
assert!(html.contains("<span class=\"str\">\"hello\"</span>"));
}
#[test]
fn highlight_comment() {
let code = "// this is a comment\nlet x = 1;";
let html = highlight(code, "rust");
assert!(html.contains("<span class=\"cm\">// this is a comment</span>"));
}
#[test]
fn highlight_bash() {
let code = "# comment\nexport PATH=/usr/bin";
let html = highlight(code, "bash");
assert!(html.contains("<span class=\"cm\"># comment</span>"));
assert!(html.contains("<span class=\"kw\">export</span>"));
}
#[test]
fn escapes_html() {
let code = "fn compare<T>(a: T, b: T) {}";
let html = highlight(code, "rust");
assert!(html.contains("&lt;"));
assert!(html.contains("&gt;"));
assert!(!html.contains("<T>"));
}
}

171
crates/wo-md/src/inline.rs Normal file
View file

@ -0,0 +1,171 @@
/// Convert inline markdown to HTML.
///
/// Handles: **bold**, *italic*, `code`, [text](url), and HTML entity escaping.
/// Does not handle block-level elements (headers, lists, etc.) — those are
/// already structured in the article JSON.
pub fn markdown_to_html(text: &str) -> String {
let mut result = String::with_capacity(text.len() * 2);
let chars: Vec<char> = text.chars().collect();
let len = chars.len();
let mut i = 0;
while i < len {
match chars[i] {
// HTML entity escaping.
'&' => { result.push_str("&amp;"); i += 1; }
'<' => { result.push_str("&lt;"); i += 1; }
'>' => { result.push_str("&gt;"); i += 1; }
// **bold** or *italic*
'*' => {
if i + 1 < len && chars[i + 1] == '*' {
// **bold**
if let Some(end) = find_closing(&chars, i + 2, "**") {
result.push_str("<strong>");
let inner: String = chars[i + 2..end].iter().collect();
result.push_str(&escape_html(&inner));
result.push_str("</strong>");
i = end + 2;
} else {
result.push('*');
i += 1;
}
} else {
// *italic*
if let Some(end) = find_closing_char(&chars, i + 1, '*') {
result.push_str("<em>");
let inner: String = chars[i + 1..end].iter().collect();
result.push_str(&escape_html(&inner));
result.push_str("</em>");
i = end + 1;
} else {
result.push('*');
i += 1;
}
}
}
// `inline code`
'`' => {
if let Some(end) = find_closing_char(&chars, i + 1, '`') {
result.push_str("<code>");
let inner: String = chars[i + 1..end].iter().collect();
result.push_str(&escape_html(&inner));
result.push_str("</code>");
i = end + 1;
} else {
result.push('`');
i += 1;
}
}
// [text](url)
'[' => {
if let Some((text_end, url_start, url_end)) = parse_link(&chars, i) {
let link_text: String = chars[i + 1..text_end].iter().collect();
let url: String = chars[url_start..url_end].iter().collect();
result.push_str(&format!(
"<a href=\"{}\">{}</a>",
escape_html(&url),
escape_html(&link_text)
));
i = url_end + 1;
} else {
result.push('[');
i += 1;
}
}
c => {
result.push(c);
i += 1;
}
}
}
result
}
fn escape_html(s: &str) -> String {
s.replace('&', "&amp;")
.replace('<', "&lt;")
.replace('>', "&gt;")
.replace('"', "&quot;")
}
fn find_closing(chars: &[char], start: usize, marker: &str) -> Option<usize> {
let marker_chars: Vec<char> = marker.chars().collect();
let mlen = marker_chars.len();
for i in start..chars.len().saturating_sub(mlen - 1) {
if chars[i..i + mlen] == marker_chars[..] {
return Some(i);
}
}
None
}
fn find_closing_char(chars: &[char], start: usize, marker: char) -> Option<usize> {
for i in start..chars.len() {
if chars[i] == marker {
return Some(i);
}
}
None
}
fn parse_link(chars: &[char], start: usize) -> Option<(usize, usize, usize)> {
// [text](url)
let text_end = find_closing_char(chars, start + 1, ']')?;
if text_end + 1 >= chars.len() || chars[text_end + 1] != '(' {
return None;
}
let url_start = text_end + 2;
let url_end = find_closing_char(chars, url_start, ')')?;
Some((text_end, url_start, url_end))
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn plain_text() {
assert_eq!(markdown_to_html("hello world"), "hello world");
}
#[test]
fn bold() {
assert_eq!(markdown_to_html("**bold**"), "<strong>bold</strong>");
}
#[test]
fn italic() {
assert_eq!(markdown_to_html("*italic*"), "<em>italic</em>");
}
#[test]
fn inline_code() {
assert_eq!(markdown_to_html("`code`"), "<code>code</code>");
}
#[test]
fn link() {
assert_eq!(
markdown_to_html("[click](https://example.com)"),
"<a href=\"https://example.com\">click</a>"
);
}
#[test]
fn mixed() {
assert_eq!(
markdown_to_html("Use **Rust** with `cargo` for [docs](https://doc.rust-lang.org)"),
"Use <strong>Rust</strong> with <code>cargo</code> for <a href=\"https://doc.rust-lang.org\">docs</a>"
);
}
#[test]
fn html_escaping() {
assert_eq!(markdown_to_html("<script>alert('xss')</script>"), "&lt;script&gt;alert('xss')&lt;/script&gt;");
}
}

7
crates/wo-md/src/lib.rs Normal file
View file

@ -0,0 +1,7 @@
mod block;
mod inline;
mod highlight;
pub use block::markdown_to_html_block;
pub use inline::markdown_to_html;
pub use highlight::highlight;

View file

@ -0,0 +1,13 @@
[package]
name = "wo-model"
version = "0.1.0"
edition = "2021"
[dependencies]
serde = { workspace = true }
serde_json = { workspace = true }
bincode = { workspace = true }
wo-md = { path = "../wo-md" }
[dev-dependencies]
tempfile = "3"

View file

@ -0,0 +1,177 @@
use serde::{Deserialize, Serialize};
/// Markdown-first article: minimal metadata + pre-rendered HTML content.
///
/// The `sys_title` field is the primary key — a URL-safe slug
/// used for lookups and routing (e.g., "linux-misc").
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
pub struct Article {
pub sys_title: String,
pub title: String,
pub published: bool,
pub author: String,
#[serde(default)]
pub tags: Vec<String>,
#[serde(default)]
pub published_on: Option<i64>,
/// Pre-rendered HTML from the .md file. Populated by ContentLoader.
#[serde(default)]
pub content_html: String,
}
// --- Legacy types for backwards compatibility with old JSON format ---
/// Legacy article format (nested JSON with sections/codes/etc.).
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct LegacyArticle {
pub title: String,
pub sys_title: String,
pub published: bool,
pub content: LegacyArticleContent,
#[serde(default)]
pub do_aws_sync: Option<bool>,
#[serde(default)]
pub published_on: Option<i64>,
}
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct LegacyArticleContent {
pub author: String,
pub content: LegacyArticleBody,
#[serde(default, rename = "publishedOn")]
pub published_on: Option<i64>,
#[serde(default)]
pub references: Vec<LegacyReference>,
#[serde(default)]
pub tags: Vec<String>,
#[serde(default)]
pub title: Option<String>,
#[serde(default)]
pub systitle: Option<String>,
}
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct LegacyArticleBody {
#[serde(default)]
pub sections: Vec<LegacySection>,
#[serde(default)]
pub codes: Vec<serde_json::Value>,
#[serde(default)]
pub images: Vec<serde_json::Value>,
#[serde(default)]
pub img: Option<serde_json::Value>,
}
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct LegacySection {
pub heading: String,
#[serde(default)]
pub paragraphs: Vec<String>,
}
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct LegacyReference {
#[serde(default, rename = "dateAccessed")]
pub date_accessed: Option<i64>,
#[serde(default)]
pub title: Option<String>,
#[serde(default)]
pub url: Option<String>,
}
impl LegacyArticle {
/// Convert a legacy article to the new format by building HTML from sections.
pub fn to_article(&self) -> Article {
let mut html = String::new();
for section in &self.content.content.sections {
html.push_str(&format!("<h2>{}</h2>\n", section.heading));
for para in &section.paragraphs {
if !para.is_empty() {
html.push_str(&format!("<p>{}</p>\n", wo_md::markdown_to_html(para)));
}
}
}
Article {
sys_title: self.sys_title.clone(),
title: self.title.clone(),
published: self.published,
author: self.content.author.clone(),
tags: self.content.tags.clone(),
published_on: self.published_on.or(self.content.published_on),
content_html: html,
}
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn deserialize_new_format() {
let json = r#"{
"sys_title": "test-article",
"title": "Test Article",
"published": true,
"author": "Author",
"tags": ["test"],
"published_on": 1000
}"#;
let article: Article = serde_json::from_str(json).unwrap();
assert_eq!(article.sys_title, "test-article");
assert_eq!(article.author, "Author");
assert_eq!(article.tags, vec!["test"]);
assert!(article.published);
assert_eq!(article.content_html, "");
}
#[test]
fn round_trip_bincode() {
let article = Article {
sys_title: "test".into(),
title: "Test".into(),
published: true,
author: "Author".into(),
tags: vec!["rust".into()],
published_on: Some(1000),
content_html: "<h1>Hello</h1>".into(),
};
let bytes = bincode::serialize(&article).unwrap();
let deserialized: Article = bincode::deserialize(&bytes).unwrap();
assert_eq!(article, deserialized);
}
#[test]
fn legacy_to_article() {
let json = r#"{
"title": "Legacy Article",
"sys_title": "legacy",
"published": true,
"content": {
"author": "Author",
"content": {
"sections": [
{"heading": "Intro", "paragraphs": ["Hello **world**."]}
],
"codes": [],
"images": []
},
"tags": ["test"],
"publishedOn": 1000
},
"published_on": 2000
}"#;
let legacy: LegacyArticle = serde_json::from_str(json).unwrap();
let article = legacy.to_article();
assert_eq!(article.sys_title, "legacy");
assert_eq!(article.author, "Author");
assert_eq!(article.tags, vec!["test"]);
assert_eq!(article.published_on, Some(2000));
assert!(article.content_html.contains("<h2>Intro</h2>"));
assert!(article.content_html.contains("<strong>world</strong>"));
}
}

View file

@ -0,0 +1,5 @@
mod article;
mod loader;
pub use article::*;
pub use loader::ContentLoader;

View file

@ -0,0 +1,249 @@
use std::fs;
use std::io;
use std::path::Path;
use crate::{Article, LegacyArticle};
/// Walks a content directory and loads all articles.
///
/// Supports two formats:
/// - **New**: minimal `.json` metadata + `.md` file for content
/// - **Legacy**: full JSON with nested sections/codes (falls back if no `.md`)
///
/// ```text
/// content_dir/
/// article-slug/
/// article-slug.json # metadata
/// article-slug.md # content (new format)
/// ```
pub struct ContentLoader;
impl ContentLoader {
/// Load all articles from the given content directory.
pub fn load_all(content_dir: &Path) -> io::Result<Vec<Article>> {
let mut articles = Vec::new();
let entries = fs::read_dir(content_dir)?;
for entry in entries {
let entry = entry?;
let path = entry.path();
if !path.is_dir() {
continue;
}
match Self::load_from_dir(&path) {
Ok(article) => articles.push(article),
Err(e) => {
eprintln!(
"wo-model: skipping {:?}: {}",
path.file_name().unwrap_or_default(),
e
);
}
}
}
articles.sort_by(|a, b| {
b.published_on
.unwrap_or(0)
.cmp(&a.published_on.unwrap_or(0))
});
Ok(articles)
}
/// Load a single article from a directory.
///
/// Tries new format first (minimal JSON + .md), falls back to legacy JSON.
pub fn load_from_dir(dir: &Path) -> io::Result<Article> {
let json_file = Self::find_json_file(dir)?;
let json_contents = fs::read_to_string(&json_file)?;
// Check if a .md file exists alongside the JSON.
let md_file = Self::find_md_file(dir);
if let Some(md_path) = md_file {
// New format: minimal JSON + markdown file.
// Try parsing as new format first, then fall back to legacy.
let mut article: Article =
if let Ok(a) = serde_json::from_str::<Article>(&json_contents) {
a
} else {
let legacy: LegacyArticle =
serde_json::from_str(&json_contents).map_err(|e| {
io::Error::new(
io::ErrorKind::InvalidData,
format!("{}: {}", json_file.display(), e),
)
})?;
legacy.to_article()
};
let md_content = fs::read_to_string(&md_path)?;
article.content_html = wo_md::markdown_to_html_block(&md_content);
Ok(article)
} else {
// Legacy format: full JSON with sections, no .md file.
let legacy: LegacyArticle = serde_json::from_str(&json_contents).map_err(|e| {
io::Error::new(
io::ErrorKind::InvalidData,
format!("{}: {}", json_file.display(), e),
)
})?;
Ok(legacy.to_article())
}
}
/// Load a single article from a specific JSON file path.
pub fn load_from_file(path: &Path) -> io::Result<Article> {
let dir = path.parent().unwrap_or(Path::new("."));
Self::load_from_dir(dir)
}
fn find_json_file(dir: &Path) -> io::Result<std::path::PathBuf> {
for entry in fs::read_dir(dir)? {
let entry = entry?;
let path = entry.path();
if path.extension().and_then(|e| e.to_str()) == Some("json") {
return Ok(path);
}
}
Err(io::Error::new(
io::ErrorKind::NotFound,
format!("no .json file in {}", dir.display()),
))
}
fn find_md_file(dir: &Path) -> Option<std::path::PathBuf> {
let entries = fs::read_dir(dir).ok()?;
for entry in entries {
let entry = entry.ok()?;
let path = entry.path();
if path.extension().and_then(|e| e.to_str()) == Some("md") {
return Some(path);
}
}
None
}
}
#[cfg(test)]
mod tests {
use super::*;
use std::fs;
/// Create a new-format article: minimal JSON + .md file.
fn create_new_article(base: &Path, sys_title: &str, published: bool, tags: &[&str]) {
let dir = base.join(sys_title);
fs::create_dir_all(&dir).unwrap();
let tags_json: Vec<String> = tags.iter().map(|t| format!("\"{}\"", t)).collect();
let json = format!(
r#"{{
"sys_title": "{}",
"title": "{}",
"published": {},
"author": "Test Author",
"tags": [{}],
"published_on": 2000
}}"#,
sys_title, sys_title, published, tags_json.join(", ")
);
fs::write(dir.join(format!("{}.json", sys_title)), json).unwrap();
let md = format!("# {}\n\nThis is the **content** of the article.\n\n```rust\nfn main() {{}}\n```\n", sys_title);
fs::write(dir.join(format!("{}.md", sys_title)), md).unwrap();
}
/// Create a legacy-format article: full JSON, no .md file.
fn create_legacy_article(base: &Path, sys_title: &str, published: bool, tags: &[&str]) {
let dir = base.join(sys_title);
fs::create_dir_all(&dir).unwrap();
let tags_json: Vec<String> = tags.iter().map(|t| format!("\"{}\"", t)).collect();
let json = format!(
r#"{{
"title": "{}",
"sys_title": "{}",
"published": {},
"content": {{
"author": "Test Author",
"content": {{
"sections": [{{ "heading": "Intro", "paragraphs": ["Hello **world**."] }}],
"codes": [],
"images": []
}},
"tags": [{}],
"publishedOn": 1000
}},
"published_on": 2000
}}"#,
sys_title, sys_title, published, tags_json.join(", ")
);
fs::write(dir.join(format!("{}.json", sys_title)), json).unwrap();
}
#[test]
fn load_new_format() {
let tmp = tempfile::tempdir().unwrap();
create_new_article(tmp.path(), "my-article", true, &["rust"]);
let article = ContentLoader::load_from_dir(&tmp.path().join("my-article")).unwrap();
assert_eq!(article.sys_title, "my-article");
assert_eq!(article.author, "Test Author");
assert_eq!(article.tags, vec!["rust"]);
assert!(article.content_html.contains("<h1>my-article</h1>"));
assert!(article.content_html.contains("<strong>content</strong>"));
assert!(article.content_html.contains("<pre><code"));
}
#[test]
fn load_legacy_format() {
let tmp = tempfile::tempdir().unwrap();
create_legacy_article(tmp.path(), "legacy-art", true, &["test"]);
let article = ContentLoader::load_from_dir(&tmp.path().join("legacy-art")).unwrap();
assert_eq!(article.sys_title, "legacy-art");
assert_eq!(article.author, "Test Author");
assert!(article.content_html.contains("<h2>Intro</h2>"));
assert!(article.content_html.contains("<strong>world</strong>"));
}
#[test]
fn load_all_mixed() {
let tmp = tempfile::tempdir().unwrap();
create_new_article(tmp.path(), "new-art", true, &["rust"]);
create_legacy_article(tmp.path(), "old-art", true, &["go"]);
let articles = ContentLoader::load_all(tmp.path()).unwrap();
assert_eq!(articles.len(), 2);
let new = articles.iter().find(|a| a.sys_title == "new-art").unwrap();
assert!(new.content_html.contains("<h1>"));
let old = articles.iter().find(|a| a.sys_title == "old-art").unwrap();
assert!(old.content_html.contains("<h2>Intro</h2>"));
}
#[test]
fn skip_non_directory_entries() {
let tmp = tempfile::tempdir().unwrap();
create_new_article(tmp.path(), "valid-article", true, &[]);
fs::write(tmp.path().join("README.md"), "not an article").unwrap();
let articles = ContentLoader::load_all(tmp.path()).unwrap();
assert_eq!(articles.len(), 1);
}
#[test]
fn skip_directory_without_json() {
let tmp = tempfile::tempdir().unwrap();
create_new_article(tmp.path(), "valid-article", true, &[]);
fs::create_dir_all(tmp.path().join("images")).unwrap();
fs::write(tmp.path().join("images/photo.png"), b"fake image").unwrap();
let articles = ContentLoader::load_all(tmp.path()).unwrap();
assert_eq!(articles.len(), 1);
}
}

View file

@ -0,0 +1,7 @@
[package]
name = "wo-route"
version = "0.1.0"
edition = "2021"
[dependencies]
wo-http = { path = "../wo-http" }

View file

@ -0,0 +1,5 @@
mod pattern;
mod router;
pub use pattern::{Pattern, Segment};
pub use router::{RouteParams, Router};

View file

@ -0,0 +1,146 @@
/// A compiled URL pattern segment.
#[derive(Debug, Clone, PartialEq)]
pub enum Segment {
/// Exact literal match (e.g., "blog").
Literal(String),
/// Named parameter (e.g., ":sys_title").
Param(String),
/// Wildcard matching the rest of the path (e.g., "*path").
Wildcard(String),
}
/// A compiled URL pattern like "/blog/:sys_title" or "/static/*path".
#[derive(Debug, Clone)]
pub struct Pattern {
pub segments: Vec<Segment>,
}
impl Pattern {
/// Compile a pattern string into segments.
///
/// - `/blog/:sys_title` → `[Literal("blog"), Param("sys_title")]`
/// - `/static/*path` → `[Literal("static"), Wildcard("path")]`
/// - `/` → `[]`
pub fn compile(pattern: &str) -> Self {
let segments = pattern
.trim_start_matches('/')
.split('/')
.filter(|s| !s.is_empty())
.map(|s| {
if let Some(name) = s.strip_prefix(':') {
Segment::Param(name.to_string())
} else if let Some(name) = s.strip_prefix('*') {
Segment::Wildcard(name.to_string())
} else {
Segment::Literal(s.to_string())
}
})
.collect();
Self { segments }
}
/// Try to match a URL path against this pattern.
///
/// Returns `Some(params)` if the path matches, where params is a list
/// of `(name, value)` pairs for any `:param` or `*wildcard` segments.
pub fn matches(&self, path: &str) -> Option<Vec<(String, String)>> {
let path_segments: Vec<&str> = path
.trim_start_matches('/')
.split('/')
.filter(|s| !s.is_empty())
.collect();
let mut params = Vec::new();
let mut pi = 0; // path segment index
for seg in &self.segments {
match seg {
Segment::Literal(lit) => {
if pi >= path_segments.len() || path_segments[pi] != lit.as_str() {
return None;
}
pi += 1;
}
Segment::Param(name) => {
if pi >= path_segments.len() {
return None;
}
params.push((name.clone(), path_segments[pi].to_string()));
pi += 1;
}
Segment::Wildcard(name) => {
if pi >= path_segments.len() {
return None;
}
let rest = path_segments[pi..].join("/");
params.push((name.clone(), rest));
return Some(params);
}
}
}
// All pattern segments consumed; path must also be fully consumed.
if pi == path_segments.len() {
Some(params)
} else {
None
}
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn root_path() {
let p = Pattern::compile("/");
assert!(p.matches("/").is_some());
assert!(p.matches("/blog").is_none());
}
#[test]
fn literal_path() {
let p = Pattern::compile("/about");
assert!(p.matches("/about").is_some());
assert!(p.matches("/contact").is_none());
assert!(p.matches("/about/extra").is_none());
}
#[test]
fn param_extraction() {
let p = Pattern::compile("/blog/:sys_title");
let params = p.matches("/blog/linux-misc").unwrap();
assert_eq!(params, vec![("sys_title".into(), "linux-misc".into())]);
assert!(p.matches("/blog").is_none());
assert!(p.matches("/blog/linux-misc/extra").is_none());
}
#[test]
fn wildcard() {
let p = Pattern::compile("/static/*path");
let params = p.matches("/static/styles/main.css").unwrap();
assert_eq!(params, vec![("path".into(), "styles/main.css".into())]);
let params = p.matches("/static/logo.png").unwrap();
assert_eq!(params, vec![("path".into(), "logo.png".into())]);
assert!(p.matches("/static").is_none());
}
#[test]
fn multi_segment() {
let p = Pattern::compile("/tag/:tag");
let params = p.matches("/tag/rust").unwrap();
assert_eq!(params, vec![("tag".into(), "rust".into())]);
}
#[test]
fn no_match() {
let p = Pattern::compile("/blog/:sys_title");
assert!(p.matches("/about").is_none());
assert!(p.matches("/").is_none());
}
}

View file

@ -0,0 +1,127 @@
use std::collections::HashMap;
use wo_http::request::{Method, Request};
use crate::pattern::Pattern;
/// Extracted route parameters.
#[derive(Debug, Clone, Default)]
pub struct RouteParams {
params: HashMap<String, String>,
}
impl RouteParams {
pub fn get(&self, key: &str) -> Option<&str> {
self.params.get(key).map(|s| s.as_str())
}
pub fn from_pairs(pairs: Vec<(String, String)>) -> Self {
Self {
params: pairs.into_iter().collect(),
}
}
}
/// A route entry: method + pattern + handler name.
struct Route {
method: Method,
pattern: Pattern,
handler: String,
}
/// URL router that matches requests to named handlers.
pub struct Router {
routes: Vec<Route>,
}
impl Router {
pub fn new() -> Self {
Self { routes: Vec::new() }
}
/// Add a route. Handler is a string name that the caller maps to a function.
pub fn add(&mut self, method: Method, pattern: &str, handler: &str) {
self.routes.push(Route {
method,
pattern: Pattern::compile(pattern),
handler: handler.to_string(),
});
}
/// Match a request to a route. Returns the handler name and extracted params.
pub fn dispatch(&self, request: &Request) -> Option<(String, RouteParams)> {
for route in &self.routes {
if route.method != request.method {
continue;
}
if let Some(pairs) = route.pattern.matches(&request.path) {
return Some((
route.handler.clone(),
RouteParams::from_pairs(pairs),
));
}
}
None
}
}
#[cfg(test)]
mod tests {
use super::*;
fn make_request(method: Method, path: &str) -> Request {
Request {
method,
path: path.to_string(),
query: None,
headers: HashMap::new(),
}
}
#[test]
fn dispatch_routes() {
let mut router = Router::new();
router.add(Method::Get, "/", "home");
router.add(Method::Get, "/blog/:sys_title", "article");
router.add(Method::Get, "/about", "about");
router.add(Method::Get, "/tag/:tag", "tag_listing");
router.add(Method::Get, "/static/*path", "static_file");
// Home.
let (handler, _) = router.dispatch(&make_request(Method::Get, "/")).unwrap();
assert_eq!(handler, "home");
// Article.
let (handler, params) = router
.dispatch(&make_request(Method::Get, "/blog/linux-misc"))
.unwrap();
assert_eq!(handler, "article");
assert_eq!(params.get("sys_title"), Some("linux-misc"));
// About.
let (handler, _) = router.dispatch(&make_request(Method::Get, "/about")).unwrap();
assert_eq!(handler, "about");
// Tag.
let (handler, params) = router
.dispatch(&make_request(Method::Get, "/tag/rust"))
.unwrap();
assert_eq!(handler, "tag_listing");
assert_eq!(params.get("tag"), Some("rust"));
// Static.
let (handler, params) = router
.dispatch(&make_request(Method::Get, "/static/styles/main.css"))
.unwrap();
assert_eq!(handler, "static_file");
assert_eq!(params.get("path"), Some("styles/main.css"));
// No match.
assert!(router
.dispatch(&make_request(Method::Get, "/nonexistent"))
.is_none());
// Wrong method.
assert!(router.dispatch(&make_request(Method::Post, "/")).is_none());
}
}

28
crates/wo-rt/Cargo.toml Normal file
View file

@ -0,0 +1,28 @@
[package]
name = "wo-rt"
version = "0.1.0"
edition = "2021"
[dependencies]
wo-model = { path = "../wo-model" }
wo-store = { path = "../wo-store" }
wo-watch = { path = "../wo-watch" }
wo-event = { path = "../wo-event" }
wo-sub = { path = "../wo-sub" }
wo-http = { path = "../wo-http" }
wo-route = { path = "../wo-route" }
wo-htmlx = { path = "../wo-htmlx" }
wo-md = { path = "../wo-md" }
wo-serve = { path = "../wo-serve" }
libc = { workspace = true }
serde = { workspace = true }
serde_json = { workspace = true }
bincode = { workspace = true }
[[bin]]
name = "writeonce"
path = "src/main.rs"
[dev-dependencies]
tempfile = "3"
bincode = { workspace = true }

View file

@ -0,0 +1,31 @@
use std::path::PathBuf;
/// Runtime configuration.
#[derive(Debug, Clone)]
pub struct Config {
/// Path to the content directory (JSON + MD source files).
pub content_dir: PathBuf,
/// Path to the data directory (.seg + .idx derived files).
pub data_dir: PathBuf,
/// Path to the templates directory (.htmlx files).
pub templates_dir: PathBuf,
/// Path to the static assets directory (CSS, images).
pub static_dir: PathBuf,
/// HTTP bind address (e.g., "0.0.0.0:3000").
pub bind_addr: String,
/// Force a full rebuild on startup (ignore existing data/).
pub rebuild_on_start: bool,
}
impl Config {
pub fn new(content_dir: impl Into<PathBuf>, data_dir: impl Into<PathBuf>) -> Self {
Self {
content_dir: content_dir.into(),
data_dir: data_dir.into(),
templates_dir: PathBuf::from("templates"),
static_dir: PathBuf::from("static"),
bind_addr: "0.0.0.0:3000".into(),
rebuild_on_start: false,
}
}
}

View file

@ -0,0 +1,130 @@
use wo_htmlx::{self, Value};
use wo_http::response::Response;
use wo_route::RouteParams;
use wo_store::Store;
use wo_htmlx::TemplateRegistry;
/// Render the homepage: list of published articles.
pub fn handle_home(store: &Store, templates: &TemplateRegistry) -> Response {
let articles = match store.list_published(0, 20) {
Ok(a) => a,
Err(e) => return Response::internal_error(&e.to_string()),
};
let articles_val: Vec<Value> = articles
.iter()
.map(|a| {
let json = serde_json::to_value(a).unwrap_or_default();
Value::from_json(&json)
})
.collect();
let mut ctx = std::collections::BTreeMap::new();
ctx.insert("articles".into(), Value::List(articles_val));
ctx.insert("page_title".into(), Value::String("writeonce".into()));
let context = Value::Object(ctx);
render_page("home", &context, templates)
}
/// Render a single article page.
pub fn handle_article(params: &RouteParams, store: &Store, templates: &TemplateRegistry) -> Response {
let sys_title = match params.get("sys_title") {
Some(t) => t,
None => return Response::not_found(),
};
let article = match store.get_by_title(sys_title) {
Ok(Some(a)) => a,
Ok(None) => return Response::not_found(),
Err(e) => return Response::internal_error(&e.to_string()),
};
let json = serde_json::to_value(&article).unwrap_or_default();
let mut ctx = std::collections::BTreeMap::new();
ctx.insert("article".into(), Value::from_json(&json));
ctx.insert("page_title".into(), Value::String(article.title.clone()));
let context = Value::Object(ctx);
render_page("article", &context, templates)
}
/// Render a tag listing page.
pub fn handle_tag(params: &RouteParams, store: &Store, templates: &TemplateRegistry) -> Response {
let tag = match params.get("tag") {
Some(t) => t,
None => return Response::not_found(),
};
let articles = match store.list_by_tag(tag) {
Ok(a) => a,
Err(e) => return Response::internal_error(&e.to_string()),
};
let articles_val: Vec<Value> = articles
.iter()
.map(|a| {
let json = serde_json::to_value(a).unwrap_or_default();
Value::from_json(&json)
})
.collect();
let mut ctx = std::collections::BTreeMap::new();
ctx.insert("articles".into(), Value::List(articles_val));
ctx.insert("page_title".into(), Value::String(format!("tag: {}", tag)));
ctx.insert("tag".into(), Value::String(tag.to_string()));
let context = Value::Object(ctx);
render_page("home", &context, templates)
}
/// Render a static page (about, contact).
pub fn handle_static_page(name: &str, templates: &TemplateRegistry) -> Response {
let mut ctx = std::collections::BTreeMap::new();
ctx.insert("page_title".into(), Value::String(name.to_string()));
let context = Value::Object(ctx);
render_page(name, &context, templates)
}
/// Render a page template composed with layout, header, and footer.
fn render_page(template_name: &str, context: &Value, templates: &TemplateRegistry) -> Response {
let page_nodes = match templates.get(template_name) {
Some(nodes) => nodes,
None => return Response::not_found(),
};
let partials = templates.partials();
// Render the page content.
let page_html = wo_htmlx::render(page_nodes, context, partials);
// Compose with layout if it exists.
let html = if let Some(layout_nodes) = templates.get("layout") {
// Inject page content and header/footer.
let header_html = templates
.get("header")
.map(|n| wo_htmlx::render(n, context, partials))
.unwrap_or_default();
let footer_html = templates
.get("footer")
.map(|n| wo_htmlx::render(n, context, partials))
.unwrap_or_default();
let mut layout_ctx = std::collections::BTreeMap::new();
layout_ctx.insert("content".into(), Value::String(page_html));
layout_ctx.insert("header".into(), Value::String(header_html));
layout_ctx.insert("footer".into(), Value::String(footer_html));
if let Value::Object(map) = context {
for (k, v) in map {
layout_ctx.insert(k.clone(), v.clone());
}
}
let layout_context = Value::Object(layout_ctx);
wo_htmlx::render(layout_nodes, &layout_context, partials)
} else {
page_html
};
Response::html(html)
}

6
crates/wo-rt/src/lib.rs Normal file
View file

@ -0,0 +1,6 @@
mod config;
mod handlers;
mod runtime;
pub use config::Config;
pub use runtime::{Runtime, RuntimeHandle};

34
crates/wo-rt/src/main.rs Normal file
View file

@ -0,0 +1,34 @@
use std::path::PathBuf;
use wo_rt::{Config, Runtime};
fn main() {
let content_dir = PathBuf::from("content");
let data_dir = PathBuf::from("data");
let config = Config {
content_dir,
data_dir,
templates_dir: PathBuf::from("templates"),
static_dir: PathBuf::from("static"),
bind_addr: "0.0.0.0:3000".into(),
rebuild_on_start: true,
};
let mut rt = match Runtime::new(&config) {
Ok(rt) => rt,
Err(e) => {
eprintln!("failed to start: {}", e);
std::process::exit(1);
}
};
eprintln!("writeonce started — content={} data={}",
config.content_dir.display(), config.data_dir.display());
if let Err(e) = rt.run() {
eprintln!("runtime error: {}", e);
std::process::exit(1);
}
eprintln!("writeonce shut down");
}

481
crates/wo-rt/src/runtime.rs Normal file
View file

@ -0,0 +1,481 @@
use std::collections::HashMap;
use std::io;
use std::os::unix::io::RawFd;
use std::sync::atomic::{AtomicBool, Ordering};
use std::sync::Arc;
use std::time::Duration;
use wo_event::{EventLoop, Interest, SignalFd, TimerFd, Token};
use wo_htmlx::TemplateRegistry;
use wo_http::connection::{Connection, State};
use wo_http::listener::TcpListener;
use wo_http::request::Method;
use wo_http::response::Response;
use wo_route::{RouteParams, Router};
use wo_store::Store;
use wo_sub::{ChangeType, Subscription, SubscriptionManager};
use wo_watch::{ContentChange, ContentWatcher};
use crate::handlers;
use crate::Config;
// Well-known tokens.
const TOKEN_WATCHER: Token = 1;
const TOKEN_SIGNAL: Token = 2;
const TOKEN_TIMER: Token = 3;
const TOKEN_NOTIFY: Token = 4;
const TOKEN_HTTP_LISTENER: Token = 5;
// Token ranges.
const TOKEN_HTTP_BASE: Token = 10000;
const TOKEN_HTTP_MAX: Token = 19999;
const TOKEN_SUB_BASE: Token = 20000;
/// The writeonce runtime: single process, single event loop, all fds on one epoll.
pub struct Runtime {
event_loop: EventLoop,
store: Store,
watcher: ContentWatcher,
subscriptions: SubscriptionManager,
signal_fd: SignalFd,
timer_fd: TimerFd,
running: Arc<AtomicBool>,
// HTTP
listener: Option<TcpListener>,
connections: HashMap<Token, Connection>,
next_http_token: Token,
router: Router,
templates: TemplateRegistry,
static_dir: std::path::PathBuf,
// Subscribed browser connections.
next_sub_token: Token,
sub_token_to_fd: HashMap<Token, RawFd>,
}
/// Handle returned by `Runtime::start()` for external interaction.
pub struct RuntimeHandle {
running: Arc<AtomicBool>,
}
impl RuntimeHandle {
pub fn shutdown(&self) {
self.running.store(false, Ordering::SeqCst);
}
}
impl Runtime {
pub fn new(config: &Config) -> io::Result<Self> {
let event_loop = EventLoop::new()?;
let store = if config.rebuild_on_start {
let mut s = Store::open(&config.content_dir, &config.data_dir)?;
s.rebuild()?;
s
} else {
Store::open(&config.content_dir, &config.data_dir)?
};
let watcher = ContentWatcher::new(&config.content_dir)?;
let subscriptions = SubscriptionManager::new()?;
let signal_fd = SignalFd::new()?;
let timer_fd = TimerFd::new()?;
timer_fd.set(Duration::from_secs(60), Duration::from_secs(60))?;
// Register system fds.
event_loop.register(watcher.fd(), Interest::Readable, TOKEN_WATCHER)?;
event_loop.register(signal_fd.fd(), Interest::Readable, TOKEN_SIGNAL)?;
event_loop.register(timer_fd.fd(), Interest::Readable, TOKEN_TIMER)?;
event_loop.register(subscriptions.notify_fd(), Interest::Readable, TOKEN_NOTIFY)?;
// HTTP listener (optional — may not bind in test mode).
let listener = if !config.bind_addr.is_empty() {
match TcpListener::bind(&config.bind_addr) {
Ok(l) => {
event_loop.register(l.fd(), Interest::Readable, TOKEN_HTTP_LISTENER)?;
Some(l)
}
Err(e) => {
eprintln!("wo-rt: failed to bind {}: {}", config.bind_addr, e);
None
}
}
} else {
None
};
// Load templates.
let templates = TemplateRegistry::load(&config.templates_dir)?;
// Build router.
let mut router = Router::new();
router.add(Method::Get, "/", "home");
router.add(Method::Get, "/blog/:sys_title", "article");
router.add(Method::Get, "/about", "about");
router.add(Method::Get, "/contact", "contact");
router.add(Method::Get, "/tag/:tag", "tag_listing");
router.add(Method::Get, "/static/*path", "static_file");
Ok(Self {
event_loop,
store,
watcher,
subscriptions,
signal_fd,
timer_fd,
running: Arc::new(AtomicBool::new(false)),
listener,
connections: HashMap::new(),
next_http_token: TOKEN_HTTP_BASE,
router,
templates,
static_dir: config.static_dir.clone(),
next_sub_token: TOKEN_SUB_BASE,
sub_token_to_fd: HashMap::new(),
})
}
pub fn run(&mut self) -> io::Result<()> {
self.running.store(true, Ordering::SeqCst);
if let Some(ref listener) = self.listener {
let (_, port) = listener.local_addr()?;
eprintln!("writeonce listening on port {}", port);
}
while self.running.load(Ordering::SeqCst) {
let events = self.event_loop.poll(Some(Duration::from_millis(500)))?;
for event in events {
match event.token {
TOKEN_WATCHER if event.readable => {
self.handle_watcher_event()?;
}
TOKEN_SIGNAL if event.readable => {
let _ = self.signal_fd.read();
self.running.store(false, Ordering::SeqCst);
}
TOKEN_TIMER if event.readable => {
let _ = self.timer_fd.read();
}
TOKEN_NOTIFY if event.readable => {
// Notifications already written by SubscriptionManager.
}
TOKEN_HTTP_LISTENER if event.readable => {
self.handle_accept()?;
}
token if token >= TOKEN_SUB_BASE => {
// Subscribed connection.
if event.hangup || event.error {
if let Some(&fd) = self.sub_token_to_fd.get(&token) {
self.subscriptions.unsubscribe(fd);
let _ = self.event_loop.deregister(fd);
self.sub_token_to_fd.remove(&token);
// Don't close fd here — Connection owns it.
}
}
}
token if token >= TOKEN_HTTP_BASE => {
self.handle_http_event(token, &event)?;
}
_ => {}
}
}
// Clean up completed connections.
let done_tokens: Vec<Token> = self
.connections
.iter()
.filter(|(_, c)| c.state == State::Done)
.map(|(&t, _)| t)
.collect();
for token in done_tokens {
if let Some(conn) = self.connections.remove(&token) {
let _ = self.event_loop.deregister(conn.fd);
}
}
}
Ok(())
}
pub fn handle(&self) -> RuntimeHandle {
RuntimeHandle {
running: self.running.clone(),
}
}
pub fn store(&self) -> &Store {
&self.store
}
pub fn store_mut(&mut self) -> &mut Store {
&mut self.store
}
/// Subscribe a fd (binary format, for internal use).
pub fn subscribe(&mut self, fd: RawFd, query: Subscription) -> io::Result<Token> {
let token = self.next_sub_token;
self.next_sub_token += 1;
self.event_loop.register(fd, Interest::Readable, token)?;
self.subscriptions.subscribe(fd, query);
self.sub_token_to_fd.insert(token, fd);
Ok(token)
}
fn handle_accept(&mut self) -> io::Result<()> {
let listener = match &self.listener {
Some(l) => l,
None => return Ok(()),
};
// Accept all pending connections.
while let Some(client_fd) = listener.accept()? {
if self.next_http_token > TOKEN_HTTP_MAX {
self.next_http_token = TOKEN_HTTP_BASE; // recycle tokens
}
let token = self.next_http_token;
self.next_http_token += 1;
self.event_loop
.register(client_fd, Interest::Readable, token)?;
self.connections.insert(token, Connection::new(client_fd));
}
Ok(())
}
fn handle_http_event(&mut self, token: Token, event: &wo_event::Event) -> io::Result<()> {
if event.hangup || event.error {
if let Some(conn) = self.connections.remove(&token) {
let _ = self.event_loop.deregister(conn.fd);
}
return Ok(());
}
// Read phase: parse the request without holding mutable borrow during dispatch.
let conn = match self.connections.get_mut(&token) {
Some(c) => c,
None => return Ok(()),
};
match conn.state {
State::ReadingRequest if event.readable => {
let open = conn.read()?;
if !open {
conn.set_done();
return Ok(());
}
if let Some(request) = conn.try_parse() {
// Dispatch the handler (borrows self immutably via store/templates).
let dispatch_result = self.router.dispatch(&request);
let response = match dispatch_result {
Some((ref handler, ref params)) => {
dispatch_handler(handler, params, &self.store, &self.templates, &self.static_dir)
}
None => Response::not_found(),
};
// Now get the connection again to set the response.
let conn = self.connections.get_mut(&token).unwrap();
conn.set_response(&response);
if conn.write()? {
conn.set_done();
} else {
let fd = conn.fd;
self.event_loop.modify(fd, Interest::Writable, token)?;
}
}
}
State::WritingResponse if event.writable => {
if conn.write()? {
conn.set_done();
}
}
_ => {}
}
Ok(())
}
fn handle_watcher_event(&mut self) -> io::Result<()> {
let changes = self.watcher.process_events()?;
for change in &changes {
match change {
ContentChange::Created(sys_title) | ContentChange::Modified(sys_title) => {
self.store.rebuild()?;
let version = self.store.article_version(sys_title).unwrap_or(0);
let tags = self
.store
.get_by_title(sys_title)?
.map(|a| a.tags.clone())
.unwrap_or_default();
let change_type = match change {
ContentChange::Created(_) => ChangeType::Created,
_ => ChangeType::Updated,
};
self.subscriptions.notify(sys_title, &tags, change_type, version)?;
}
ContentChange::Deleted(sys_title) => {
let version = self.store.article_version(sys_title).unwrap_or(0);
self.subscriptions.notify(sys_title, &[], ChangeType::Deleted, version)?;
self.store.rebuild()?;
}
}
}
Ok(())
}
}
fn dispatch_handler(
handler: &str,
params: &RouteParams,
store: &Store,
templates: &TemplateRegistry,
static_dir: &std::path::Path,
) -> Response {
match handler {
"home" => handlers::handle_home(store, templates),
"article" => handlers::handle_article(params, store, templates),
"tag_listing" => handlers::handle_tag(params, store, templates),
"about" => handlers::handle_static_page("about", templates),
"contact" => handlers::handle_static_page("contact", templates),
"static_file" => {
let path = params.get("path").unwrap_or("");
wo_serve::send_file(static_dir, path)
}
_ => Response::not_found(),
}
}
#[cfg(test)]
mod tests {
use super::*;
use std::fs;
fn create_test_article(base: &std::path::Path, sys_title: &str, published: bool) {
let dir = base.join(sys_title);
fs::create_dir_all(&dir).unwrap();
let json = format!(
r#"{{
"title": "{}",
"sys_title": "{}",
"published": {},
"content": {{
"author": "Author",
"content": {{
"sections": [{{ "heading": "Intro", "paragraphs": ["Hello."] }}],
"codes": [],
"images": []
}},
"tags": ["test"],
"publishedOn": 1000
}},
"published_on": 1000
}}"#,
sys_title, sys_title, published
);
fs::write(dir.join(format!("{}.json", sys_title)), json).unwrap();
}
fn test_config(content: &std::path::Path, data: &std::path::Path) -> Config {
Config {
bind_addr: String::new(), // no HTTP in tests
..Config::new(content, data)
}
}
#[test]
fn runtime_creates_and_queries() {
let tmp = tempfile::tempdir().unwrap();
let content = tmp.path().join("content");
let data = tmp.path().join("data");
create_test_article(&content, "art-one", true);
create_test_article(&content, "art-two", true);
let config = test_config(&content, &data);
let rt = Runtime::new(&config).unwrap();
let art = rt.store().get_by_title("art-one").unwrap().unwrap();
assert_eq!(art.sys_title, "art-one");
assert_eq!(rt.store().count_published().unwrap(), 2);
}
#[test]
fn runtime_subscribe_and_notify() {
let tmp = tempfile::tempdir().unwrap();
let content = tmp.path().join("content");
let data = tmp.path().join("data");
create_test_article(&content, "art-one", true);
let config = test_config(&content, &data);
let mut rt = Runtime::new(&config).unwrap();
let mut fds = [0i32; 2];
assert_eq!(unsafe { libc::pipe2(fds.as_mut_ptr(), libc::O_NONBLOCK) }, 0);
let (read_fd, write_fd) = (fds[0], fds[1]);
rt.subscribe(write_fd, Subscription::ByTitle("art-one".into())).unwrap();
rt.subscriptions
.notify("art-one", &["test".into()], ChangeType::Updated, 1)
.unwrap();
let mut len_buf = [0u8; 4];
let ret = unsafe { libc::read(read_fd, len_buf.as_mut_ptr() as *mut libc::c_void, 4) };
assert_eq!(ret, 4);
let len = u32::from_le_bytes(len_buf) as usize;
let mut payload = vec![0u8; len];
let ret = unsafe { libc::read(read_fd, payload.as_mut_ptr() as *mut libc::c_void, len) };
assert_eq!(ret as usize, len);
let notification: wo_sub::Notification = bincode::deserialize(&payload).unwrap();
assert_eq!(notification.sys_title, "art-one");
unsafe { libc::close(read_fd); libc::close(write_fd); }
}
#[test]
fn runtime_shutdown_via_handle() {
let tmp = tempfile::tempdir().unwrap();
let content = tmp.path().join("content");
let data = tmp.path().join("data");
fs::create_dir_all(&content).unwrap();
let config = test_config(&content, &data);
let mut rt = Runtime::new(&config).unwrap();
let handle = rt.handle();
std::thread::spawn(move || {
std::thread::sleep(Duration::from_millis(100));
handle.shutdown();
});
rt.run().unwrap();
}
#[test]
fn runtime_rebuild_on_start() {
let tmp = tempfile::tempdir().unwrap();
let content = tmp.path().join("content");
let data = tmp.path().join("data");
create_test_article(&content, "art-one", true);
let config = test_config(&content, &data);
let rt = Runtime::new(&config).unwrap();
assert_eq!(rt.store().count_published().unwrap(), 1);
drop(rt);
let config = Config {
rebuild_on_start: true,
..test_config(&content, &data)
};
let rt = Runtime::new(&config).unwrap();
assert_eq!(rt.store().count_published().unwrap(), 1);
}
}

13
crates/wo-seg/Cargo.toml Normal file
View file

@ -0,0 +1,13 @@
[package]
name = "wo-seg"
version = "0.1.0"
edition = "2021"
[dependencies]
wo-model = { path = "../wo-model" }
serde = { workspace = true }
bincode = { workspace = true }
libc = { workspace = true }
[dev-dependencies]
tempfile = "3"

111
crates/wo-seg/src/header.rs Normal file
View file

@ -0,0 +1,111 @@
use std::io;
/// Magic bytes: "WOSF" (WriteOnce Segment File).
pub const MAGIC: [u8; 4] = *b"WOSF";
/// Segment file version.
pub const VERSION: u16 = 1;
/// Fixed header size in bytes.
pub const HEADER_SIZE: u64 = 32;
/// The segment file header, stored at byte 0 of every .seg file.
///
/// Layout (32 bytes):
/// ```text
/// [0..4] magic: b"WOSF"
/// [4..6] version: u16 LE
/// [6..8] flags: u16 LE (reserved)
/// [8..16] record_count: u64 LE
/// [16..24] data_start: u64 LE
/// [24..32] reserved: 8 bytes
/// ```
#[derive(Debug, Clone, Copy, PartialEq)]
pub struct SegHeader {
pub version: u16,
pub flags: u16,
pub record_count: u64,
pub data_start: u64,
}
impl SegHeader {
/// Create a new header with default values.
pub fn new() -> Self {
Self {
version: VERSION,
flags: 0,
record_count: 0,
data_start: HEADER_SIZE,
}
}
/// Serialize the header into a 32-byte buffer.
pub fn to_bytes(&self) -> [u8; 32] {
let mut buf = [0u8; 32];
buf[0..4].copy_from_slice(&MAGIC);
buf[4..6].copy_from_slice(&self.version.to_le_bytes());
buf[6..8].copy_from_slice(&self.flags.to_le_bytes());
buf[8..16].copy_from_slice(&self.record_count.to_le_bytes());
buf[16..24].copy_from_slice(&self.data_start.to_le_bytes());
// [24..32] reserved, stays zero
buf
}
/// Parse a header from a 32-byte buffer.
pub fn from_bytes(buf: &[u8; 32]) -> io::Result<Self> {
if &buf[0..4] != &MAGIC {
return Err(io::Error::new(
io::ErrorKind::InvalidData,
format!(
"invalid magic bytes: expected {:?}, got {:?}",
MAGIC,
&buf[0..4]
),
));
}
let version = u16::from_le_bytes([buf[4], buf[5]]);
if version != VERSION {
return Err(io::Error::new(
io::ErrorKind::InvalidData,
format!("unsupported version: {}", version),
));
}
let flags = u16::from_le_bytes([buf[6], buf[7]]);
let record_count = u64::from_le_bytes(buf[8..16].try_into().unwrap());
let data_start = u64::from_le_bytes(buf[16..24].try_into().unwrap());
Ok(Self {
version,
flags,
record_count,
data_start,
})
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn round_trip() {
let header = SegHeader {
version: VERSION,
flags: 0,
record_count: 42,
data_start: HEADER_SIZE,
};
let bytes = header.to_bytes();
let parsed = SegHeader::from_bytes(&bytes).unwrap();
assert_eq!(header, parsed);
}
#[test]
fn bad_magic() {
let mut bytes = SegHeader::new().to_bytes();
bytes[0] = b'X';
assert!(SegHeader::from_bytes(&bytes).is_err());
}
}

14
crates/wo-seg/src/lib.rs Normal file
View file

@ -0,0 +1,14 @@
mod header;
mod reader;
mod writer;
pub use header::{SegHeader, HEADER_SIZE, MAGIC};
pub use reader::SegReader;
pub use writer::SegWriter;
/// Record flags.
pub const FLAG_ACTIVE: u8 = 0x00;
pub const FLAG_TOMBSTONED: u8 = 0x01;
/// Size of the per-record header: 4 bytes length + 1 byte flags.
pub const RECORD_HEADER_SIZE: usize = 5;

221
crates/wo-seg/src/reader.rs Normal file
View file

@ -0,0 +1,221 @@
use std::fs::File;
use std::io;
use std::os::unix::io::AsRawFd;
use std::path::Path;
use wo_model::Article;
use crate::header::SegHeader;
use crate::{FLAG_TOMBSTONED, RECORD_HEADER_SIZE};
/// Reads articles from a .seg file.
pub struct SegReader {
file: File,
header: SegHeader,
}
impl SegReader {
/// Open an existing .seg file and validate its header.
pub fn open(path: &Path) -> io::Result<Self> {
let file = File::open(path)?;
let mut header_buf = [0u8; 32];
pread_exact(&file, &mut header_buf, 0)?;
let header = SegHeader::from_bytes(&header_buf)?;
Ok(Self { file, header })
}
/// Read a single article record at the given byte offset.
///
/// Returns `None` if the record is tombstoned.
pub fn read_at(&self, offset: u64) -> io::Result<Option<Article>> {
// Read the record header: [u32 length][u8 flags]
let mut rec_header = [0u8; RECORD_HEADER_SIZE];
pread_exact(&self.file, &mut rec_header, offset)?;
let payload_len = u32::from_le_bytes(rec_header[0..4].try_into().unwrap()) as usize;
let flags = rec_header[4];
if flags == FLAG_TOMBSTONED {
return Ok(None);
}
// Read the payload.
let mut payload = vec![0u8; payload_len];
pread_exact(&self.file, &mut payload, offset + RECORD_HEADER_SIZE as u64)?;
let article: Article = bincode::deserialize(&payload).map_err(|e| {
io::Error::new(io::ErrorKind::InvalidData, format!("bincode: {}", e))
})?;
Ok(Some(article))
}
/// Iterate all active (non-tombstoned) records.
///
/// Returns `(byte_offset, Article)` pairs.
pub fn iter(&self) -> io::Result<Vec<(u64, Article)>> {
let mut results = Vec::new();
let mut pos = self.header.data_start;
for _ in 0..self.header.record_count {
// Read record header.
let mut rec_header = [0u8; RECORD_HEADER_SIZE];
pread_exact(&self.file, &mut rec_header, pos)?;
let payload_len = u32::from_le_bytes(rec_header[0..4].try_into().unwrap()) as usize;
let flags = rec_header[4];
let record_offset = pos;
pos += RECORD_HEADER_SIZE as u64 + payload_len as u64;
if flags == FLAG_TOMBSTONED {
continue;
}
let mut payload = vec![0u8; payload_len];
pread_exact(
&self.file,
&mut payload,
record_offset + RECORD_HEADER_SIZE as u64,
)?;
let article: Article = bincode::deserialize(&payload).map_err(|e| {
io::Error::new(io::ErrorKind::InvalidData, format!("bincode: {}", e))
})?;
results.push((record_offset, article));
}
Ok(results)
}
/// Return the parsed file header.
pub fn header(&self) -> &SegHeader {
&self.header
}
}
/// Positional read using pread(2). Does not modify the file offset.
fn pread_exact(file: &File, buf: &mut [u8], offset: u64) -> io::Result<()> {
let fd = file.as_raw_fd();
let mut read = 0usize;
while read < buf.len() {
let ret = unsafe {
libc::pread(
fd,
buf[read..].as_mut_ptr() as *mut libc::c_void,
buf.len() - read,
(offset + read as u64) as libc::off_t,
)
};
if ret < 0 {
return Err(io::Error::last_os_error());
}
if ret == 0 {
return Err(io::Error::new(
io::ErrorKind::UnexpectedEof,
"pread returned 0",
));
}
read += ret as usize;
}
Ok(())
}
#[cfg(test)]
mod tests {
use super::*;
use crate::SegWriter;
use wo_model::*;
fn sample_article(sys_title: &str) -> Article {
Article {
sys_title: sys_title.to_string(),
title: format!("Title: {}", sys_title),
published: true,
author: "Author".into(),
tags: vec!["test".into()],
published_on: Some(2000),
content_html: "<h1>Hello world.</h1>".into(),
}
}
#[test]
fn write_and_read_single() {
let tmp = tempfile::tempdir().unwrap();
let path = tmp.path().join("test.seg");
let article = sample_article("test-one");
let offset = {
let mut writer = SegWriter::create(&path).unwrap();
writer.append(&article).unwrap()
};
let reader = SegReader::open(&path).unwrap();
assert_eq!(reader.header().record_count, 1);
let loaded = reader.read_at(offset).unwrap().unwrap();
assert_eq!(loaded.sys_title, "test-one");
assert_eq!(loaded, article);
}
#[test]
fn write_and_read_multiple() {
let tmp = tempfile::tempdir().unwrap();
let path = tmp.path().join("multi.seg");
let articles: Vec<Article> = (0..5).map(|i| sample_article(&format!("art-{}", i))).collect();
let offsets: Vec<u64> = {
let mut writer = SegWriter::create(&path).unwrap();
articles.iter().map(|a| writer.append(a).unwrap()).collect()
};
let reader = SegReader::open(&path).unwrap();
assert_eq!(reader.header().record_count, 5);
// Read each by offset.
for (i, offset) in offsets.iter().enumerate() {
let loaded = reader.read_at(*offset).unwrap().unwrap();
assert_eq!(loaded.sys_title, format!("art-{}", i));
}
// Iterate all.
let all = reader.iter().unwrap();
assert_eq!(all.len(), 5);
}
#[test]
fn tombstone_record() {
let tmp = tempfile::tempdir().unwrap();
let path = tmp.path().join("tomb.seg");
let offset = {
let mut writer = SegWriter::create(&path).unwrap();
let o = writer.append(&sample_article("to-delete")).unwrap();
writer.append(&sample_article("to-keep")).unwrap();
writer.tombstone(o).unwrap();
o
};
let reader = SegReader::open(&path).unwrap();
// Direct read returns None for tombstoned.
assert!(reader.read_at(offset).unwrap().is_none());
// Iter skips tombstoned.
let all = reader.iter().unwrap();
assert_eq!(all.len(), 1);
assert_eq!(all[0].1.sys_title, "to-keep");
}
#[test]
fn corrupted_header() {
let tmp = tempfile::tempdir().unwrap();
let path = tmp.path().join("bad.seg");
std::fs::write(&path, b"not a segment file at all!!!!!xx").unwrap();
assert!(SegReader::open(&path).is_err());
}
}

129
crates/wo-seg/src/writer.rs Normal file
View file

@ -0,0 +1,129 @@
use std::fs::{File, OpenOptions};
use std::io;
use std::os::unix::io::AsRawFd;
use std::path::Path;
use wo_model::Article;
use crate::header::{SegHeader, HEADER_SIZE};
use crate::{FLAG_ACTIVE, RECORD_HEADER_SIZE};
/// Writes articles to a .seg file.
///
/// Records are appended sequentially. Each append returns the byte offset
/// of the record, which can be stored in an index for direct access.
pub struct SegWriter {
file: File,
header: SegHeader,
/// Current write position (end of file).
pos: u64,
}
impl SegWriter {
/// Create a new .seg file at the given path.
///
/// Writes the initial header and optionally pre-allocates disk space
/// using `fallocate` to reduce fragmentation.
pub fn create(path: &Path) -> io::Result<Self> {
let file = OpenOptions::new()
.read(true)
.write(true)
.create(true)
.truncate(true)
.open(path)?;
let header = SegHeader::new();
// Write the header at position 0.
pwrite_all(&file, &header.to_bytes(), 0)?;
// Pre-allocate 1 MB to reduce fragmentation.
let _ = fallocate_safe(&file, HEADER_SIZE as i64, 1024 * 1024);
Ok(Self {
file,
header,
pos: HEADER_SIZE,
})
}
/// Append an article as an active record.
///
/// Returns the byte offset of the record start (the position of the
/// length prefix), which can be used for direct reads via `SegReader::read_at`.
pub fn append(&mut self, article: &Article) -> io::Result<u64> {
let payload = bincode::serialize(article).map_err(|e| {
io::Error::new(io::ErrorKind::InvalidData, format!("bincode: {}", e))
})?;
let record_offset = self.pos;
let payload_len = payload.len() as u32;
// Build record: [u32 length][u8 flags][payload]
let mut record = Vec::with_capacity(RECORD_HEADER_SIZE + payload.len());
record.extend_from_slice(&payload_len.to_le_bytes());
record.push(FLAG_ACTIVE);
record.extend_from_slice(&payload);
pwrite_all(&self.file, &record, record_offset)?;
self.pos += record.len() as u64;
self.header.record_count += 1;
// Update the header with the new record count.
pwrite_all(&self.file, &self.header.to_bytes(), 0)?;
Ok(record_offset)
}
/// Tombstone a record at the given offset.
///
/// Sets the flags byte to `FLAG_TOMBSTONED` without modifying the payload.
pub fn tombstone(&self, record_offset: u64) -> io::Result<()> {
let flags_offset = record_offset + 4; // skip the u32 length
pwrite_all(&self.file, &[crate::FLAG_TOMBSTONED], flags_offset)
}
/// Return the current header (record count, etc.).
pub fn header(&self) -> &SegHeader {
&self.header
}
/// Sync file data and metadata to disk.
pub fn sync(&self) -> io::Result<()> {
self.file.sync_all()
}
}
/// Positional write using pwrite(2). Does not modify the file offset.
fn pwrite_all(file: &File, buf: &[u8], offset: u64) -> io::Result<()> {
let fd = file.as_raw_fd();
let mut written = 0usize;
while written < buf.len() {
let ret = unsafe {
libc::pwrite(
fd,
buf[written..].as_ptr() as *const libc::c_void,
buf.len() - written,
(offset + written as u64) as libc::off_t,
)
};
if ret < 0 {
return Err(io::Error::last_os_error());
}
if ret == 0 {
return Err(io::Error::new(io::ErrorKind::WriteZero, "pwrite returned 0"));
}
written += ret as usize;
}
Ok(())
}
/// Best-effort fallocate. Non-fatal if unsupported.
fn fallocate_safe(file: &File, offset: i64, len: i64) -> io::Result<()> {
let ret = unsafe { libc::fallocate(file.as_raw_fd(), 0, offset, len) };
if ret < 0 {
Err(io::Error::last_os_error())
} else {
Ok(())
}
}

View file

@ -0,0 +1,11 @@
[package]
name = "wo-serve"
version = "0.1.0"
edition = "2021"
[dependencies]
wo-http = { path = "../wo-http" }
libc = { workspace = true }
[dev-dependencies]
tempfile = "3"

View file

@ -0,0 +1,7 @@
mod mime;
mod resolve;
mod sendfile;
pub use mime::content_type_for;
pub use resolve::resolve_path;
pub use sendfile::send_file;

View file

@ -0,0 +1,44 @@
/// Map a file extension to its HTTP Content-Type.
pub fn content_type_for(path: &str) -> &'static str {
let ext = path.rsplit('.').next().unwrap_or("");
match ext {
"html" | "htmlx" => "text/html; charset=utf-8",
"css" => "text/css; charset=utf-8",
"js" => "application/javascript; charset=utf-8",
"json" => "application/json; charset=utf-8",
"png" => "image/png",
"jpg" | "jpeg" => "image/jpeg",
"gif" => "image/gif",
"svg" => "image/svg+xml",
"ico" => "image/x-icon",
"woff" => "font/woff",
"woff2" => "font/woff2",
"ttf" => "font/ttf",
"txt" => "text/plain; charset=utf-8",
"xml" => "application/xml; charset=utf-8",
_ => "application/octet-stream",
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn known_types() {
assert_eq!(content_type_for("main.css"), "text/css; charset=utf-8");
assert_eq!(content_type_for("logo.png"), "image/png");
assert_eq!(content_type_for("app.js"), "application/javascript; charset=utf-8");
assert_eq!(content_type_for("favicon.ico"), "image/x-icon");
}
#[test]
fn unknown_type() {
assert_eq!(content_type_for("file.xyz"), "application/octet-stream");
}
#[test]
fn nested_path() {
assert_eq!(content_type_for("styles/code-theme.css"), "text/css; charset=utf-8");
}
}

View file

@ -0,0 +1,80 @@
use std::io;
use std::path::{Path, PathBuf};
/// Resolve a request path relative to a static directory.
///
/// Returns the absolute path to the file, or an error if:
/// - The path contains `..` (directory traversal)
/// - The resolved path escapes the base directory
/// - The file doesn't exist
pub fn resolve_path(base: &Path, request_path: &str) -> io::Result<PathBuf> {
// Reject any path with `..`
if request_path.contains("..") {
return Err(io::Error::new(
io::ErrorKind::PermissionDenied,
"path traversal rejected",
));
}
let clean = request_path.trim_start_matches('/');
let resolved = base.join(clean);
// Canonicalize and verify it's still under base.
let canonical = resolved.canonicalize().map_err(|_| {
io::Error::new(io::ErrorKind::NotFound, "file not found")
})?;
let canonical_base = base.canonicalize().map_err(|_| {
io::Error::new(io::ErrorKind::NotFound, "base directory not found")
})?;
if !canonical.starts_with(&canonical_base) {
return Err(io::Error::new(
io::ErrorKind::PermissionDenied,
"path traversal rejected",
));
}
if !canonical.is_file() {
return Err(io::Error::new(io::ErrorKind::NotFound, "not a file"));
}
Ok(canonical)
}
#[cfg(test)]
mod tests {
use super::*;
use std::fs;
#[test]
fn valid_path() {
let tmp = tempfile::tempdir().unwrap();
fs::write(tmp.path().join("test.css"), "body {}").unwrap();
let path = resolve_path(tmp.path(), "test.css").unwrap();
assert!(path.ends_with("test.css"));
}
#[test]
fn nested_path() {
let tmp = tempfile::tempdir().unwrap();
fs::create_dir_all(tmp.path().join("styles")).unwrap();
fs::write(tmp.path().join("styles/main.css"), "body {}").unwrap();
let path = resolve_path(tmp.path(), "styles/main.css").unwrap();
assert!(path.ends_with("styles/main.css"));
}
#[test]
fn reject_traversal() {
let tmp = tempfile::tempdir().unwrap();
assert!(resolve_path(tmp.path(), "../etc/passwd").is_err());
}
#[test]
fn reject_nonexistent() {
let tmp = tempfile::tempdir().unwrap();
assert!(resolve_path(tmp.path(), "nonexistent.css").is_err());
}
}

View file

@ -0,0 +1,109 @@
use std::fs;
use std::io;
use std::os::unix::io::RawFd;
use std::path::Path;
use wo_http::response::Response;
use crate::mime::content_type_for;
use crate::resolve::resolve_path;
/// Serve a static file. Returns a Response with the file contents.
///
/// For small files, reads into memory and returns a normal Response.
/// For zero-copy serving via sendfile, use `send_file_zero_copy` instead.
pub fn send_file(static_dir: &Path, request_path: &str) -> Response {
let file_path = match resolve_path(static_dir, request_path) {
Ok(p) => p,
Err(e) => {
return match e.kind() {
io::ErrorKind::NotFound => Response::not_found(),
io::ErrorKind::PermissionDenied => {
Response::new(403, "Forbidden")
}
_ => Response::internal_error(&e.to_string()),
};
}
};
let body = match fs::read(&file_path) {
Ok(b) => b,
Err(_) => return Response::internal_error("failed to read file"),
};
let ct = content_type_for(file_path.to_str().unwrap_or(""));
let mut response = Response::ok(body, ct);
response.header("Cache-Control", "public, max-age=3600");
response
}
/// Zero-copy file serving using sendfile(2).
///
/// Sends the file directly from the page cache to the socket fd.
/// Returns the number of bytes sent, or an error.
pub fn sendfile_to_fd(socket_fd: RawFd, file_path: &Path) -> io::Result<usize> {
let file = fs::File::open(file_path)?;
let file_fd = {
use std::os::unix::io::AsRawFd;
file.as_raw_fd()
};
let metadata = file.metadata()?;
let file_size = metadata.len() as usize;
let mut offset: libc::off_t = 0;
let mut sent = 0usize;
while sent < file_size {
let n = unsafe {
libc::sendfile(
socket_fd,
file_fd,
&mut offset,
file_size - sent,
)
};
if n < 0 {
let err = io::Error::last_os_error();
if err.raw_os_error() == Some(libc::EAGAIN) {
break; // partial send, caller should retry on EPOLLOUT
}
return Err(err);
}
if n == 0 {
break;
}
sent += n as usize;
}
Ok(sent)
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn serve_existing_file() {
let tmp = tempfile::tempdir().unwrap();
fs::write(tmp.path().join("test.css"), "body { color: red; }").unwrap();
let response = send_file(tmp.path(), "test.css");
assert_eq!(response.status, 200);
assert!(String::from_utf8_lossy(&response.body).contains("color: red"));
}
#[test]
fn serve_nonexistent_file() {
let tmp = tempfile::tempdir().unwrap();
let response = send_file(tmp.path(), "nope.css");
assert_eq!(response.status, 404);
}
#[test]
fn serve_traversal_attempt() {
let tmp = tempfile::tempdir().unwrap();
let response = send_file(tmp.path(), "../etc/passwd");
assert_eq!(response.status, 403);
}
}

View file

@ -0,0 +1,13 @@
[package]
name = "wo-store"
version = "0.1.0"
edition = "2021"
[dependencies]
wo-model = { path = "../wo-model" }
wo-seg = { path = "../wo-seg" }
wo-index = { path = "../wo-index" }
serde_json = { workspace = true }
[dev-dependencies]
tempfile = "3"

355
crates/wo-store/src/lib.rs Normal file
View file

@ -0,0 +1,355 @@
use std::collections::HashMap;
use std::io;
use std::path::{Path, PathBuf};
use wo_index::{DateIndex, TagIndex, TitleIndex};
use wo_model::{Article, ContentLoader};
use wo_seg::{SegReader, SegWriter};
/// Unified storage engine composing .seg files and indexes.
///
/// Provides the query API that the rest of the system uses.
/// Handles cold-start (rebuild from content/) and incremental ingestion.
pub struct Store {
seg_path: PathBuf,
index_dir: PathBuf,
content_dir: PathBuf,
title_idx: TitleIndex,
date_idx: DateIndex,
tag_idx: TagIndex,
/// Monotonic version counter per sys_title (for subscription diffs).
versions: HashMap<String, u64>,
version_counter: u64,
}
impl Store {
/// Open an existing data directory, or cold-start from content/.
///
/// If `data_dir` contains valid .seg and index files, opens them.
/// Otherwise, rebuilds everything from `content_dir`.
pub fn open(content_dir: &Path, data_dir: &Path) -> io::Result<Self> {
let seg_path = data_dir.join("articles.seg");
let index_dir = data_dir.join("index");
if seg_path.exists() && index_dir.join("title.idx").exists() {
Self::open_existing(content_dir, &seg_path, &index_dir)
} else {
let mut store = Self::init_empty(content_dir, &seg_path, &index_dir)?;
store.rebuild()?;
Ok(store)
}
}
fn open_existing(
content_dir: &Path,
seg_path: &Path,
index_dir: &Path,
) -> io::Result<Self> {
let title_idx = TitleIndex::open(&index_dir.join("title.idx"))?;
let date_idx = DateIndex::open(&index_dir.join("date.idx"))?;
let tag_idx = TagIndex::open(&index_dir.join("tags.idx"))?;
// Build version map from existing seg records.
let reader = SegReader::open(seg_path)?;
let mut versions = HashMap::new();
let mut version_counter = 0u64;
for (_, article) in reader.iter()? {
version_counter += 1;
versions.insert(article.sys_title.clone(), version_counter);
}
Ok(Self {
seg_path: seg_path.to_path_buf(),
index_dir: index_dir.to_path_buf(),
content_dir: content_dir.to_path_buf(),
title_idx,
date_idx,
tag_idx,
versions,
version_counter,
})
}
fn init_empty(
content_dir: &Path,
seg_path: &Path,
index_dir: &Path,
) -> io::Result<Self> {
std::fs::create_dir_all(seg_path.parent().unwrap())?;
std::fs::create_dir_all(index_dir)?;
// Create empty seg file.
let _writer = SegWriter::create(seg_path)?;
// Create empty indexes.
TitleIndex::build(&index_dir.join("title.idx"), &[])?;
DateIndex::build(&index_dir.join("date.idx"), &mut [])?;
TagIndex::build(&index_dir.join("tags.idx"), &[])?;
let title_idx = TitleIndex::open(&index_dir.join("title.idx"))?;
let date_idx = DateIndex::open(&index_dir.join("date.idx"))?;
let tag_idx = TagIndex::open(&index_dir.join("tags.idx"))?;
Ok(Self {
seg_path: seg_path.to_path_buf(),
index_dir: index_dir.to_path_buf(),
content_dir: content_dir.to_path_buf(),
title_idx,
date_idx,
tag_idx,
versions: HashMap::new(),
version_counter: 0,
})
}
/// Full rebuild: load all articles from content/, rewrite .seg and indexes.
pub fn rebuild(&mut self) -> io::Result<()> {
let articles = ContentLoader::load_all(&self.content_dir)?;
// Write all articles to a new .seg file.
let mut writer = SegWriter::create(&self.seg_path)?;
let mut index_entries: Vec<(String, i64, Vec<String>, u64)> = Vec::new();
self.versions.clear();
self.version_counter = 0;
for article in &articles {
let offset = writer.append(article)?;
let timestamp = article.published_on.unwrap_or(0);
index_entries.push((
article.sys_title.clone(),
timestamp,
article.tags.clone(),
offset,
));
self.version_counter += 1;
self.versions
.insert(article.sys_title.clone(), self.version_counter);
}
writer.sync()?;
// Rebuild all indexes.
wo_index::rebuild_all(&self.index_dir, &index_entries)?;
// Reload indexes.
self.title_idx = TitleIndex::open(&self.index_dir.join("title.idx"))?;
self.date_idx = DateIndex::open(&self.index_dir.join("date.idx"))?;
self.tag_idx = TagIndex::open(&self.index_dir.join("tags.idx"))?;
Ok(())
}
/// Look up a single article by sys_title.
pub fn get_by_title(&self, sys_title: &str) -> io::Result<Option<Article>> {
match self.title_idx.get(sys_title) {
Some(offset) => {
let reader = SegReader::open(&self.seg_path)?;
reader.read_at(offset)
}
None => Ok(None),
}
}
/// List published articles, most recent first, with pagination.
pub fn list_published(&self, skip: usize, limit: usize) -> io::Result<Vec<Article>> {
let offsets = self.date_idx.latest(skip + limit);
let reader = SegReader::open(&self.seg_path)?;
let mut articles = Vec::new();
for &offset in offsets.iter().skip(skip).take(limit) {
if let Some(article) = reader.read_at(offset)? {
if article.published {
articles.push(article);
}
}
}
Ok(articles)
}
/// List all articles with a given tag.
pub fn list_by_tag(&self, tag: &str) -> io::Result<Vec<Article>> {
let offsets = match self.tag_idx.get(tag) {
Some(offsets) => offsets,
None => return Ok(vec![]),
};
let reader = SegReader::open(&self.seg_path)?;
let mut articles = Vec::new();
for &offset in offsets {
if let Some(article) = reader.read_at(offset)? {
articles.push(article);
}
}
Ok(articles)
}
/// List articles published within a timestamp range.
pub fn list_by_date_range(&self, start: i64, end: i64) -> io::Result<Vec<Article>> {
let offsets = self.date_idx.range(start, end);
let reader = SegReader::open(&self.seg_path)?;
let mut articles = Vec::new();
for offset in offsets {
if let Some(article) = reader.read_at(offset)? {
articles.push(article);
}
}
Ok(articles)
}
/// Count of published articles.
pub fn count_published(&self) -> io::Result<usize> {
// For now, iterate and count. With small article counts this is fine.
let reader = SegReader::open(&self.seg_path)?;
let count = reader
.iter()?
.into_iter()
.filter(|(_, a)| a.published)
.count();
Ok(count)
}
/// Ingest a single article from a JSON file path.
///
/// Appends to .seg and triggers a full index rebuild.
/// Returns the sys_title of the ingested article.
pub fn ingest_article(&mut self, json_path: &Path) -> io::Result<String> {
let article = ContentLoader::load_from_file(json_path)?;
let sys_title = article.sys_title.clone();
// For simplicity, rebuild the entire store.
// A future optimization can do incremental append + index update.
self.rebuild()?;
Ok(sys_title)
}
/// Get the current version number for a sys_title.
pub fn article_version(&self, sys_title: &str) -> Option<u64> {
self.versions.get(sys_title).copied()
}
/// Get a reference to the content directory path.
pub fn content_dir(&self) -> &Path {
&self.content_dir
}
}
#[cfg(test)]
mod tests {
use super::*;
use std::fs;
fn create_test_article(base: &Path, sys_title: &str, published: bool, tags: &[&str], ts: i64) {
let dir = base.join(sys_title);
fs::create_dir_all(&dir).unwrap();
let tags_json: Vec<String> = tags.iter().map(|t| format!("\"{}\"", t)).collect();
let json = format!(
r#"{{
"title": "{}",
"sys_title": "{}",
"published": {},
"content": {{
"author": "Author",
"content": {{
"sections": [{{ "heading": "Intro", "paragraphs": ["Hello."] }}],
"codes": [],
"images": []
}},
"tags": [{}],
"publishedOn": {}
}},
"published_on": {}
}}"#,
sys_title,
sys_title,
published,
tags_json.join(", "),
ts,
ts
);
fs::write(dir.join(format!("{}.json", sys_title)), json).unwrap();
}
#[test]
fn cold_start_and_queries() {
let tmp = tempfile::tempdir().unwrap();
let content = tmp.path().join("content");
let data = tmp.path().join("data");
create_test_article(&content, "art-one", true, &["rust"], 1000);
create_test_article(&content, "art-two", true, &["rust", "linux"], 2000);
create_test_article(&content, "art-three", false, &["draft"], 3000);
create_test_article(&content, "art-four", true, &["linux"], 4000);
let store = Store::open(&content, &data).unwrap();
// Title lookup.
let art = store.get_by_title("art-one").unwrap().unwrap();
assert_eq!(art.sys_title, "art-one");
assert!(store.get_by_title("nonexistent").unwrap().is_none());
// Tag query.
let rust_articles = store.list_by_tag("rust").unwrap();
assert_eq!(rust_articles.len(), 2);
let linux_articles = store.list_by_tag("linux").unwrap();
assert_eq!(linux_articles.len(), 2);
// Date range.
let range = store.list_by_date_range(1500, 3500).unwrap();
assert_eq!(range.len(), 2); // art-two(2000) and art-three(3000)
// Published count.
let count = store.count_published().unwrap();
assert_eq!(count, 3);
// Version tracking.
assert!(store.article_version("art-one").is_some());
assert!(store.article_version("nonexistent").is_none());
}
#[test]
fn rebuild_after_delete() {
let tmp = tempfile::tempdir().unwrap();
let content = tmp.path().join("content");
let data = tmp.path().join("data");
create_test_article(&content, "art-one", true, &["test"], 1000);
// First open: cold start.
let store = Store::open(&content, &data).unwrap();
assert_eq!(store.count_published().unwrap(), 1);
drop(store);
// Delete data/ and reopen: should rebuild.
fs::remove_dir_all(&data).unwrap();
let store = Store::open(&content, &data).unwrap();
assert_eq!(store.count_published().unwrap(), 1);
}
#[test]
fn reopen_existing() {
let tmp = tempfile::tempdir().unwrap();
let content = tmp.path().join("content");
let data = tmp.path().join("data");
create_test_article(&content, "art-one", true, &["test"], 1000);
// First open: cold start builds .seg + indexes.
let store = Store::open(&content, &data).unwrap();
assert_eq!(store.count_published().unwrap(), 1);
drop(store);
// Second open: loads existing files, no rebuild.
let store = Store::open(&content, &data).unwrap();
assert_eq!(store.count_published().unwrap(), 1);
let art = store.get_by_title("art-one").unwrap().unwrap();
assert_eq!(art.sys_title, "art-one");
}
}

16
crates/wo-sub/Cargo.toml Normal file
View file

@ -0,0 +1,16 @@
[package]
name = "wo-sub"
version = "0.1.0"
edition = "2021"
[dependencies]
wo-model = { path = "../wo-model" }
wo-store = { path = "../wo-store" }
wo-event = { path = "../wo-event" }
libc = { workspace = true }
bincode = { workspace = true }
serde = { workspace = true }
serde_json = { workspace = true }
[dev-dependencies]
tempfile = "3"

470
crates/wo-sub/src/lib.rs Normal file
View file

@ -0,0 +1,470 @@
use std::collections::HashMap;
use std::io;
use std::os::unix::io::RawFd;
use serde::{Deserialize, Serialize};
use wo_event::EventFd;
/// A subscription query pattern.
#[derive(Debug, Clone, PartialEq)]
pub enum Subscription {
/// Subscribe to changes for a single article by sys_title.
ByTitle(String),
/// Subscribe to changes for all articles with a given tag.
ByTag(String),
/// Subscribe to all content changes.
All,
}
/// The type of content change.
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
pub enum ChangeType {
Created,
Updated,
Deleted,
}
/// Notification payload written to subscriber fds.
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
pub struct Notification {
pub change_type: ChangeType,
pub sys_title: String,
pub version: u64,
}
/// Wire format for notification payloads.
#[derive(Debug, Clone, Copy, PartialEq)]
pub enum FdFormat {
/// Length-prefixed bincode (for internal subscribers).
Binary,
/// Length-prefixed JSON (for browser socket fds).
Json,
}
/// Manages subscriptions and delivers notifications to subscriber fds.
///
/// Subscribers register interest via `subscribe()` with a query pattern.
/// When content changes, `notify()` evaluates which subscriptions match
/// and writes a length-prefixed notification to each matching fd.
pub struct SubscriptionManager {
/// sys_title → list of subscriber fds.
by_title: HashMap<String, Vec<RawFd>>,
/// tag → list of subscriber fds.
by_tag: HashMap<String, Vec<RawFd>>,
/// Fds subscribed to all changes.
global: Vec<RawFd>,
/// fd → list of subscriptions (for cleanup on disconnect).
fd_registry: HashMap<RawFd, Vec<Subscription>>,
/// fd → wire format.
fd_format: HashMap<RawFd, FdFormat>,
/// EventFd for signaling the event loop that notifications are pending.
notify_efd: EventFd,
}
impl SubscriptionManager {
/// Create a new subscription manager.
pub fn new() -> io::Result<Self> {
Ok(Self {
by_title: HashMap::new(),
by_tag: HashMap::new(),
global: Vec::new(),
fd_registry: HashMap::new(),
fd_format: HashMap::new(),
notify_efd: EventFd::new()?,
})
}
/// The eventfd for signaling the event loop.
pub fn notify_fd(&self) -> RawFd {
self.notify_efd.fd()
}
/// Register a subscription for the given fd (binary format).
pub fn subscribe(&mut self, fd: RawFd, query: Subscription) {
self.subscribe_with_format(fd, query, FdFormat::Binary);
}
/// Register a subscription for the given fd (JSON format for browsers).
pub fn subscribe_json(&mut self, fd: RawFd, query: Subscription) {
self.subscribe_with_format(fd, query, FdFormat::Json);
}
fn subscribe_with_format(&mut self, fd: RawFd, query: Subscription, format: FdFormat) {
match &query {
Subscription::ByTitle(title) => {
self.by_title.entry(title.clone()).or_default().push(fd);
}
Subscription::ByTag(tag) => {
self.by_tag.entry(tag.clone()).or_default().push(fd);
}
Subscription::All => {
self.global.push(fd);
}
}
self.fd_registry.entry(fd).or_default().push(query);
self.fd_format.insert(fd, format);
}
/// Remove all subscriptions for a fd (e.g., on EPOLLHUP).
pub fn unsubscribe(&mut self, fd: RawFd) {
self.fd_format.remove(&fd);
if let Some(subs) = self.fd_registry.remove(&fd) {
for sub in subs {
match sub {
Subscription::ByTitle(title) => {
if let Some(fds) = self.by_title.get_mut(&title) {
fds.retain(|&f| f != fd);
}
}
Subscription::ByTag(tag) => {
if let Some(fds) = self.by_tag.get_mut(&tag) {
fds.retain(|&f| f != fd);
}
}
Subscription::All => {
self.global.retain(|&f| f != fd);
}
}
}
}
}
/// Notify subscribers about a content change.
///
/// Evaluates which subscriptions match the changed sys_title and tags,
/// then writes a length-prefixed notification to each matching fd.
pub fn notify(
&self,
sys_title: &str,
tags: &[String],
change_type: ChangeType,
version: u64,
) -> io::Result<usize> {
let notification = Notification {
change_type,
sys_title: sys_title.to_string(),
version,
};
// Pre-serialize both formats (only actually used if there are subscribers).
let bincode_payload = bincode::serialize(&notification).map_err(|e| {
io::Error::new(io::ErrorKind::InvalidData, format!("bincode: {}", e))
})?;
let json_payload = serde_json::to_vec(&notification).map_err(|e| {
io::Error::new(io::ErrorKind::InvalidData, format!("json: {}", e))
})?;
// Build wire formats: [u32 length][payload]
let bincode_wire = build_wire(&bincode_payload);
let json_wire = build_wire(&json_payload);
// Collect all fds that should receive this notification.
let mut target_fds: Vec<RawFd> = Vec::new();
// Title subscribers.
if let Some(fds) = self.by_title.get(sys_title) {
target_fds.extend(fds);
}
// Tag subscribers.
for tag in tags {
if let Some(fds) = self.by_tag.get(tag) {
target_fds.extend(fds);
}
}
// Global subscribers.
target_fds.extend(&self.global);
// Deduplicate.
target_fds.sort_unstable();
target_fds.dedup();
let mut sent = 0;
for &fd in &target_fds {
let format = self.fd_format.get(&fd).copied().unwrap_or(FdFormat::Binary);
let wire = match format {
FdFormat::Binary => &bincode_wire,
FdFormat::Json => &json_wire,
};
if write_all_fd(fd, wire).is_ok() {
sent += 1;
}
}
// Signal the event loop.
if sent > 0 {
let _ = self.notify_efd.write(1);
}
Ok(sent)
}
/// Number of active subscriber fds.
pub fn subscriber_count(&self) -> usize {
self.fd_registry.len()
}
}
fn build_wire(payload: &[u8]) -> Vec<u8> {
let len = payload.len() as u32;
let mut wire = Vec::with_capacity(4 + payload.len());
wire.extend_from_slice(&len.to_le_bytes());
wire.extend_from_slice(payload);
wire
}
/// Write all bytes to a raw fd.
fn write_all_fd(fd: RawFd, buf: &[u8]) -> io::Result<()> {
let mut written = 0;
while written < buf.len() {
let ret = unsafe {
libc::write(
fd,
buf[written..].as_ptr() as *const libc::c_void,
buf.len() - written,
)
};
if ret < 0 {
return Err(io::Error::last_os_error());
}
if ret == 0 {
return Err(io::Error::new(io::ErrorKind::WriteZero, "write returned 0"));
}
written += ret as usize;
}
Ok(())
}
/// Convenience macro for compile-time subscription wiring.
///
/// ```ignore
/// register!(sub_manager, fd, ByTitle("linux-misc"));
/// register!(sub_manager, fd, ByTag("rust"));
/// register!(sub_manager, fd, All);
/// ```
#[macro_export]
macro_rules! register {
($mgr:expr, $fd:expr, ByTitle($title:expr)) => {
$mgr.subscribe($fd, $crate::Subscription::ByTitle($title.into()))
};
($mgr:expr, $fd:expr, ByTag($tag:expr)) => {
$mgr.subscribe($fd, $crate::Subscription::ByTag($tag.into()))
};
($mgr:expr, $fd:expr, All) => {
$mgr.subscribe($fd, $crate::Subscription::All)
};
}
#[cfg(test)]
mod tests {
use super::*;
/// Create a pipe and return (read_fd, write_fd).
fn make_pipe() -> (RawFd, RawFd) {
let mut fds = [0i32; 2];
assert_eq!(unsafe { libc::pipe2(fds.as_mut_ptr(), libc::O_NONBLOCK) }, 0);
(fds[0], fds[1])
}
fn read_notification(fd: RawFd) -> Notification {
let mut len_buf = [0u8; 4];
let ret = unsafe { libc::read(fd, len_buf.as_mut_ptr() as *mut libc::c_void, 4) };
assert!(ret == 4, "failed to read length prefix");
let len = u32::from_le_bytes(len_buf) as usize;
let mut payload = vec![0u8; len];
let ret = unsafe { libc::read(fd, payload.as_mut_ptr() as *mut libc::c_void, len) };
assert_eq!(ret as usize, len);
bincode::deserialize(&payload).unwrap()
}
fn close_fd(fd: RawFd) {
unsafe { libc::close(fd) };
}
#[test]
fn title_subscription() {
let mut mgr = SubscriptionManager::new().unwrap();
let (r1, w1) = make_pipe();
let (r2, w2) = make_pipe();
mgr.subscribe(w1, Subscription::ByTitle("art-one".into()));
mgr.subscribe(w2, Subscription::ByTitle("art-one".into()));
let sent = mgr
.notify("art-one", &[], ChangeType::Updated, 1)
.unwrap();
assert_eq!(sent, 2);
let n1 = read_notification(r1);
assert_eq!(n1.sys_title, "art-one");
assert_eq!(n1.change_type, ChangeType::Updated);
assert_eq!(n1.version, 1);
let n2 = read_notification(r2);
assert_eq!(n2.sys_title, "art-one");
// Unrelated title: no notifications.
let sent = mgr
.notify("art-two", &[], ChangeType::Created, 2)
.unwrap();
assert_eq!(sent, 0);
close_fd(r1); close_fd(w1);
close_fd(r2); close_fd(w2);
}
#[test]
fn tag_subscription() {
let mut mgr = SubscriptionManager::new().unwrap();
let (r, w) = make_pipe();
mgr.subscribe(w, Subscription::ByTag("rust".into()));
let sent = mgr
.notify("art-one", &["rust".into(), "linux".into()], ChangeType::Updated, 1)
.unwrap();
assert_eq!(sent, 1);
let n = read_notification(r);
assert_eq!(n.sys_title, "art-one");
// No "rust" tag: no notification.
let sent = mgr
.notify("art-two", &["python".into()], ChangeType::Created, 2)
.unwrap();
assert_eq!(sent, 0);
close_fd(r); close_fd(w);
}
#[test]
fn global_subscription() {
let mut mgr = SubscriptionManager::new().unwrap();
let (r, w) = make_pipe();
mgr.subscribe(w, Subscription::All);
let sent = mgr.notify("anything", &[], ChangeType::Deleted, 5).unwrap();
assert_eq!(sent, 1);
let n = read_notification(r);
assert_eq!(n.change_type, ChangeType::Deleted);
close_fd(r); close_fd(w);
}
#[test]
fn unsubscribe_removes_fd() {
let mut mgr = SubscriptionManager::new().unwrap();
let (r, w) = make_pipe();
mgr.subscribe(w, Subscription::ByTitle("art-one".into()));
mgr.subscribe(w, Subscription::All);
assert_eq!(mgr.subscriber_count(), 1);
mgr.unsubscribe(w);
assert_eq!(mgr.subscriber_count(), 0);
let sent = mgr.notify("art-one", &[], ChangeType::Updated, 1).unwrap();
assert_eq!(sent, 0);
close_fd(r); close_fd(w);
}
#[test]
fn dedup_notifications() {
let mut mgr = SubscriptionManager::new().unwrap();
let (r, w) = make_pipe();
// Same fd subscribed to both title AND tag — should only get one notification.
mgr.subscribe(w, Subscription::ByTitle("art-one".into()));
mgr.subscribe(w, Subscription::ByTag("rust".into()));
let sent = mgr
.notify("art-one", &["rust".into()], ChangeType::Updated, 1)
.unwrap();
assert_eq!(sent, 1); // deduped
let n = read_notification(r);
assert_eq!(n.sys_title, "art-one");
close_fd(r); close_fd(w);
}
#[test]
fn json_format_subscription() {
let mut mgr = SubscriptionManager::new().unwrap();
let (r, w) = make_pipe();
mgr.subscribe_json(w, Subscription::ByTitle("art-one".into()));
mgr.notify("art-one", &[], ChangeType::Updated, 1).unwrap();
// Read length-prefixed JSON.
let mut len_buf = [0u8; 4];
let ret = unsafe { libc::read(r, len_buf.as_mut_ptr() as *mut libc::c_void, 4) };
assert_eq!(ret, 4);
let len = u32::from_le_bytes(len_buf) as usize;
let mut payload = vec![0u8; len];
let ret = unsafe { libc::read(r, payload.as_mut_ptr() as *mut libc::c_void, len) };
assert_eq!(ret as usize, len);
// Should be valid JSON.
let json: serde_json::Value = serde_json::from_slice(&payload).unwrap();
assert_eq!(json["sys_title"], "art-one");
assert_eq!(json["change_type"], "Updated");
assert_eq!(json["version"], 1);
close_fd(r); close_fd(w);
}
#[test]
fn mixed_format_subscribers() {
let mut mgr = SubscriptionManager::new().unwrap();
let (r_bin, w_bin) = make_pipe();
let (r_json, w_json) = make_pipe();
mgr.subscribe(w_bin, Subscription::ByTitle("art-one".into()));
mgr.subscribe_json(w_json, Subscription::ByTitle("art-one".into()));
let sent = mgr.notify("art-one", &[], ChangeType::Created, 5).unwrap();
assert_eq!(sent, 2);
// Binary subscriber gets bincode.
let n = read_notification(r_bin);
assert_eq!(n.sys_title, "art-one");
assert_eq!(n.version, 5);
// JSON subscriber gets JSON.
let mut len_buf = [0u8; 4];
unsafe { libc::read(r_json, len_buf.as_mut_ptr() as *mut libc::c_void, 4) };
let len = u32::from_le_bytes(len_buf) as usize;
let mut payload = vec![0u8; len];
unsafe { libc::read(r_json, payload.as_mut_ptr() as *mut libc::c_void, len) };
let json: serde_json::Value = serde_json::from_slice(&payload).unwrap();
assert_eq!(json["sys_title"], "art-one");
close_fd(r_bin); close_fd(w_bin);
close_fd(r_json); close_fd(w_json);
}
#[test]
fn register_macro() {
let mut mgr = SubscriptionManager::new().unwrap();
let (r, w) = make_pipe();
register!(mgr, w, ByTitle("test-article"));
register!(mgr, w, ByTag("rust"));
register!(mgr, w, All);
assert_eq!(mgr.subscriber_count(), 1);
close_fd(r); close_fd(w);
}
}

View file

@ -0,0 +1,11 @@
[package]
name = "wo-watch"
version = "0.1.0"
edition = "2021"
[dependencies]
wo-store = { path = "../wo-store" }
libc = { workspace = true }
[dev-dependencies]
tempfile = "3"

280
crates/wo-watch/src/lib.rs Normal file
View file

@ -0,0 +1,280 @@
use std::collections::HashMap;
use std::io;
use std::os::unix::io::RawFd;
use std::path::{Path, PathBuf};
/// A content change detected by the watcher.
#[derive(Debug, Clone, PartialEq)]
pub enum ContentChange {
Created(String), // sys_title
Modified(String), // sys_title
Deleted(String), // sys_title
}
/// Watches a content directory for article file changes using inotify.
///
/// Expects the directory structure:
/// ```text
/// content_dir/
/// {sys_title}/
/// {sys_title}.json
/// ```
pub struct ContentWatcher {
inotify_fd: RawFd,
/// Maps inotify watch descriptor → directory path.
wd_to_path: HashMap<i32, PathBuf>,
/// Maps directory path → watch descriptor.
path_to_wd: HashMap<PathBuf, i32>,
content_dir: PathBuf,
}
const EVENT_MASK: u32 =
(libc::IN_CREATE | libc::IN_MODIFY | libc::IN_DELETE | libc::IN_MOVED_TO | libc::IN_MOVED_FROM)
as u32;
impl ContentWatcher {
/// Create a new watcher on the given content directory.
///
/// Adds inotify watches on the content dir itself and all subdirectories.
pub fn new(content_dir: &Path) -> io::Result<Self> {
let fd = unsafe { libc::inotify_init1(libc::IN_NONBLOCK | libc::IN_CLOEXEC) };
if fd < 0 {
return Err(io::Error::last_os_error());
}
let mut watcher = Self {
inotify_fd: fd,
wd_to_path: HashMap::new(),
path_to_wd: HashMap::new(),
content_dir: content_dir.to_path_buf(),
};
// Watch the root content directory (for new article dirs).
watcher.add_watch(content_dir)?;
// Watch each existing subdirectory.
if content_dir.exists() {
for entry in std::fs::read_dir(content_dir)? {
let entry = entry?;
if entry.path().is_dir() {
watcher.add_watch(&entry.path())?;
}
}
}
Ok(watcher)
}
/// The inotify file descriptor, for registration on an event loop.
pub fn fd(&self) -> RawFd {
self.inotify_fd
}
/// Read and process pending inotify events.
///
/// Returns a list of content changes. Call this when the event loop
/// signals that the inotify fd is readable.
pub fn process_events(&mut self) -> io::Result<Vec<ContentChange>> {
let mut buf = [0u8; 4096];
let n = unsafe {
libc::read(
self.inotify_fd,
buf.as_mut_ptr() as *mut libc::c_void,
buf.len(),
)
};
if n < 0 {
let err = io::Error::last_os_error();
if err.raw_os_error() == Some(libc::EAGAIN) {
return Ok(vec![]);
}
return Err(err);
}
let mut changes = Vec::new();
let mut offset = 0usize;
while offset < n as usize {
let event = unsafe { &*(buf.as_ptr().add(offset) as *const libc::inotify_event) };
let name_len = event.len as usize;
let name = if name_len > 0 {
let name_ptr = unsafe { buf.as_ptr().add(offset + std::mem::size_of::<libc::inotify_event>()) };
let name_bytes = unsafe { std::slice::from_raw_parts(name_ptr, name_len) };
let end = name_bytes.iter().position(|&b| b == 0).unwrap_or(name_len);
Some(String::from_utf8_lossy(&name_bytes[..end]).to_string())
} else {
None
};
offset += std::mem::size_of::<libc::inotify_event>() + name_len;
let Some(name) = name else { continue };
let mask = event.mask;
// Event on the root content directory: subdirectory created/deleted.
if let Some(dir_path) = self.wd_to_path.get(&event.wd) {
let dir_path = dir_path.clone();
if dir_path == self.content_dir {
let sub_path = self.content_dir.join(&name);
if (mask & libc::IN_CREATE as u32) != 0 || (mask & libc::IN_MOVED_TO as u32) != 0 {
if sub_path.is_dir() {
let _ = self.add_watch(&sub_path);
changes.push(ContentChange::Created(name.clone()));
}
}
if (mask & libc::IN_DELETE as u32) != 0 || (mask & libc::IN_MOVED_FROM as u32) != 0 {
self.remove_watch(&sub_path);
changes.push(ContentChange::Deleted(name));
}
} else {
// Event inside an article subdirectory.
if name.ends_with(".json") || name.ends_with(".md") {
let sys_title = dir_path
.file_name()
.unwrap()
.to_string_lossy()
.to_string();
if (mask & libc::IN_MODIFY as u32) != 0 {
changes.push(ContentChange::Modified(sys_title));
} else if (mask & libc::IN_CREATE as u32) != 0
|| (mask & libc::IN_MOVED_TO as u32) != 0
{
changes.push(ContentChange::Created(sys_title));
} else if (mask & libc::IN_DELETE as u32) != 0
|| (mask & libc::IN_MOVED_FROM as u32) != 0
{
changes.push(ContentChange::Deleted(sys_title));
}
}
}
}
}
// Deduplicate: keep only the last change per sys_title.
let mut seen = HashMap::new();
for change in changes {
let key = match &change {
ContentChange::Created(t) | ContentChange::Modified(t) | ContentChange::Deleted(t) => {
t.clone()
}
};
seen.insert(key, change);
}
Ok(seen.into_values().collect())
}
fn add_watch(&mut self, path: &Path) -> io::Result<()> {
let c_path = std::ffi::CString::new(path.to_str().unwrap_or("")).map_err(|_| {
io::Error::new(io::ErrorKind::InvalidInput, "invalid path for inotify")
})?;
let wd = unsafe { libc::inotify_add_watch(self.inotify_fd, c_path.as_ptr(), EVENT_MASK) };
if wd < 0 {
return Err(io::Error::last_os_error());
}
self.wd_to_path.insert(wd, path.to_path_buf());
self.path_to_wd.insert(path.to_path_buf(), wd);
Ok(())
}
fn remove_watch(&mut self, path: &Path) {
if let Some(wd) = self.path_to_wd.remove(path) {
unsafe { libc::inotify_rm_watch(self.inotify_fd, wd) };
self.wd_to_path.remove(&wd);
}
}
}
impl Drop for ContentWatcher {
fn drop(&mut self) {
unsafe { libc::close(self.inotify_fd) };
}
}
#[cfg(test)]
mod tests {
use super::*;
use std::fs;
#[test]
fn detect_json_modification() {
let tmp = tempfile::tempdir().unwrap();
let content = tmp.path().join("content");
let art_dir = content.join("my-article");
fs::create_dir_all(&art_dir).unwrap();
fs::write(art_dir.join("my-article.json"), r#"{"test": true}"#).unwrap();
let mut watcher = ContentWatcher::new(&content).unwrap();
// Modify the file.
fs::write(art_dir.join("my-article.json"), r#"{"test": false}"#).unwrap();
// Give inotify a moment.
std::thread::sleep(std::time::Duration::from_millis(50));
let changes = watcher.process_events().unwrap();
assert!(!changes.is_empty());
assert!(changes.iter().any(|c| matches!(c, ContentChange::Modified(t) if t == "my-article")));
}
#[test]
fn detect_new_directory() {
let tmp = tempfile::tempdir().unwrap();
let content = tmp.path().join("content");
fs::create_dir_all(&content).unwrap();
let mut watcher = ContentWatcher::new(&content).unwrap();
// Create a new article directory.
let art_dir = content.join("new-article");
fs::create_dir_all(&art_dir).unwrap();
std::thread::sleep(std::time::Duration::from_millis(50));
let changes = watcher.process_events().unwrap();
assert!(changes.iter().any(|c| matches!(c, ContentChange::Created(t) if t == "new-article")));
}
#[test]
fn detect_json_delete() {
let tmp = tempfile::tempdir().unwrap();
let content = tmp.path().join("content");
let art_dir = content.join("delete-me");
fs::create_dir_all(&art_dir).unwrap();
fs::write(art_dir.join("delete-me.json"), "{}").unwrap();
let mut watcher = ContentWatcher::new(&content).unwrap();
// Delete the JSON file.
fs::remove_file(art_dir.join("delete-me.json")).unwrap();
std::thread::sleep(std::time::Duration::from_millis(50));
let changes = watcher.process_events().unwrap();
assert!(changes.iter().any(|c| matches!(c, ContentChange::Deleted(t) if t == "delete-me")));
}
#[test]
fn ignore_non_json_files() {
let tmp = tempfile::tempdir().unwrap();
let content = tmp.path().join("content");
let art_dir = content.join("my-article");
fs::create_dir_all(&art_dir).unwrap();
let mut watcher = ContentWatcher::new(&content).unwrap();
// Create a non-JSON file.
fs::write(art_dir.join("notes.txt"), "not json").unwrap();
std::thread::sleep(std::time::Duration::from_millis(50));
let changes = watcher.process_events().unwrap();
// Non-JSON files should not trigger Modified events.
assert!(changes.iter().all(|c| !matches!(c, ContentChange::Modified(_))));
}
}

122
docs/00-linux.md Normal file
View file

@ -0,0 +1,122 @@
## Linux Kernel Features
Kernel primitives that the writeonce binary can leverage, mapped to the architectural needs identified in [01-problem.md](./01-problem.md) and [02-recovery.md](./02-recovery.md).
### File Watching — Content Directory
| Syscall | Purpose |
| ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| `inotify` | Watch the content directory for file creates, modifications, and deletes. Triggers re-indexing and subscriber notification when articles change. Replaces the S3 + Lambda event pipeline entirely. |
| `fanotify` | Alternative to inotify with broader scope (filesystem-level events). Useful if watching needs to span mount points or require permission-based filtering. |
### Async I/O — Server and Subscription Manager
| Syscall | Purpose |
| ---------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| `epoll` | Event-driven I/O multiplexing for the HTTP server and SSE connections. Handles many concurrent subscriber connections without a thread per client. Foundation for the async runtime. |
| `io_uring` | Modern async I/O interface (Linux 5.1+). Supports batched, zero-copy submission of read/write/accept operations. Candidate for the embedded storage engine's disk reads and the HTTP server's socket handling in a single event loop. |
### Efficient File Serving — Content Delivery
| Syscall | Purpose |
| ---------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| `sendfile` | Zero-copy transfer from file descriptor to socket. Serves markdown files and static frontend assets directly from disk to the client without copying through userspace. |
| `splice` / `tee` | Zero-copy data transfer between file descriptors via kernel pipe buffers. Useful for streaming .seg file reads directly to HTTP responses. |
### Embedded Storage — .seg Files and Indexing
| Syscall | Purpose |
| -------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| `mmap` | Memory-map .seg index files for O(1)/O(log n) lookups by `blog-title` without loading entire files into heap memory. Keeps the storage engine's memory footprint proportional to working set, not total content size. |
| `madvise` | Hint to the kernel about mmap access patterns (`MADV_SEQUENTIAL` for full scans, `MADV_RANDOM` for index lookups). Improves page cache behavior for both date-ordered listing and title-based retrieval. |
| `fallocate` | Pre-allocate disk space for .seg files and append-only logs. Prevents fragmentation and ensures writes don't fail mid-operation due to disk pressure. |
| `pread` / `pwritev2` | Positional read/write without seeking. Allows concurrent reads from different offsets in the same .seg file without locking a shared file offset. Pairs well with io_uring for batched operations. |
### Event Notification — Subscription Manager
| Syscall | Purpose |
| --------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| `eventfd` | Lightweight signaling between the file watcher thread and the subscription manager. When inotify detects a content change, eventfd wakes the async event loop to push diffs to SSE subscribers. |
| `timerfd` | Timer as a file descriptor. Can drive periodic tasks (index compaction, subscriber keepalive pings) within the same epoll/io_uring event loop without a separate timer thread. |
### Process and Resource Management
| Syscall | Purpose |
| -------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------- |
| `pidfd` | File descriptor for the process itself. Enables clean self-monitoring and graceful shutdown signaling within the single-binary model. |
| `memfd_create` | Anonymous memory-backed file descriptors. Useful for transient storage (e.g., building a .seg index in memory before atomically swapping it to disk). |
## Runtime
- Develop async runtime environment built on the kernel primitives above.
- Core event loop: `epoll` or `io_uring` driving all I/O (file watch, HTTP, SSE, storage reads).
- Single-threaded event loop with thread pool for blocking operations (disk I/O, index rebuilds).
- All file descriptors (inotify, eventfd, timerfd, sockets) registered on one multiplexer — one process, one loop, no external dependencies.
## Database Subscription
No SSE. No WebSocket. Subscriptions are handled at the kernel level using file descriptors and the event loop that already drives the rest of the system.
### Why Not SSE or WebSocket
Both are HTTP-layer protocols that add framing, connection management, and protocol negotiation on top of what the kernel already provides. If the binary already owns the event loop (`epoll` / `io_uring`), the content directory (`inotify`), and the subscriber sockets (raw fds), there is no reason to layer another protocol on top. The kernel *is* the subscription engine.
### Register Macro
A subscription is a registration of interest in a content query, bound to a file descriptor. The `register` macro wires a query pattern to the event loop at compile time:
```rust
register! {
#{blog-title} => notify(fd)
}
```
This expands to:
1. **Parse the query pattern** — `#{blog-title}` resolves to a specific article's `sys_title` key in the .seg index.
2. **Bind to inotify watch** — The content file backing that `blog-title` gets an inotify watch descriptor. When the file is modified, inotify fires.
3. **Map watch → subscriber fd** — The event loop maps the inotify event to every file descriptor that registered interest in that `blog-title`.
4. **Write to fd** — The subscriber's socket fd receives the diff (or a notification payload) via a direct `write()` — no HTTP framing, no SSE `data:` lines, no WebSocket frames.
### Flow
```
inotify (content file changed)
|
v
epoll wakes → lookup blog-title in subscription table
|
v
for each registered fd:
write(fd, diff_payload)
```
### Subscription Table
The subscription table is an in-process map:
```
blog-title → Vec<RawFd>
```
- **Register**: client connects → fd is added to the vec for the requested `blog-title`.
- **Deregister**: client disconnects → fd is removed (detected via `EPOLLHUP` / `EPOLLRDHUP`).
- **Notify**: inotify fires for a content file → resolve `blog-title` from path → write to all registered fds.
No broker, no message queue, no protocol layer. The kernel's fd lifecycle (`epoll` for readiness, `inotify` for content changes, `close` for cleanup) handles the full subscription lifecycle.
### Query Patterns
The `register` macro supports patterns beyond single titles:
```rust
register! {
#{blog-title} // single article by sys_title
#[tag="rust"] // all articles matching a tag
#[published > "2026-01-01"] // date-range filter
#* // all content changes
}
```
Each pattern resolves to a set of inotify watch descriptors. When the watched set changes (new article added that matches the filter), the subscription table updates automatically during re-indexing.

66
docs/01-problem.md Normal file
View file

@ -0,0 +1,66 @@
# Problem Statement
The current writeonce architecture works, but it carries weight that the project doesn't need. This document identifies the structural problems that motivate the redesign described in [02-recovery.md](./02-recovery.md).
## Too Many Moving Parts
A single article edit touches five systems before it reaches a reader:
```
local file --> S3 bucket --> Lambda --> Rust API --> PostgreSQL --> Angular app
```
Each hop is a failure point. Each system has its own deployment, its own logs, its own configuration. The Lambda needs AWS credentials and an API token. The API needs a DATABASE_URL and an AWS_INFRA_BASE_URL. The infrastructure API needs its own AWS SDK config. For a platform that serves markdown files, this is disproportionate complexity.
## External Database for Derived Data
PostgreSQL stores articles as JSONB — but the database is not the source of truth. The articles repository is. The database is a derived cache that requires:
- A running PostgreSQL 17 instance
- Diesel ORM with migrations
- Connection pooling (r2d2)
- A separate Rust API server to mediate access
If the database dies, recovery means re-syncing every article from S3 through the Lambda pipeline. The database adds operational burden without adding authority.
## Five Repositories for One Product
The project spans five git submodules across four languages:
| Repo | Language | Purpose |
|------|----------|---------|
| writeonce-articles-s3 | JSON + MD | Content |
| lambda-function | Go | Sync trigger |
| writeonce-api | Rust | API server |
| aws-infra | Rust | AWS bridge |
| writeonce-app | Angular/TS | Frontend |
Each has its own CI/CD pipeline, Dockerfile, and deployment target. Coordinating changes across repos (e.g., adding a new article field) requires touching multiple codebases, multiple pipelines, and multiple deploys.
## No Real-Time Content Updates
The current flow is request-response only. When an article is updated:
1. Author syncs files to S3
2. Lambda fires and upserts via API
3. The frontend knows nothing until the next page load
There is no mechanism for the client to learn that content has changed. No subscriptions, no push, no invalidation. The architecture explicitly avoids WebSocket, but offers no alternative for real-time awareness.
## AWS Dependency for a File-Based System
The content is markdown and JSON files — inherently local, portable, and simple. But the current pipeline requires:
- An S3 bucket to host them
- A Lambda function to watch for changes
- AWS SDK configuration in two Rust services
- Pulumi infrastructure-as-code to manage the Lambda + S3 setup
- IAM credentials across multiple components
The cloud infrastructure exists to shuttle files from one place to another. The files themselves don't need the cloud — they need to be read, indexed, and served.
## Summary
The core problems are **accidental complexity** and **infrastructure overhead**. The content model (markdown + JSON metadata) is sound. The content-as-code principle is right. But the delivery mechanism — five repos, three languages on the backend, an external database, a cloud event pipeline — is heavier than the problem requires.
The question `02-recovery.md` answers: what if the database, the server, and the client were one thing?

158
docs/02-recovery.md Normal file
View file

@ -0,0 +1,158 @@
# Recovery — The Target Architecture
This document describes where writeonce is going: a single, self-contained binary that owns its own storage, serves its own content, and pushes updates to connected clients in real-time — with no external database, no cloud pipeline, and no separate API server.
## Guiding Principle
**Everything in one process.** The database, the server logic, and the client-facing interface all live in a single codebase and ship as a single executable. If you can run the binary, you have the full platform.
## Own Database
The current PostgreSQL instance is a derived cache — it stores JSONB copies of files that already exist as the source of truth. The recovery architecture eliminates this indirection entirely.
### What Changes
- **No external database.** No PostgreSQL, no Diesel ORM, no connection pooling, no migrations.
- **Local file storage.** Markdown files and JSON metadata files are stored in a local directory, just as they are today in `writeonce-articles-s3/`. The file system *is* the database.
- **Custom storage segments (.seg files).** Research area: segment files that provide efficient read access, indexing, and potentially append-only writes for content. Think of these as a lightweight, purpose-built storage layer — not a general-purpose database engine, but enough to support indexed lookups by `blog-title` and ordered listing by date.
- **Indexed by blog-title.** The `sys_title` / blog-title field remains the primary key for content retrieval. The embedded storage must support O(1) or O(log n) lookups by this field.
### What Stays the Same
- Articles are still structured as JSON metadata + Markdown content pairs.
- The `sys_title`, `published`, `tags`, `author`, and section structure remain the content model.
- Content is still the source of truth — but now it's read directly from local storage instead of being derived through a sync pipeline.
## No AWS Infrastructure
The current architecture uses S3 as a file host and Lambda as a sync trigger. In the target architecture, there is nothing to sync *to* — the files are already where they need to be.
### What Gets Removed
| Current Component | Why It Existed | Why It's No Longer Needed |
|---|---|---|
| S3 bucket | Remote file storage | Files live locally alongside the binary |
| Lambda function (Go) | Watch S3 for changes, call API | No remote store to watch — file changes are local |
| aws-infra service (Rust) | Bridge to AWS S3/EC2 APIs | No AWS dependency |
| Pulumi IaC | Manage Lambda + S3 resources | No cloud resources to manage |
### What Replaces It
The binary watches its own content directory. When a file changes (new article, updated metadata), the embedded database re-indexes and notifies subscribers. The deployment model becomes:
```
1. Place the binary on a server
2. Point it at a content directory
3. It serves
```
No credentials, no IAM roles, no SDK configuration.
## No Separate API
Today, `writeonce-api` is a standalone Actix-web server that mediates between the frontend and the database. In the target architecture, the server logic is embedded in the same process as the database and the content renderer.
### What This Means
- **No HTTP hop between database and server.** Queries go directly from the request handler to the storage engine in-process. No network serialization, no connection pool, no ORM layer.
- **Single codebase.** No multi-repo coordination. A new article field is added once — in the content model — and it flows through storage, indexing, and rendering in the same compilation unit.
- **Single deployment.** One binary, one container, one process. No docker-compose orchestrating API + database + infra services.
The binary still exposes HTTP endpoints — it's still a web server. But it's a web server with an embedded database, not a web server that talks to an external one.
## Real-Time Subscriptions Without WebSocket
The current architecture has no mechanism for pushing content updates to connected clients. The target architecture adds real-time subscriptions, but explicitly without WebSocket.
### Why Not WebSocket
WebSocket adds connection state management, heartbeat logic, reconnection handling, and protocol upgrade complexity. For a content platform where updates are infrequent (articles are published, not streamed), the overhead isn't justified.
### Subscription Model
The target is a subscription mechanism where:
- A client subscribes to a content query (e.g., "all published articles" or "article with sys_title X")
- When the underlying data changes, the server pushes the relevant diff to the subscriber
- No polling from the client side
Candidate approaches to research:
- **Server-Sent Events (SSE)** — unidirectional push over HTTP. Simple, well-supported, no protocol upgrade. Natural fit for infrequent content updates.
- **SpacetimeDB-style subscriptions** — clients register queries, the engine tracks which rows match, and only sends diffs when the result set changes. This is the aspirational model.
- **Long polling** — fallback option. Simple but less efficient than SSE for multiple subscribers.
The key constraint: the subscription mechanism must work without requiring clients to maintain persistent bidirectional connections.
## Target Architecture
```
content directory
(JSON + MD files, .seg index)
|
| file watch + re-index
v
+---------------------------+
| writeonce binary |
| |
| +-------------------+ |
| | embedded storage | | .seg files, blog-title index
| | (read/write/index)| |
| +-------------------+ |
| | |
| +-------------------+ |
| | server logic | | route handlers, content queries
| | (HTTP endpoints) | |
| +-------------------+ |
| | |
| +-------------------+ |
| | subscription mgr | | SSE / query-based push
| | (real-time push) | |
| +-------------------+ |
| |
+---------------------------+
|
HTTP / SSE
|
v
+-------------------+
| frontend app | Angular or successor
| (browser client) |
+-------------------+
```
## Single Repository
The five current repos collapse into one:
```
writeonce/
content/ # articles (JSON + MD), images, assets
storage/ # embedded database engine (.seg files, indexing)
server/ # HTTP handlers, subscription manager
frontend/ # client application
writeonce.toml # configuration (port, content dir, index settings)
```
One repo. One build. One deploy artifact.
## What Needs Research
| Area | Question | Notes |
|------|----------|-------|
| **.seg file format** | What storage format gives efficient indexed reads over JSON+MD content? | Look at LSM trees, append-only logs, SQLite's page format for inspiration |
| **File watching** | How to efficiently detect content changes on Linux/macOS? | `inotify` on Linux, `kqueue` on macOS, or cross-platform via `notify` crate |
| **SSE vs alternatives** | Is SSE sufficient for the subscription model, or is something custom needed? | SSE handles the "push diffs to subscribers" case well for low-frequency updates |
| **Index structure** | What index structure supports `blog-title` lookup + date-ordered listing? | B-tree or hash index for title, sorted set for date ordering |
| **Language choice** | Continue with Rust for the unified binary? | Rust fits: single binary output, no runtime, strong typing, existing team knowledge |
| **Frontend coupling** | Should the frontend be embedded in the binary (serve static assets) or remain separate? | Embedding simplifies deployment; separate allows independent frontend iteration |
## Migration Path
The transition from current to target doesn't have to be all-or-nothing:
1. **Phase 1** — Build the embedded storage engine. Read JSON+MD files from a local directory, index by `blog-title`, serve via HTTP. No AWS, no PostgreSQL. This alone replaces `writeonce-api` + `aws-infra` + `lambda-function` + PostgreSQL.
2. **Phase 2** — Add real-time subscriptions (SSE). Clients subscribe to content queries and receive push updates when files change.
3. **Phase 3** — Collapse repositories. Move frontend into the unified codebase. Ship as a single binary that serves both API and static assets.
Each phase produces a working system. The current architecture can run in parallel until the new one is ready.

184
docs/03-data.md Normal file
View file

@ -0,0 +1,184 @@
# Data Layer — Local Storage with Subscriptions
This document describes the embedded data layer that replaces PostgreSQL: local `.seg` files with indexing, and a subscription model where clients register queries and receive diffs on route visit — no polling required.
## .seg File Storage
The `.seg` (segment) format is the on-disk representation of article data. Each segment file holds serialized article content with positional indexing for fast lookups.
### Design Goals
- **No external database process.** The binary reads and writes `.seg` files directly. No socket connections, no protocol negotiation, no separate daemon.
- **Indexed by blog-title.** The primary access pattern is `GET /blog/:sys_title`. The storage layer must resolve a `sys_title` to its article content without scanning all files.
- **Append-friendly.** New articles and updates append to the segment. Deletes are tombstoned and compacted later.
- **Human-readable source.** The JSON + Markdown files remain the authoring format. `.seg` files are a derived index — if they're deleted, they can be rebuilt from the content directory.
### Proposed Structure
```
content/
linux-misc/
linux-misc.json # authored metadata (source of truth)
linux-misc.md # authored content (source of truth)
aws-lambda-pulumi/
aws-lambda-pulumi.json
aws-lambda-pulumi.md
data/
articles.seg # serialized article records
index/
title.idx # blog-title -> offset mapping
date.idx # publish date -> offset (sorted)
tags.idx # tag -> [offsets] (inverted index)
```
The `content/` directory is what the author edits. The `data/` directory is what the engine builds and queries. Losing `data/` is a cold start, not data loss.
### Segment File Internals
```
+------------------+
| segment header | magic bytes, version, record count
+------------------+
| record 0 | length-prefixed serialized article
+------------------+
| record 1 |
+------------------+
| ... |
+------------------+
| record N |
+------------------+
```
Each record is a length-prefixed byte sequence containing the full article (metadata + content merged). Records are addressed by byte offset from the start of the file.
### Index Files
**title.idx** — Hash map serialized to disk. Maps `sys_title` (string) to byte offset in `articles.seg`. Loaded into memory at startup for O(1) lookups.
**date.idx** — Sorted array of `(timestamp, offset)` pairs. Supports range queries for "articles published between X and Y" and ordered listing for the homepage.
**tags.idx** — Inverted index. Maps each tag string to a list of offsets. Supports "all articles tagged with X" queries.
On startup, index files are memory-mapped or loaded into heap. On content change, affected indexes are rebuilt incrementally.
## Subscription Model
The subscription model is inspired by SpacetimeDB: clients register queries, and the engine tracks which results match. When underlying data changes, only the relevant diffs are pushed to subscribers.
### How It Works
```
Client A Server Content Dir
| | |
|--- GET /blog/linux-misc -| |
| |-- read from .seg index ---->|
|<-- article + SSE stream -| |
| | |
| (subscribed to | |
| sys_title=linux-misc) | |
| | |
| |<-- file change detected ----|
| | |
| |-- re-index article -------->|
| |-- diff against last push -->|
| | |
|<-- SSE: updated content -| |
| | |
```
### Route-Based Subscription
When a user visits a route, the response includes both the current content and an SSE stream. The client is automatically subscribed to changes for that query — no explicit subscription handshake needed.
```
GET /blog/linux-misc
```
Response:
```
HTTP/1.1 200 OK
Content-Type: text/html
<!-- full article content rendered -->
<!-- SSE connection opened for this query -->
<script>
const source = new EventSource('/subscribe/blog/linux-misc');
source.onmessage = (event) => {
// apply diff to current content
};
</script>
```
The subscription lives as long as the browser tab is open. When the user navigates away, the EventSource closes and the server drops the subscription. No heartbeat management, no reconnection logic beyond what SSE provides natively (automatic reconnect is built into the EventSource API).
### Query Registration
Subscriptions are not limited to single-article lookups. The engine supports registering arbitrary content queries:
| Query Type | Example | Subscription Behavior |
|---|---|---|
| Single article | `sys_title = "linux-misc"` | Push when this specific article changes |
| All published | `published = true` | Push when any article is published or unpublished |
| By tag | `tags contains "rust"` | Push when a rust-tagged article is added, removed, or updated |
| Homepage list | `published = true ORDER BY date DESC LIMIT 10` | Push when the top-10 list changes |
The server maintains a registry of active subscriptions. On each content change, it evaluates which subscriptions are affected and pushes diffs only to those clients.
### Diff Format
When content changes, the server doesn't resend the full article. It sends a minimal diff:
```json
{
"type": "update",
"sys_title": "linux-misc",
"changes": {
"content.sections[2].paragraphs[0]": "Updated paragraph text...",
"content.tags": ["linux", "kernel", "new-tag"]
},
"version": 42
}
```
The `version` field enables clients to detect missed updates and request a full resync if needed.
## Sample Dataset
To validate the storage engine and subscription model, a sample dataset should exercise the core access patterns:
### Articles
| sys_title | tags | published | purpose |
|---|---|---|---|
| `sample-getting-started` | `[tutorial, beginner]` | true | Basic article, tests single-article subscription |
| `sample-rust-patterns` | `[rust, patterns]` | true | Tests tag-based queries |
| `sample-draft-wip` | `[draft]` | false | Tests published filter — should not appear in public queries |
| `sample-long-form` | `[deep-dive, rust]` | true | Multiple sections, images, code snippets — tests complex content rendering |
| `sample-frequently-updated` | `[changelog]` | true | Updated often — tests subscription diff delivery |
### Test Scenarios
1. **Cold start** — Delete `data/`, start the binary. It should rebuild `.seg` and index files from `content/` and serve all articles.
2. **Single article query** — `GET /blog/sample-getting-started` returns the article and opens an SSE subscription.
3. **Live update** — Edit `sample-frequently-updated.json` while a client is subscribed. The client should receive an SSE event with the diff.
4. **Tag query** — Subscribe to `tags contains "rust"`. Both `sample-rust-patterns` and `sample-long-form` should be in the result set. Adding a new article tagged `rust` should trigger a push.
5. **Publish toggle** — Change `sample-draft-wip` from `published: false` to `true`. Clients subscribed to the homepage list should receive a push with the new article added.
## SpacetimeDB Reference
SpacetimeDB is the primary architectural inspiration for the subscription model. Key concepts to study:
- **Modules** — server logic that runs inside the database, not beside it
- **Subscription queries** — clients register SQL-like queries; the engine evaluates them incrementally on each transaction
- **Incremental view maintenance** — only recompute the parts of a query result that changed
- **Client SDK generation** — type-safe client code generated from the server schema
Add SpacetimeDB as a reference submodule for quick access to their implementation patterns:
```bash
git submodule add https://github.com/clockworklabs/SpacetimeDB.git references/spacetimedb
```
The goal is not to replicate SpacetimeDB — it's to take its subscription semantics and apply them to a much narrower domain (blog content), where the simplicity of the problem allows a simpler implementation.

216
docs/04-ui.md Normal file
View file

@ -0,0 +1,216 @@
# User Interface — Server-Rendered HTMLX
No Angular. No React. No frontend framework. The UI is a set of `.htmlx` template files that the server parses, populates with content from the embedded database, and serves as plain HTML. Real-time updates arrive via SSE and are applied with minimal client-side scripting.
## Why Not Angular
The current `writeonce-app` is an Angular 18 SPA with Tailwind, PrismJS, ngx-markdown, and FontAwesome. It works, but it's a heavy delivery mechanism for what is fundamentally a read-heavy content site:
- **~200MB of `node_modules`** for a site that renders markdown articles
- **Client-side routing** for content that doesn't need it — every article is a distinct URL, not an interactive application
- **JavaScript-dependent rendering** — content doesn't exist until Angular boots, hydrates, and fetches from the API
- **Separate build pipeline** — `npm run build` produces static assets that must be deployed to nginx independently of the API
The content is static between updates. The interactivity is limited to navigation and code highlighting. A server-rendered approach matches the actual requirements.
## HTMLX Templates
The author defines the site layout using `.htmlx` files — HTML with embedded data bindings that the server resolves at render time.
### Template Structure
```
templates/
layout.htmlx # outer shell: <html>, <head>, <body>
header.htmlx # site header, navigation
footer.htmlx # site footer
home.htmlx # homepage: article list
article.htmlx # single article view
about.htmlx # static page
contact.htmlx # static page
components/
article-card.htmlx # summary card for article listings
code-snippet.htmlx # code block with language + title
img-caption.htmlx # image with caption
section.htmlx # article section (heading + paragraphs)
```
### Template Syntax
Templates use a binding syntax that references content from the database. The server parses these bindings, resolves them against the current content, and outputs plain HTML.
```html
<!-- header.htmlx -->
<header>
<nav>
<a href="/">writeonce</a>
<a href="/about">about</a>
<a href="/contact">contact</a>
</nav>
</header>
```
```html
<!-- article.htmlx -->
<article>
<h1>{{article.title}}</h1>
<p class="meta">by {{article.author}} &middot; {{article.tags}}</p>
{{#each article.sections}}
<section>
<h2>{{heading}}</h2>
{{#each paragraphs}}
<p>{{this}}</p>
{{/each}}
</section>
{{/each}}
{{#each article.codes}}
{{> code-snippet snippet=this}}
{{/each}}
{{#each article.images}}
{{> img-caption image=this}}
{{/each}}
</article>
```
```html
<!-- home.htmlx -->
<main>
<h1>articles</h1>
{{#each articles}}
{{> article-card article=this}}
{{/each}}
</main>
```
The `{{> partial}}` syntax includes another `.htmlx` file as a component. The server resolves these at render time — no client-side component tree.
### Content Subscription in Templates
Templates declare what data they need. The server resolves these declarations against the embedded database and subscribes the client to changes:
```html
<!-- article.htmlx -->
<!-- subscribe: article WHERE sys_title = :route_param -->
<article>
<h1>{{article.title}}</h1>
...
</article>
```
```html
<!-- home.htmlx -->
<!-- subscribe: articles WHERE published = true ORDER BY date DESC LIMIT 10 -->
<main>
{{#each articles}}
{{> article-card article=this}}
{{/each}}
</main>
```
The `<!-- subscribe: ... -->` comment is a directive to the server. It declares the query that populates the template's data context. The same query is used to register an SSE subscription for live updates (as described in [03-data.md](./03-data.md)).
## Rendering Pipeline
```
Browser request
|
v
Route match (/blog/linux-misc)
|
v
Load template (article.htmlx)
|
v
Parse subscribe directive
(article WHERE sys_title = "linux-misc")
|
v
Query embedded database (.seg index)
|
v
Resolve template bindings ({{article.title}}, etc.)
|
v
Compose with layout.htmlx + header.htmlx + footer.htmlx
|
v
Inject SSE subscription script
|
v
Send complete HTML response
```
The browser receives a fully rendered page on first load. No JavaScript framework boots. No API call fires. The content is already in the HTML.
## Live Updates via SSE
After the initial HTML is delivered, a small inline script opens an SSE connection for the page's subscription query:
```html
<script>
const source = new EventSource('/subscribe/blog/linux-misc');
source.onmessage = (event) => {
const diff = JSON.parse(event.data);
applyDiff(diff);
};
</script>
```
The `applyDiff` function is a lightweight client-side updater — it targets DOM elements by data attribute and patches their content. No virtual DOM, no reconciliation, no framework. For a content site where updates are infrequent and localized (a paragraph changed, a tag was added), direct DOM manipulation is sufficient.
```html
<h1 data-bind="article.title">Linux Misc</h1>
<p data-bind="article.sections[0].paragraphs[0]">First paragraph...</p>
```
When a diff arrives for `article.title`, the script finds the element with `data-bind="article.title"` and replaces its text content. This is the minimal client-side code the architecture requires.
## Code Highlighting
The current frontend uses PrismJS for syntax highlighting. In the server-rendered model, highlighting can happen at either layer:
**Server-side (preferred):** The server parses code blocks during template rendering and emits pre-highlighted HTML with CSS classes. The browser only needs the PrismJS CSS theme, not the JavaScript library. This eliminates client-side parsing entirely.
**Client-side (fallback):** Include PrismJS as a small script that runs on page load and on SSE update. Simpler to implement initially but adds a JavaScript dependency.
## Markdown Rendering
The current frontend uses `ngx-markdown` and `marked` to parse markdown in the browser. In the target architecture, markdown is rendered to HTML on the server during template composition. The browser never sees raw markdown.
This aligns with the content model: the JSON metadata already defines the article structure (sections, paragraphs, code snippets, images). The markdown file provides prose content. The server combines both into final HTML — the template just places the pre-rendered blocks.
## What Gets Removed
| Current (Angular) | Target (HTMLX) |
|---|---|
| `writeonce-app/` (full Angular project) | `templates/` (handful of .htmlx files) |
| `node_modules/` (~200MB) | None |
| `angular.json`, `tsconfig.json`, `karma.conf.js` | None |
| npm build pipeline | Template parsed at request time |
| Nginx static file serving | Binary serves its own HTML |
| Client-side routing | Server-side route matching |
| Client-side markdown parsing | Server-side rendering |
| Client-side code highlighting | Server-side or minimal JS |
## Styling
Templates use plain CSS. Tailwind can optionally be used as a build-time utility (generating a static CSS file), but there is no runtime CSS framework. The author writes styles in a `styles.css` file that the server serves as a static asset.
```
templates/
styles/
main.css # site-wide styles
article.css # article-specific styles
code-theme.css # syntax highlighting theme (PrismJS compatible)
```
## Template Authoring Experience
The `.htmlx` files are editable by the same author who writes articles. The template syntax is intentionally close to HTML — there's no JSX, no TypeScript, no build step. An author who knows HTML can modify the site layout.
This closes the loop on the writeonce philosophy: the author writes content (markdown + JSON) and layout (`.htmlx` + CSS) as files, and the binary turns them into a live site.

139
docs/05-datalayer.md Normal file
View file

@ -0,0 +1,139 @@
# Data Layer — Implementation Status
The embedded data layer described in [02-recovery.md](./02-recovery.md) and [03-data.md](./03-data.md) has been implemented as a Cargo workspace with 8 crates. All 44 tests pass. No external database, no AWS, no tokio — direct Linux syscalls on a custom event loop.
## Workspace Structure
```
writeonce-all/
Cargo.toml # workspace root
docs/ # architecture documentation
sample-content/ # 5 test articles for validation
crates/
wo-model/ # content model
wo-seg/ # .seg file format
wo-index/ # index files
wo-store/ # unified storage engine
wo-watch/ # inotify file watcher
wo-event/ # epoll event loop
wo-sub/ # subscription system
wo-rt/ # custom runtime
```
## Crate Summary
| Crate | Purpose | Tests | Key Types |
|-------|---------|-------|-----------|
| **wo-model** | Article structs matching existing JSON schema, `ContentLoader` for directory walking | 8 | `Article`, `ArticleContent`, `ArticleBody`, `Section`, `CodeSnippet`, `ContentLoader` |
| **wo-seg** | Binary `.seg` file format — length-prefixed records, tombstoning, positional I/O | 6 | `SegWriter`, `SegReader`, `SegHeader` |
| **wo-index** | Three index types for O(1) and O(log n) access patterns | 8 | `TitleIndex`, `DateIndex`, `TagIndex` |
| **wo-store** | Unified storage engine composing seg + indexes, cold-start rebuild | 3 | `Store` |
| **wo-watch** | Content directory watcher using inotify | 4 | `ContentWatcher`, `ContentChange` |
| **wo-event** | Custom event loop on epoll with eventfd, timerfd, signalfd | 5 | `EventLoop`, `EventFd`, `TimerFd`, `SignalFd` |
| **wo-sub** | Subscription manager with fd-based notifications, `register!` macro | 6 | `SubscriptionManager`, `Subscription`, `Notification` |
| **wo-rt** | Runtime tying all crates together — single process, single event loop | 4 | `Runtime`, `RuntimeHandle`, `Config` |
## Linux Kernel Syscalls Used
| Syscall | Crate | Purpose |
|---------|-------|---------|
| `pread` / `pwrite` | wo-seg | Positional read/write for .seg records without seeking |
| `fallocate` | wo-seg | Pre-allocate .seg file space to reduce fragmentation |
| `epoll_create1` / `epoll_ctl` / `epoll_wait` | wo-event | Event-driven I/O multiplexing for the main loop |
| `eventfd` | wo-event, wo-sub | Lightweight signaling between watcher and subscription manager |
| `timerfd_create` / `timerfd_settime` | wo-event | Periodic tasks (compaction, keepalive) as file descriptors |
| `signalfd` | wo-event | SIGINT/SIGTERM delivered as fd events for graceful shutdown |
| `inotify_init1` / `inotify_add_watch` | wo-watch | File system change detection on the content directory |
| `pipe2` | wo-sub (tests) | Mock subscriber fds for testing notification delivery |
## .seg File Format
```
Offset Size Field
0 4 Magic: b"WOSF"
4 2 Version: u16 LE (1)
6 2 Flags: u16 LE (reserved)
8 8 Record count: u64 LE
16 8 Data start offset: u64 LE
24 8 Reserved
32+ variable Records: [u32 length][u8 flags][bincode payload]...
```
- Records are addressed by byte offset from file start
- Flags: `0x00` = active, `0x01` = tombstoned
- Payload: bincode-serialized `Article` struct
## Index Files
| File | Format | Access Pattern |
|------|--------|----------------|
| `title.idx` | On-disk hash table (Robin Hood, load factor 0.5), 138 bytes/slot | O(1) lookup by `sys_title` |
| `date.idx` | Sorted `(i64 timestamp, u64 offset)` array, 16 bytes/entry | Binary search for date ranges, latest N |
| `tags.idx` | Bincode-serialized `HashMap<String, Vec<u64>>` | Tag-to-offsets inverted index |
All indexes are derived from `.seg` and rebuildable from `content/` on cold start.
## Subscription Model
No SSE. No WebSocket. Notifications are written directly to subscriber file descriptors.
- **Subscribe**: `SubscriptionManager::subscribe(fd, Subscription::ByTitle("linux-misc"))`
- **Notify**: on content change, length-prefixed `Notification` written to matching fds
- **Cleanup**: `EPOLLHUP` on epoll triggers automatic `unsubscribe(fd)`
- **Dedup**: if a fd matches multiple patterns (title + tag), it receives only one notification
Subscription patterns:
- `Subscription::ByTitle(sys_title)` — single article
- `Subscription::ByTag(tag)` — all articles with tag
- `Subscription::All` — all content changes
## Store Query API
```rust
store.get_by_title("linux-misc") -> Option<Article>
store.list_published(skip, limit) -> Vec<Article>
store.list_by_tag("rust") -> Vec<Article>
store.list_by_date_range(start, end) -> Vec<Article>
store.count_published() -> usize
store.article_version("linux-misc") -> Option<u64>
store.rebuild() // full rebuild from content/
```
## Runtime Event Loop
Single `epoll` instance multiplexing all file descriptors:
| Token | Fd | Handler |
|-------|----|---------|
| `WATCHER` | inotify fd | Process file changes → update store → notify subscribers |
| `SIGNAL` | signalfd | SIGINT/SIGTERM → graceful shutdown |
| `TIMER` | timerfd | Periodic tasks (compaction, stats) |
| `NOTIFY` | eventfd | Subscription notification signal |
| `1000+` | subscriber fds | Hangup detection → unsubscribe + cleanup |
## External Dependencies
| Crate | Version | Purpose |
|-------|---------|---------|
| `serde` | 1.x | Serialization derives |
| `serde_json` | 1.x | JSON parsing for article files |
| `bincode` | 1.x | Compact binary serialization for .seg records and notifications |
| `libc` | 0.2.x | Raw Linux syscall bindings |
No tokio. No async-std. No database driver. No HTTP framework (yet).
## What Comes Next
The data layer delivers everything the HTTP server and UI layers need:
1. **`Store` with zero-copy query access** — all article queries resolve in-process
2. **Subscription system accepting raw fds** — HTTP layer hands socket fds to `subscribe()`
3. **Shared event loop** — HTTP listener socket registers on the same epoll
4. **Automatic cold-start** — if `data/` is missing, rebuilds from `content/` on startup
5. **Graceful shutdown** — SIGTERM triggers clean fd cleanup
Next phases per [02-recovery.md](./02-recovery.md):
- **HTTP server** — route handlers using the `Store` query API, embedded in the same binary
- **HTMLX templates** — server-rendered HTML with `{{bindings}}` per [04-ui.md](./04-ui.md)
- **Frontend collapse** — serve static assets from the binary, eliminate the Angular app
3

246
docs/06-markdown-render.md Normal file
View file

@ -0,0 +1,246 @@
# Markdown File Rendering
## Current State (writeonce-articles-s3)
Each article is a directory containing a JSON metadata file and one or more `.md` files:
```
auto-scale-gitlab-runner-using-aws-spot-instance/
docker-machine-test-with-t2.md
gitlab-runner-config.md
stop-test-gitlab-docker-machine.md
gitlab-runner-with-kubernetes-executor/
gitlab-runner-with-kubernetes-executor.json
deploy.md
permission.md
role-binding.md
role-defination.md
gitlab-runnergitlab-runner-deploy.md
```
The JSON metadata currently defines the full article structure — sections, headings, paragraphs, and code snippet references. Markdown files are limited to code blocks referenced via the `codes[].snippet` field.
## Problem
The JSON metadata carries too much content. Headings, paragraphs, prose — all of this is duplicated as JSON strings inside `content.content.sections`. The markdown files only hold code snippets, referenced by `sectionIndex` and `paragraphIndex`.
This is backwards. The markdown file should be the content. The JSON should be minimal metadata.
## Target: Markdown-First Content Model
**The markdown file is the article.** All prose, headings, code blocks, and inline formatting live in the `.md` file. The JSON metadata file holds only what markdown cannot express: system fields, tags, publication state, and author.
### Minimal JSON Metadata
```json
{
"sys_title": "gitlab-runner-with-kubernetes-executor",
"title": "Gitlab Runner with Kubernetes Executor",
"published": true,
"author": "Shoney Arickathil",
"tags": ["kubernetes", "gitlab", "ci-cd"],
"published_on": 1740950884
}
```
No `content.content.sections`. No `content.content.codes`. No `paragraphs[]` arrays. No `sectionIndex`/`paragraphIndex` mapping.
### Markdown File = Full Article Content
````markdown
# Introduction
Deploying a Gitlab runner using kubernetes is a great option to overcome
the limitations of other gitlab runner executor such as docker and docker machine.
## Running Gitlab Runner in gitlab namespace
Create the namespace and apply the deployment:
```yaml
apiVersion: apps/v1
kind: Deployment
metadata:
name: gitlab-runner
namespace: gitlab
```
````
## Permissions
The runner needs RBAC permissions to create pods:
```yaml
apiVersion: rbac.authorization.k8s.io/v1
kind: Role
metadata:
name: gitlab-runner
```
Everything is in the markdown — headings, paragraphs, code blocks with language hints, links, images. The rendering pipeline parses the markdown directly.
### Directory Structure
```
content/
gitlab-runner-with-kubernetes-executor/
gitlab-runner-with-kubernetes-executor.json # minimal metadata
gitlab-runner-with-kubernetes-executor.md # full article content
linux-misc/
linux-misc.json
linux-misc.md
```
One JSON for metadata. One markdown for content. No scattered `.md` files per code snippet.
## What Changes
| Before | After |
| -------------------------------------------------------------- | ----------------------------------------------------------------------- |
| JSON holds sections, headings, paragraphs as structured arrays | JSON holds only sys_title, title, published, author, tags, published_on |
| Markdown files hold only code snippets | Markdown file holds the entire article |
| `codes[].snippet` maps filename to sectionIndex/paragraphIndex | No mapping needed — headings and code blocks are inline in markdown |
| Renderer reads JSON structure, injects code from .md files | Renderer parses markdown directly into HTML |
| Multiple .md files per article (one per code snippet) | One .md file per article |
## Impact on the Data Layer
### wo-model
The `Article` struct simplifies:
```rust
pub struct Article {
pub sys_title: String,
pub title: String,
pub published: bool,
pub author: String,
pub tags: Vec<String>,
pub published_on: Option<i64>,
}
```
The nested `ArticleContent` / `ArticleBody` / `Section` / `CodeSnippet` hierarchy is no longer needed. Article content comes from parsing the `.md` file at render time, not from the JSON.
### wo-md
Currently handles only inline markdown (`**bold**`, `` `code` ``, links). Needs to become a full markdown-to-HTML renderer:
- Block elements: headings (`#`, `##`), paragraphs, code fences (` `lang ```), lists, blockquotes
- Inline elements: bold, italic, code, links, images
- Code fence language extraction for `wo-md::highlight()`
- The renderer reads `{sys_title}/{sys_title}.md`, parses it, and returns HTML
### wo-htmlx
The `article.htmlx` template simplifies. Instead of iterating `{{#each article.content.content.sections}}`, it renders the pre-parsed markdown HTML:
```html
<article>
<h1>{{article.title}}</h1>
<p class="meta">by {{article.author}} &middot; {{article.tags}}</p>
{{article.content_html}}
</article>
```
Where `content_html` is the full HTML output from the markdown renderer.
### wo-store
`ContentLoader` reads the `.json` for metadata and the `.md` for content. The `.seg` file stores both. At query time, the markdown is either:
- Pre-rendered to HTML during ingestion (stored in .seg alongside metadata)
- Rendered on-demand at request time (read .md from disk)
Pre-rendering is preferred — it avoids parsing markdown on every HTTP request.
## Migration Path
1. Update `wo-model` with the simplified `Article` struct
2. Extend `wo-md` to handle full markdown (block-level parsing, code fences)
3. Update `ContentLoader` to read `.json` + `.md` pairs
4. Update `wo-store` to store pre-rendered HTML in the .seg file
5. Simplify `article.htmlx` template
6. Migrate existing articles: extract prose from JSON into `.md` files
Existing articles with the old JSON format can coexist during migration — `ContentLoader` checks for a `.md` file and falls back to the JSON structure if none exists.
## Blog Subscription — Live Content Reload
When a user visits `http://localhost:3000/blog/sample-rust-patterns`, the content should stay live. Any edit to `sample-content/sample-rust-patterns/sample-rust-patterns.md` must auto-reflect in the browser without a page refresh.
### How It Works
```
Browser visits /blog/sample-rust-patterns
│
▼
1. Server renders article HTML from .seg (pre-rendered from .md)
2. Server writes HTML response to socket fd
3. Server registers socket fd in subscription table:
register!(sub_manager, socket_fd, ByTitle("sample-rust-patterns"))
4. Connection transitions to Subscribed state (stays open)
│
│ (user edits sample-rust-patterns.md)
│
▼
5. inotify fires IN_MODIFY on sample-rust-patterns.md
6. ContentWatcher maps file → sys_title "sample-rust-patterns"
7. Store rebuilds: re-reads .json + .md, re-renders markdown to HTML, updates .seg + indexes
8. SubscriptionManager::notify("sample-rust-patterns", ...) fires
9. For each subscribed fd: write(fd, diff_payload)
│
▼
10. Browser receives payload on the open connection
11. Client-side script applies the update to the DOM
```
### What Needs to Work
| Component | Requirement |
|-----------|-------------|
| **inotify** (wo-watch) | Already watches `content/` directory. `.md` file changes must trigger `ContentChange::Modified(sys_title)` |
| **Store rebuild** (wo-store) | On `.md` change: re-read file, re-render markdown to HTML, update `.seg` and indexes |
| **Subscription table** (wo-sub) | Route handler registers the browser's socket fd via `register!` after sending initial HTML |
| **Notification** (wo-sub) | On content change, write updated `content_html` to all subscribed fds as JSON payload |
| **Event loop** (wo-rt) | After writing initial response, transition connection to `Subscribed` state. Keep fd on epoll for hangup detection. |
| **Client script** | Injected in the HTML. Reads payloads from the open connection. Replaces article content in the DOM. |
### Client-Side Script
Injected by the template renderer into every article page:
```html
<script>
// Connection stays open after initial HTML.
// Server writes length-prefixed JSON payloads when content changes.
const decoder = new TextDecoder();
const articleEl = document.querySelector('article');
fetch(window.location.href, { headers: { 'X-Subscribe': '1' } })
.then(r => r.body.getReader())
.then(reader => {
(function read() {
reader.read().then(({ done, value }) => {
if (done) return;
try {
const payload = JSON.parse(decoder.decode(value));
if (payload.content_html) {
articleEl.innerHTML = payload.content_html;
}
} catch (e) {}
read();
});
})();
});
</script>
```
### inotify and .md Files
The current `ContentWatcher` watches for `.json` file changes. It must also trigger on `.md` file changes:
- `IN_MODIFY` on `*.md` → `ContentChange::Modified(sys_title)`
- The sys_title is derived from the parent directory name (same as for JSON)
- Both `.json` and `.md` changes trigger a store rebuild and subscriber notification

252
docs/07-ssl.md Normal file
View file

@ -0,0 +1,252 @@
# SSL and Deployment
## Problem
In [01-problem.md](./01-problem.md), the infrastructure overhead was identified — multiple repos, AWS dependencies, separate deployment pipelines. But one problem went unaddressed: the server-side infrastructure that sits in front of the application — nginx reverse proxy, SSL certificates, systemd service management, and deployment to the production host.
Currently this requires manual SSH, manual nginx config, manual certbot runs. For a single-binary platform, the deployment should be as simple as the architecture.
## Target
Given:
- SSH access to `writeonce.de` is configured
- nginx exists at the default path `/etc/nginx/`
- The writeonce binary listens on a local port (e.g., `127.0.0.1:3000`)
The deployment pipeline should:
1. Build the binary
2. Copy it to the server
3. Create/update the systemd service
4. Restart the service
5. Configure nginx as a reverse proxy
6. Obtain and auto-renew SSL certificates via Let's Encrypt
## Systemd Service
The writeonce binary runs as a systemd service for automatic restart, logging, and boot-start.
### Service File
```ini
# /etc/systemd/system/writeonce.service
[Unit]
Description=writeonce content platform
After=network.target
[Service]
Type=simple
User=writeonce
Group=writeonce
WorkingDirectory=/opt/writeonce
ExecStart=/opt/writeonce/writeonce
Restart=on-failure
RestartSec=5
StandardOutput=journal
StandardError=journal
# Security hardening
NoNewPrivileges=true
ProtectSystem=strict
ProtectHome=true
ReadWritePaths=/opt/writeonce/data
PrivateTmp=true
[Install]
WantedBy=multi-user.target
```
### Directory Layout on Server
```
/opt/writeonce/
writeonce # the binary
content/ # article .json + .md files
data/ # derived .seg + .idx (rebuilt on start)
templates/ # .htmlx templates
static/ # CSS, images
```
### Service Management
```bash
# Install / update
sudo systemctl daemon-reload
sudo systemctl enable writeonce
sudo systemctl restart writeonce
# Check status
sudo systemctl status writeonce
journalctl -u writeonce -f
```
## Nginx Reverse Proxy
Nginx sits in front of the writeonce binary, handling SSL termination and proxying requests to `127.0.0.1:3000`.
### Nginx Config
```nginx
# /etc/nginx/sites-available/writeonce.de
server {
listen 80;
server_name writeonce.de www.writeonce.de;
return 301 https://$server_name$request_uri;
}
server {
listen 443 ssl http2;
server_name writeonce.de www.writeonce.de;
ssl_certificate /etc/letsencrypt/live/writeonce.de/fullchain.pem;
ssl_certificate_key /etc/letsencrypt/live/writeonce.de/privkey.pem;
ssl_protocols TLSv1.2 TLSv1.3;
ssl_ciphers HIGH:!aNULL:!MD5;
ssl_prefer_server_ciphers on;
# HSTS
add_header Strict-Transport-Security "max-age=31536000; includeSubDomains" always;
location / {
proxy_pass http://127.0.0.1:3000;
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $scheme;
# Keep connections open for database subscriptions
proxy_http_version 1.1;
proxy_set_header Connection "";
proxy_read_timeout 86400s;
proxy_send_timeout 86400s;
}
# Static assets — let nginx serve directly for better caching
location /static/ {
alias /opt/writeonce/static/;
expires 1y;
add_header Cache-Control "public, immutable";
}
}
```
### Enable Site
```bash
sudo ln -sf /etc/nginx/sites-available/writeonce.de /etc/nginx/sites-enabled/
sudo nginx -t
sudo systemctl reload nginx
```
## SSL with Let's Encrypt
### Initial Certificate
```bash
sudo apt install certbot python3-certbot-nginx
sudo certbot --nginx -d writeonce.de -d www.writeonce.de
```
Certbot modifies the nginx config to add SSL directives and obtains the certificate.
### Auto-Renewal
Certbot installs a systemd timer that runs twice daily:
```bash
# Check timer
systemctl list-timers | grep certbot
# Manual test
sudo certbot renew --dry-run
```
Certificates auto-renew before expiry. Nginx reloads automatically via certbot's deploy hook.
### Deploy Hook for Nginx Reload
```bash
# /etc/letsencrypt/renewal-hooks/deploy/reload-nginx.sh
#!/bin/bash
systemctl reload nginx
```
## Deployment Script
A single script that builds, copies, and restarts:
```bash
#!/bin/bash
# deploy.sh — run from the development machine
set -e
SERVER="writeonce.de"
REMOTE_DIR="/opt/writeonce"
echo "Building release binary..."
cargo build --release -p wo-rt --bin writeonce
echo "Copying binary to server..."
scp target/release/writeonce $SERVER:$REMOTE_DIR/writeonce.new
echo "Syncing content and templates..."
rsync -az --delete content/ $SERVER:$REMOTE_DIR/content/
rsync -az --delete templates/ $SERVER:$REMOTE_DIR/templates/
rsync -az --delete static/ $SERVER:$REMOTE_DIR/static/
echo "Swapping binary and restarting..."
ssh $SERVER "
sudo mv $REMOTE_DIR/writeonce.new $REMOTE_DIR/writeonce
sudo systemctl restart writeonce
"
echo "Deployed. Checking status..."
ssh $SERVER "sudo systemctl status writeonce --no-pager"
```
### First-Time Setup
Run once on the server to create the user, directory, and service:
```bash
#!/bin/bash
# setup.sh — run on the server
set -e
# Create user
sudo useradd -r -s /bin/false writeonce
# Create directory
sudo mkdir -p /opt/writeonce/{content,data,templates,static}
sudo chown -R writeonce:writeonce /opt/writeonce
# Install service
sudo cp writeonce.service /etc/systemd/system/
sudo systemctl daemon-reload
sudo systemctl enable writeonce
# Configure nginx
sudo cp writeonce.de.nginx /etc/nginx/sites-available/writeonce.de
sudo ln -sf /etc/nginx/sites-available/writeonce.de /etc/nginx/sites-enabled/
sudo nginx -t
sudo systemctl reload nginx
# SSL
sudo certbot --nginx -d writeonce.de -d www.writeonce.de
```
## What This Replaces
| Before | After |
|--------|-------|
| Pulumi IaC managing Lambda + S3 | `deploy.sh` with scp + rsync |
| AWS Lambda deployment pipeline | `systemctl restart writeonce` |
| S3 bucket for content hosting | `rsync content/` to server |
| Docker Compose for API + DB | Single binary, one systemd service |
| Multiple nginx configs for API + frontend | One nginx config, one proxy_pass |
| Manual SSL setup | `certbot --nginx` with auto-renewal |
## Connection Keepalive for Subscriptions
The nginx config sets `proxy_read_timeout 86400s` (24 hours) to keep persistent connections open for the database subscription model. When a browser visits an article page and the connection transitions to `Subscribed` state, nginx must not timeout and close the upstream connection.
If nginx is removed in the future (the binary handles TLS directly via `rustls`), this concern disappears — the binary owns the socket end-to-end.

View file

@ -0,0 +1,175 @@
# AI Agents and Content Management
## Context
AI agents (Claude Code, Copilot, Cursor, custom agents) work within a specific project or working directory. Their sessions, context, and understanding are scoped to that directory. This works well when projects are completely different domains.
But writeonce content is not isolated — articles reference each other, share tags, build on concepts from other articles. An agent editing `gitlab-runner-with-kubernetes-executor.md` would benefit from knowing that `auto-scale-gitlab-runner-using-aws-spot-instance.md` exists and covers related infrastructure. Without explicit mappings, the agent treats each article as an island.
## Problem
1. **Agents lack cross-article awareness.** When asked to write or update an article about Kubernetes, the agent doesn't know that related articles about Docker, CI/CD, or AWS already exist in the content directory — unless it manually searches.
2. **No semantic grouping.** Tags provide flat categorization (`kubernetes`, `ci-cd`), but they don't express relationships: "this article is a prerequisite for that one", "these three articles form a series", "this article supersedes that one."
3. **Context window waste.** Without mappings, the agent must scan all articles to find related content. With explicit mappings, it can load exactly the relevant files.
## Solution: Metadata-Driven Content Mappings
Users define relationships between articles in the JSON metadata. These mappings serve two purposes:
1. **Human navigation** — rendered as "related articles" links on the site
2. **Agent context** — when an agent works on an article, it loads the mapped articles into its context for cross-referencing
### Mapping Fields in JSON Metadata
Per [06-markdown-render.md](./06-markdown-render.md), the JSON metadata is minimal. Add a `mappings` field:
```json
{
"sys_title": "gitlab-runner-with-kubernetes-executor",
"title": "Gitlab Runner with Kubernetes Executor",
"published": true,
"author": "Shoney Arickathil",
"tags": ["kubernetes", "gitlab", "ci-cd"],
"published_on": 1740950884,
"mappings": {
"related": ["auto-scale-gitlab-runner-using-aws-spot-instance"],
"prerequisite": ["linux-misc"],
"series": {
"name": "gitlab-runner",
"order": 2
}
}
}
```
### Mapping Types
| Type | Meaning | Agent Use |
|------|---------|-----------|
| `related` | Topically related articles | Agent loads these for cross-reference when editing |
| `prerequisite` | Articles the reader should read first | Agent ensures no concept duplication, references prerequisites instead of re-explaining |
| `series` | Articles that form an ordered sequence | Agent maintains narrative continuity across the series |
| `supersedes` | This article replaces an older one | Agent can mark the old article as outdated or unpublished |
| `references` | External articles or URLs the content builds on | Agent checks links are still valid, cites them properly |
### Directory Structure with Mappings
```
content/
gitlab-runner-with-kubernetes-executor/
gitlab-runner-with-kubernetes-executor.json # metadata + mappings
gitlab-runner-with-kubernetes-executor.md # full article
auto-scale-gitlab-runner-using-aws-spot-instance/
auto-scale-gitlab-runner-using-aws-spot-instance.json
auto-scale-gitlab-runner-using-aws-spot-instance.md
linux-misc/
linux-misc.json
linux-misc.md
```
## Agent Workflows
### 1. Writing a New Article
The author asks an agent: "Write an article about deploying GitLab Runner on ECS."
The agent:
1. Scans the content directory for existing articles with tags `gitlab`, `ci-cd`, `aws`
2. Finds `gitlab-runner-with-kubernetes-executor` and `auto-scale-gitlab-runner-using-aws-spot-instance`
3. Reads their `.md` files to understand what's already covered
4. Writes the new article, referencing existing articles rather than re-explaining shared concepts
5. Suggests `mappings.related` entries for the new article's JSON
### 2. Updating an Existing Article
The author asks: "Update the Kubernetes executor article with the new runner token format."
The agent:
1. Reads the article's JSON metadata and `.md` content
2. Reads the `mappings.related` articles to check for consistency
3. Makes the update in the `.md` file
4. Checks if the change affects any prerequisite or series articles
5. inotify detects the `.md` change → store rebuilds → subscribers notified
### 3. Content Audit
The author asks: "Which articles reference outdated AWS configurations?"
The agent:
1. Loads all article metadata (the Store already indexes everything)
2. Follows `mappings` to build a dependency graph
3. Reads the `.md` files of articles tagged with `aws`
4. Identifies outdated patterns (old SDK versions, deprecated services)
5. Reports findings with links to specific articles and line numbers
### 4. Series Management
The author asks: "Add a new part to the gitlab-runner series."
The agent:
1. Finds all articles with `mappings.series.name == "gitlab-runner"`
2. Reads them in order to understand the narrative arc
3. Writes the new article continuing from where the series left off
4. Sets `mappings.series.order` to the next number
5. Updates the previous article's mappings to reference the new one
## Integration with writeonce Architecture
### Store Index
Add a mappings index alongside the existing title, date, and tag indexes:
```
data/
articles.seg
index/
title.idx
date.idx
tags.idx
mappings.idx # sys_title → related sys_titles
```
The mappings index allows efficient traversal: "give me all articles related to X" without scanning every article's JSON.
### Template Rendering
The `article.htmlx` template can render related articles:
```html
<article>
<h1>{{article.title}}</h1>
{{article.content_html}}
{{#each article.related}}
<aside class="related">
<h3>Related</h3>
<ul>
<li><a href="/blog/{{sys_title}}">{{title}}</a></li>
</ul>
</aside>
{{/each}}
</article>
```
### Subscription
When a mapped article changes, subscribers to related articles can optionally be notified. If article A lists article B in `mappings.related`, and article B is updated, subscribers to article A can receive a notification that related content changed.
## Agent Configuration
For agents to use the mappings effectively, the project can include an agent instruction file (e.g., `CLAUDE.md` or `.agent/instructions.md`):
```markdown
## Content Management
- Articles are in `content/{sys_title}/{sys_title}.md`
- Metadata is in `content/{sys_title}/{sys_title}.json`
- Before writing or editing an article, read its `mappings` field and load related articles for context
- When creating a new article, suggest appropriate `mappings` based on tags and content overlap
- Maintain narrative continuity within `series` mappings
- Do not duplicate explanations that exist in `prerequisite` articles — reference them instead
```
This turns the content directory into an agent-navigable knowledge graph where the metadata provides the edges and the markdown files provide the nodes.

338
docs/runtime/async.md Normal file
View file

@ -0,0 +1,338 @@
# Creating an Async Runtime Environment
## Goal
Build a minimal async runtime from scratch using Linux kernel primitives. The runtime is not specific to writeonce — it is a general-purpose event loop that any project can use to multiplex I/O without threads, without tokio, and without any external async framework.
The writeonce project uses this runtime (the `wo-event` and `wo-rt` crates), but the concepts apply to any server, daemon, or event-driven application on Linux.
## What Is a Runtime?
A runtime is the loop that decides **what code runs next**. In a synchronous program, the OS scheduler picks the next thread. In an async runtime, a single thread asks the kernel: "which of my file descriptors are ready?" — and runs the corresponding handler.
```
loop {
ready_fds = ask_kernel_which_fds_are_ready()
for fd in ready_fds {
run_handler(fd)
}
}
```
That's the entire concept. Everything else — epoll, tokens, interest flags — is implementation detail around this loop.
## Explaining It with C
Before Rust, before abstractions, here is a minimal async runtime in C that watches two file descriptors on one thread.
### Step 1: Create an epoll instance
```c
#include <sys/epoll.h>
#include <unistd.h>
#include <stdio.h>
int main() {
// Create the event loop
int epoll_fd = epoll_create1(0);
// This single fd is the "runtime" — all other fds register on it
printf("epoll fd: %d\n", epoll_fd);
}
```
`epoll_create1` returns a file descriptor. This fd *is* the runtime. Every other fd in the system registers itself on this one fd, and the kernel tracks readiness for all of them.
### Step 2: Register file descriptors
```c
#include <sys/epoll.h>
#include <sys/eventfd.h>
#include <sys/timerfd.h>
#include <time.h>
int main() {
int epoll_fd = epoll_create1(0);
// Create an eventfd (like a semaphore as a file descriptor)
int event_fd = eventfd(0, EFD_NONBLOCK);
// Create a timerfd (fires every 2 seconds)
int timer_fd = timerfd_create(CLOCK_MONOTONIC, TFD_NONBLOCK);
struct itimerspec spec = {
.it_interval = { .tv_sec = 2, .tv_nsec = 0 },
.it_value = { .tv_sec = 2, .tv_nsec = 0 }
};
timerfd_settime(timer_fd, 0, &spec, NULL);
// Register both on epoll
struct epoll_event ev1 = { .events = EPOLLIN, .data.fd = event_fd };
epoll_ctl(epoll_fd, EPOLL_CTL_ADD, event_fd, &ev1);
struct epoll_event ev2 = { .events = EPOLLIN, .data.fd = timer_fd };
epoll_ctl(epoll_fd, EPOLL_CTL_ADD, timer_fd, &ev2);
}
```
Two different kinds of fd — an event signal and a timer — both registered on the same epoll instance. The kernel will wake us when either is ready.
### Step 3: The event loop
```c
#include <sys/epoll.h>
#include <sys/eventfd.h>
#include <sys/timerfd.h>
#include <unistd.h>
#include <stdio.h>
#include <stdint.h>
#include <time.h>
int main() {
int epoll_fd = epoll_create1(0);
int event_fd = eventfd(0, EFD_NONBLOCK);
int timer_fd = timerfd_create(CLOCK_MONOTONIC, TFD_NONBLOCK);
struct itimerspec spec = {
.it_interval = { .tv_sec = 2, .tv_nsec = 0 },
.it_value = { .tv_sec = 2, .tv_nsec = 0 }
};
timerfd_settime(timer_fd, 0, &spec, NULL);
struct epoll_event ev1 = { .events = EPOLLIN, .data.fd = event_fd };
epoll_ctl(epoll_fd, EPOLL_CTL_ADD, event_fd, &ev1);
struct epoll_event ev2 = { .events = EPOLLIN, .data.fd = timer_fd };
epoll_ctl(epoll_fd, EPOLL_CTL_ADD, timer_fd, &ev2);
printf("Runtime started. Timer fires every 2s.\n");
printf("Write to eventfd to trigger it: echo 1 > /proc/%d/fd/%d\n",
getpid(), event_fd);
// The event loop — this IS the runtime
struct epoll_event events[10];
while (1) {
int n = epoll_wait(epoll_fd, events, 10, -1); // block until ready
for (int i = 0; i < n; i++) {
int fd = events[i].data.fd;
if (fd == timer_fd) {
uint64_t expirations;
read(timer_fd, &expirations, sizeof(expirations));
printf("[timer] fired (%lu expirations)\n", expirations);
}
else if (fd == event_fd) {
uint64_t val;
read(event_fd, &val, sizeof(val));
printf("[event] signaled (value: %lu)\n", val);
}
}
}
close(epoll_fd);
close(event_fd);
close(timer_fd);
return 0;
}
```
Compile and run:
```bash
gcc -o runtime runtime.c
./runtime
```
Output:
```
Runtime started. Timer fires every 2s.
[timer] fired (1 expirations)
[timer] fired (1 expirations)
[timer] fired (1 expirations)
...
```
One thread. Two fd types. One loop. The kernel does the scheduling.
### Step 4: Add a TCP server to the same loop
```c
#include <sys/epoll.h>
#include <sys/socket.h>
#include <netinet/in.h>
#include <unistd.h>
#include <stdio.h>
#include <fcntl.h>
#include <string.h>
int main() {
int epoll_fd = epoll_create1(0);
// Create a non-blocking TCP listener
int listen_fd = socket(AF_INET, SOCK_STREAM | SOCK_NONBLOCK, 0);
int opt = 1;
setsockopt(listen_fd, SOL_SOCKET, SO_REUSEADDR, &opt, sizeof(opt));
struct sockaddr_in addr = {
.sin_family = AF_INET,
.sin_port = htons(8080),
.sin_addr.s_addr = INADDR_ANY
};
bind(listen_fd, (struct sockaddr*)&addr, sizeof(addr));
listen(listen_fd, 128);
// Register listener on epoll
struct epoll_event ev = { .events = EPOLLIN, .data.fd = listen_fd };
epoll_ctl(epoll_fd, EPOLL_CTL_ADD, listen_fd, &ev);
printf("Listening on port 8080\n");
struct epoll_event events[64];
while (1) {
int n = epoll_wait(epoll_fd, events, 64, -1);
for (int i = 0; i < n; i++) {
int fd = events[i].data.fd;
if (fd == listen_fd) {
// Accept new connection
int client_fd = accept4(listen_fd, NULL, NULL, SOCK_NONBLOCK);
if (client_fd >= 0) {
struct epoll_event cev = { .events = EPOLLIN, .data.fd = client_fd };
epoll_ctl(epoll_fd, EPOLL_CTL_ADD, client_fd, &cev);
printf("[accept] client fd=%d\n", client_fd);
}
} else {
// Read from client
char buf[4096];
int nbytes = read(fd, buf, sizeof(buf));
if (nbytes <= 0) {
// Client disconnected
epoll_ctl(epoll_fd, EPOLL_CTL_DEL, fd, NULL);
close(fd);
printf("[close] fd=%d\n", fd);
} else {
// Echo response
const char *response =
"HTTP/1.1 200 OK\r\n"
"Content-Length: 13\r\n"
"\r\n"
"Hello, world!";
write(fd, response, strlen(response));
epoll_ctl(epoll_fd, EPOLL_CTL_DEL, fd, NULL);
close(fd);
}
}
}
}
}
```
This is a complete HTTP server — no threads, no framework, no library. One `epoll_wait` drives accept, read, write, and close for every connection.
## From C to Rust: The writeonce Runtime
The writeonce `wo-event` crate wraps these same syscalls in safe Rust:
| C syscall | Rust wrapper | Crate |
|-----------|-------------|-------|
| `epoll_create1` | `EventLoop::new()` | wo-event |
| `epoll_ctl(ADD)` | `EventLoop::register(fd, interest, token)` | wo-event |
| `epoll_ctl(MOD)` | `EventLoop::modify(fd, interest, token)` | wo-event |
| `epoll_ctl(DEL)` | `EventLoop::deregister(fd)` | wo-event |
| `epoll_wait` | `EventLoop::poll(timeout)` | wo-event |
| `eventfd` | `EventFd::new()` | wo-event |
| `timerfd_create` | `TimerFd::new()` | wo-event |
| `signalfd` | `SignalFd::new()` | wo-event |
The key difference from the C examples: instead of matching on raw fd numbers, the Rust runtime assigns a **token** (u64) to each fd. The event loop returns tokens, and the runtime dispatches on them:
```rust
let events = event_loop.poll(Some(Duration::from_millis(500)))?;
for event in events {
match event.token {
TOKEN_WATCHER => handle_file_change(),
TOKEN_SIGNAL => handle_shutdown(),
TOKEN_TIMER => handle_periodic_task(),
TOKEN_LISTENER => handle_new_connection(),
token if token >= 10000 => handle_http(token),
_ => {}
}
}
```
## Why Not tokio?
tokio is a production-grade async runtime. It handles everything — epoll, thread pools, work stealing, timers, I/O drivers. So why build a custom one?
| Concern | tokio | Custom runtime |
|---------|-------|---------------|
| Binary size | Adds ~2-3 MB | Zero — just libc syscalls |
| Dependencies | 50+ transitive crates | 1 crate (libc) |
| Complexity | Work-stealing scheduler, multi-threaded executor | Single-threaded loop, ~200 lines |
| Control | Opaque — runtime internals hidden behind `.await` | Every fd, every syscall, every state transition is explicit |
| Learning | Abstracts the kernel away | Forces understanding of what the kernel actually does |
For a content platform serving markdown files, the workload is: accept connection, read request, query in-memory index, render template, write response. This is microseconds of work per request. A single-threaded event loop handles thousands of concurrent connections without the complexity of a multi-threaded executor.
## Reusing the Runtime in Other Projects
The `wo-event` crate has no dependency on writeonce. It provides:
- `EventLoop` — epoll wrapper with register/deregister/poll
- `EventFd` — lightweight signaling
- `TimerFd` — periodic timers as fds
- `SignalFd` — SIGINT/SIGTERM as fd events
- `Event` — readable/writable/hangup status
- `Token` — u64 identifier for dispatch
Any project that needs non-blocking I/O on Linux can use it:
```rust
use wo_event::{EventLoop, EventFd, Interest};
use std::time::Duration;
fn main() {
let eloop = EventLoop::new().unwrap();
let efd = EventFd::new().unwrap();
eloop.register(efd.fd(), Interest::Readable, 1).unwrap();
// Signal from another thread
std::thread::spawn(move || {
std::thread::sleep(Duration::from_secs(1));
efd.write(42).unwrap();
});
// Wait for the signal
let events = eloop.poll(Some(Duration::from_secs(5))).unwrap();
assert_eq!(events[0].token, 1);
println!("Event received!");
}
```
## The Mental Model
```
┌─────────────────────────────────┐
│ your code │
│ (handlers, business logic) │
└──────────────┬──────────────────┘
│ dispatches on token
┌──────────────┴──────────────────┐
│ event loop │
│ epoll_wait → Vec<Event> │
└──────────────┬──────────────────┘
│ registered fds
┌──────────────┴──────────────────┐
│ Linux kernel │
│ tracks readiness for all fds │
│ inotify, sockets, timers, │
│ signals, eventfds — all fds │
└─────────────────────────────────┘
```
The kernel is the scheduler. The event loop is the dispatcher. Your code is the handler. Everything in the system — files, sockets, timers, signals — is a file descriptor. One loop to rule them all.

253
docs/runtime/fibers.md Normal file
View file

@ -0,0 +1,253 @@
# Runtime Fibers
Runtime fibers are lightweight, user-space threads managed by an application's runtime system rather than the OS kernel. They enable massive concurrency — millions per machine — because they don't carry the overhead of kernel thread stacks and scheduling. Unlike pre-emptive kernel threads, fibers use cooperative multitasking: they yield control voluntarily at known suspension points.
## Context Switching
A context switch is saving the state of one execution unit and restoring another so it can continue running. The cost of this switch is what separates kernel threads from fibers.
### Kernel Thread Context Switch
When the OS switches between threads:
1. Save all CPU registers (general purpose, floating point, SIMD) to kernel memory
2. Save the thread's stack pointer
3. Flush the TLB (translation lookaside buffer) if switching processes
4. Update scheduler data structures
5. Restore the next thread's registers and stack pointer
6. Return to userspace
Cost: **1-10 microseconds**, involves a kernel trap (syscall boundary crossing), cache pollution from TLB flush.
### Fiber Context Switch
When a runtime switches between fibers:
1. Save a few registers (stack pointer, instruction pointer, callee-saved registers)
2. Swap the stack pointer to the next fiber's stack
3. Jump to the next fiber's saved instruction pointer
Cost: **~10-100 nanoseconds**, entirely in userspace, no kernel involvement, no TLB flush, cache stays warm.
```
Kernel thread switch: ~1,000-10,000 ns (kernel trap + TLB flush)
Fiber switch: ~10-100 ns (register swap in userspace)
100x cheaper
```
## Types of Multitasking
### Pre-emptive (Kernel Threads)
The OS scheduler interrupts threads at arbitrary points using timer interrupts. The thread does not choose when to yield — the kernel forces it.
```
Thread A: ████████──┐ (interrupted by OS)
│
Thread B: └──████████──┐ (interrupted by OS)
│
Thread A: └──████████
```
- Threads can be interrupted mid-instruction
- Requires locks/mutexes to protect shared state
- Fairness guaranteed by the scheduler
- Used by: pthreads, std::thread, OS processes
### Cooperative (Fibers / Green Threads)
Fibers explicitly yield at known points (I/O boundaries, channel sends, `.await` in Rust). The runtime only switches when the fiber says "I'm done for now."
```
Fiber A: ████████ yield ──┐
│
Fiber B: └── ████████ yield ──┐
│
Fiber A: └── ████████
```
- Fibers are never interrupted mid-computation
- No locks needed for single-threaded runtimes — only one fiber runs at a time
- Starvation possible if a fiber never yields (compute-heavy work blocks the loop)
- Used by: Go goroutines, Erlang processes, Lua coroutines, Rust async/await
### Comparison
| Property | Pre-emptive (Threads) | Cooperative (Fibers) |
|----------|----------------------|---------------------|
| Scheduling | OS kernel decides | Runtime decides at yield points |
| Context switch cost | ~1-10 us | ~10-100 ns |
| Stack size | 1-8 MB per thread (fixed) | Bytes to KB per fiber (growable) |
| Max concurrency | ~10,000 threads | ~1,000,000+ fibers |
| Synchronization | Locks, mutexes, atomics | Not needed in single-threaded runtime |
| Interruption | Any point (timer interrupt) | Only at yield points |
| Fairness | Guaranteed by scheduler | Must be designed (fiber must yield) |
## How Fibers Work Internally
A fiber needs three things:
1. **A stack** — a block of memory for local variables and call frames
2. **A saved context** — the register state at the point it yielded
3. **A function** — the code to run when resumed
### Minimal Fiber in C
```c
#include <ucontext.h>
#include <stdio.h>
static ucontext_t main_ctx, fiber_ctx;
static char fiber_stack[8192];
void fiber_fn() {
printf("Fiber: running\n");
// Yield back to main
swapcontext(&fiber_ctx, &main_ctx);
printf("Fiber: resumed\n");
// Yield again
swapcontext(&fiber_ctx, &main_ctx);
}
int main() {
// Set up fiber context
getcontext(&fiber_ctx);
fiber_ctx.uc_stack.ss_sp = fiber_stack;
fiber_ctx.uc_stack.ss_size = sizeof(fiber_stack);
fiber_ctx.uc_link = &main_ctx;
makecontext(&fiber_ctx, fiber_fn, 0);
printf("Main: starting fiber\n");
swapcontext(&main_ctx, &fiber_ctx); // switch to fiber
printf("Main: fiber yielded\n");
swapcontext(&main_ctx, &fiber_ctx); // resume fiber
printf("Main: fiber yielded again\n");
swapcontext(&main_ctx, &fiber_ctx); // resume — fiber finishes
printf("Main: done\n");
return 0;
}
```
Output:
```
Main: starting fiber
Fiber: running
Main: fiber yielded
Fiber: resumed
Main: fiber yielded again
Main: done
```
`swapcontext` saves the current register state to one context struct and loads another — that's the entire fiber switch. No kernel involved.
### Minimal Fiber in Rust (unsafe)
Rust's async/await compiles to state machines, not stack-swapping fibers. But you can build raw fibers with inline assembly:
```rust
use std::arch::asm;
struct Fiber {
stack: Vec<u8>,
sp: *mut u8, // saved stack pointer
}
impl Fiber {
fn new(func: fn()) -> Self {
let mut stack = vec![0u8; 8192];
let sp = unsafe {
let top = stack.as_mut_ptr().add(stack.len());
let aligned = (top as usize & !0xF) as *mut u8;
// Push the function pointer as the return address
let sp = aligned.sub(8);
*(sp as *mut fn()) = func;
sp
};
Fiber { stack, sp }
}
unsafe fn switch_to(&mut self, from: &mut *mut u8) {
// Save callee-saved registers and swap stack pointers
asm!(
"push rbx",
"push rbp",
"push r12",
"push r13",
"push r14",
"push r15",
"mov [{from}], rsp", // save current sp
"mov rsp, [{to}]", // load fiber sp
"pop r15",
"pop r14",
"pop r13",
"pop r12",
"pop rbp",
"pop rbx",
from = in(reg) from,
to = in(reg) &self.sp,
);
}
}
```
This is what runtimes like Go and Erlang do internally — allocate a small stack, save/restore a handful of registers, and jump. The cost is a few nanoseconds.
## Fibers vs Rust async/await
Rust chose a different approach than fibers for its async model:
| Property | Fibers (Go, Erlang) | Rust async/await |
|----------|-------------------|-----------------|
| Implementation | Stack swapping at runtime | Compiler generates state machines |
| Stack | Each fiber has its own stack | No extra stack — state stored in Future struct |
| Memory per task | ~2-8 KB minimum (stack) | Bytes — only the live variables at yield points |
| Yield mechanism | `swapcontext` / assembly | `.await` compiles to `Poll::Pending` |
| Overhead | Stack allocation + register swap | Zero-cost — state machine is a regular struct |
| Debuggability | Separate stacks in debugger | State machine is harder to trace |
| Preemption | Runtime can preempt (Go does this) | Never preempted — cooperative only |
Rust's approach is called **stackless coroutines** — no extra stack per task. The compiler transforms each `async fn` into a state machine enum where each variant holds the local variables alive across an `.await` point.
## Relation to writeonce
The writeonce runtime (`wo-event`, `wo-rt`) uses neither fibers nor Rust async/await. It uses a **single-threaded event loop with callbacks** — the simplest model:
```
loop {
events = epoll_wait()
for event in events {
match event.token {
WATCHER => handle_file_change(),
LISTENER => handle_accept(),
HTTP_CONN => handle_request(),
...
}
}
}
```
This is the same model as nginx, Redis, and Node.js (before libuv's thread pool). It works because:
- Each handler runs to completion quickly (microseconds)
- No handler blocks — all I/O is non-blocking
- No concurrent access to shared state — one thing runs at a time
- Blocking work (index rebuild) offloads to a thread pool and signals back via eventfd
If writeonce ever needed millions of concurrent long-lived tasks (not just connections), fibers would be the next step. But for a content platform serving articles, the event loop is sufficient — and far simpler to reason about.
## Summary
```
Kernel threads: OS-managed, pre-emptive, expensive switch, ~10K max
Fibers: Runtime-managed, cooperative, cheap switch, ~1M+ max
Async/await: Compiler-managed, cooperative, zero-cost, ~1M+ max
Event loop: No tasks at all — just fd readiness + callbacks
Complexity: event loop < fibers < async/await < threads
Concurrency: event loop = fibers = async/await >> threads
```
The right choice depends on the workload. For writeonce — an event loop. For a database with millions of queries in flight — fibers or async. For CPU-bound parallel work — kernel threads.

View file

@ -0,0 +1,248 @@
# Why Rust Does Not Use Fibers or Garbage Collection
## The Question
Most runtime-heavy languages ship with two things: a garbage collector (Go, Java, Python, C#, Erlang) and fiber-like concurrency (Go goroutines, Erlang processes, Java virtual threads). Rust ships with neither. Why?
The answer is the same for both: **Rust pushes the cost to compile time so there is zero cost at runtime.**
## Garbage Collection
### What a GC Does
A garbage collector tracks which objects in memory are still reachable from the program. Periodically (or continuously), it scans the heap, finds objects nothing points to, and frees them.
```
Allocate object A
Allocate object B
A.ref = B // B is reachable through A
drop(A) // A is unreachable — GC will free A
// B is now also unreachable — GC will free B
```
### How It Works (Simplified)
**Mark-and-sweep** (Go, Java):
```
1. Pause the program (or run concurrently)
2. Start from "roots" (stack variables, globals)
3. Mark every object reachable from roots
4. Sweep: free every object NOT marked
```
**Reference counting** (Python, Swift, Objective-C):
```
1. Every object has a counter
2. When a reference is created: counter++
3. When a reference is dropped: counter--
4. When counter == 0: free immediately
```
### The Costs
| Cost | Mark-and-sweep GC | Reference counting |
|------|-------------------|-------------------|
| Pause time | Stop-the-world pauses (Go: ~1ms, Java: varies) | No pauses, but slower per-operation |
| Memory overhead | 2x heap needed (live objects + garbage until collected) | Counter per object (8 bytes) |
| CPU overhead | GC thread scanning heap (10-30% throughput loss) | Increment/decrement on every pointer operation |
| Predictability | Unpredictable latency spikes | Predictable but cycles leak (need cycle collector) |
| Cache impact | GC walks heap → cache pollution | Counters spread across memory → cache misses |
For a content platform that needs predictable low-latency responses, GC pauses are the enemy. Even Go's ~1ms pauses compound under load — if a GC pause hits during `epoll_wait`, every pending connection stalls.
### What Rust Does Instead: Ownership
Rust replaces garbage collection with a compile-time ownership system:
```rust
fn main() {
let s = String::from("hello"); // s owns the string, allocated on heap
let t = s; // ownership moves to t — s is invalid
// println!("{}", s); // compile error: s was moved
println!("{}", t); // ok
} // t goes out of scope → String freed here. Deterministic. No GC.
```
The rules:
1. **Every value has exactly one owner**
2. **When the owner goes out of scope, the value is dropped (freed)**
3. **Ownership can be moved or borrowed, but never duplicated**
The compiler enforces these rules at compile time. At runtime, there is:
- No GC thread
- No mark phase
- No sweep phase
- No reference counters
- No heap scanning
- No pauses
Memory is freed at the exact point it is no longer needed — deterministically, at the closing brace.
### Lifetimes: The Compile-Time GC
References (borrows) have lifetimes — the compiler tracks how long each reference lives and ensures no reference outlives its data:
```rust
fn longest<'a>(x: &'a str, y: &'a str) -> &'a str {
if x.len() > y.len() { x } else { y }
}
```
The `'a` lifetime annotation tells the compiler: "the returned reference lives as long as both inputs." If you try to return a reference to a local variable, the compiler rejects it — at compile time, not at runtime.
```rust
fn bad() -> &str {
let s = String::from("hello");
&s // compile error: s is dropped at end of function, reference would dangle
}
```
This is what a GC does at runtime (detect unreachable memory). Rust does it at compile time (detect impossible references). Zero runtime cost.
### The Tradeoff
| Aspect | GC languages | Rust |
|--------|-------------|------|
| Developer effort | Low — just allocate, GC handles cleanup | Higher — must think about ownership and lifetimes |
| Compile time | Fast | Slower (borrow checker analysis) |
| Runtime cost | GC pauses, heap scanning, memory overhead | Zero — deterministic drop at scope exit |
| Latency | Unpredictable (GC can pause anytime) | Predictable — no hidden pauses |
| Memory usage | 2x+ (garbage accumulates between collections) | Tight — freed immediately when unused |
Rust trades developer convenience for runtime performance. For systems software (databases, runtimes, web servers), this is the right trade.
## Fibers
### Why Other Languages Use Fibers
Go has goroutines. Erlang has processes. Java 21 has virtual threads. These are all fibers — lightweight user-space threads that the runtime schedules cooperatively (or semi-preemptively in Go's case).
They exist because these languages need to:
1. Handle millions of concurrent I/O tasks
2. Let developers write synchronous-looking code (`result = fetch(url)`) that blocks the fiber, not the OS thread
3. Manage scheduling without exposing the event loop
```go
// Go: goroutine blocks on I/O — runtime suspends it and runs another
go func() {
resp, _ := http.Get("https://example.com") // blocks this goroutine, not the thread
fmt.Println(resp.Status)
}()
```
### Why Rust Does Not Use Fibers
**1. Fibers require a runtime that allocates stacks.**
Each fiber needs its own stack (Go starts at 2-8 KB, grows dynamically). This means:
- A heap allocation per fiber
- Stack overflow checks on every function call
- A runtime that manages stack growth and shrinking
- Memory overhead proportional to number of concurrent tasks
Rust's goal is zero-cost abstractions. Allocating stacks at runtime is a cost.
**2. Fibers are hard to optimize across FFI boundaries.**
Rust interoperates with C libraries extensively. Fibers with tiny stacks can't safely call into C code (which expects a full OS stack). Go solves this by switching to a system stack for cgo calls — adding complexity and overhead.
**3. Async/await achieves the same concurrency without stacks.**
Rust's async/await compiles each async function into a state machine — a regular struct stored inline, no heap allocation needed:
```rust
async fn fetch_article(title: &str) -> Article {
let data = read_from_seg(title).await; // suspend point 1
let html = render_markdown(&data).await; // suspend point 2
Article { title, html }
}
```
The compiler transforms this into something like:
```rust
enum FetchArticle {
Start { title: String },
AfterRead { title: String, data: Vec<u8> },
AfterRender { title: String, html: String },
Done,
}
```
Each `.await` becomes a variant transition. The "stack" is just the live variables in the current variant — bytes, not kilobytes. No allocation, no stack, no runtime overhead.
### Fiber vs Async/Await: Memory Per Task
```
Go goroutine: ~2,048 bytes minimum (stack)
Erlang process: ~2,688 bytes minimum (stack + heap + mailbox)
Rust async task: size_of::<FetchArticle>() — often 32-128 bytes
```
For a million concurrent connections:
- Go: ~2 GB just for goroutine stacks
- Rust: ~128 MB for state machines (and often less, since the executor batches them)
### When Fibers Would Be Better
Fibers have one advantage: **deeply nested call stacks that suspend at arbitrary points.** If a function 20 calls deep needs to yield, a fiber just swaps the stack pointer. With async/await, every function in the chain must be `async` and every call must be `.await`ed — the "async infection" problem.
```
Fibers: yield anywhere in the call stack — transparent to callers
Async: yield only at .await points — every caller must be async
```
For database engines with complex query execution plans that suspend mid-evaluation, fibers are compelling. For an HTTP server that suspends at I/O boundaries, async/await is strictly better.
## How This Applies to writeonce
writeonce uses neither fibers nor async/await. It uses a plain event loop:
```rust
loop {
events = epoll_wait();
for event in events {
handle(event); // runs to completion, no suspension
}
}
```
This is the simplest model — no GC, no fibers, no async state machines. Each handler reads from the `.seg` file, renders a template, writes to the socket, and returns. Nothing suspends mid-handler.
The memory model:
| What | How it's managed |
|------|-----------------|
| Article data in .seg | Owned by `Store`, freed when `Store` drops |
| Template ASTs | Owned by `TemplateRegistry`, live for the process lifetime |
| HTTP connections | Owned by `HashMap<Token, Connection>`, freed on close/hangup |
| Subscription table | Owned by `SubscriptionManager`, entries removed on `EPOLLHUP` |
No garbage. No fibers. No async. Just ownership, scopes, and the kernel's event notification. The Rust compiler guarantees at compile time that every allocation is freed exactly once, at exactly the right time.
## Summary
```
GC languages (Go, Java): runtime scans heap → frees unreachable objects
cost: pauses, memory overhead, CPU overhead
Reference counting (Python): counter per object → free at zero
cost: per-operation overhead, cycle leaks
Rust ownership: compiler tracks ownership → free at scope exit
cost: zero at runtime, developer thinks harder
Fibers (Go, Erlang): runtime manages stacks → swap on yield
cost: stack allocation, stack checks, runtime
Async/await (Rust): compiler generates state machines → no stack
cost: zero allocation, async must propagate
Event loop (writeonce): no tasks, no suspension → handlers run to completion
cost: nothing — simplest possible model
```
Rust's answer to both GC and fibers is the same: **make the compiler do the work so the runtime doesn't have to.**

View file

@ -0,0 +1,157 @@
# SurrealDB — Runtime Case Study
Reference repository: [github.com/surrealdb/surrealdb](https://github.com/surrealdb/surrealdb)
```bash
git submodule add https://github.com/surrealdb/surrealdb.git references/surrealdb
```
## The Question
Does SurrealDB only rely on async/await for concurrency?
**No.** SurrealDB uses a layered concurrency model — async/await is one layer, but it also uses OS thread pools, CPU-affinity-pinned workers, lock-free data structures, and parallel computation frameworks. Each layer serves a different purpose.
## Architecture Overview
SurrealDB is a single Rust binary that ships a multi-model database (documents, graphs, key-value) with real-time live queries. It supports multiple deployment modes:
- **Server**: `surreal start` runs HTTP/WebSocket API via Axum + storage engine
- **Embedded**: the library crate embeds directly in Rust applications
- **WASM**: runs in the browser with IndexedDB backend
## Concurrency Layers
### Layer 1: Tokio — Async I/O and Request Handling
The primary runtime. Handles:
- HTTP/WebSocket connections (via Axum)
- Network I/O (accept, read, write)
- Timer-based operations
- Task scheduling (M:N scheduling of futures onto OS threads)
```
Client request → Axum handler (async) → parse query → execute → respond
```
Every request handler is an async function. Tokio's multi-threaded executor distributes tasks across OS threads using work-stealing.
### Layer 2: Rayon — Parallel CPU-Bound Computation
For operations that are compute-heavy, not I/O-bound:
- Query plan execution across partitions
- Data processing and transformation
- Parallel iteration over result sets
Rayon provides `par_iter()` — automatic parallelism across CPU cores. It has its own thread pool, separate from tokio's.
### Layer 3: affinitypool — CPU-Pinned Storage I/O
SurrealDB's custom crate. Runs blocking storage operations on a dedicated thread pool where **each thread is pinned to a specific CPU core** via `libc` CPU affinity syscalls.
Used by:
- RocksDB backend (blocking disk I/O)
- SurrealKV embedded storage
- In-memory engine for heavy operations
This bridges the async world (tokio) and the blocking world (disk I/O) without polluting the tokio thread pool with blocking calls.
### Layer 4: Lock-Free Data Structures
The hot path in storage engines uses concurrent data structures that avoid locks entirely:
| Crate | Data Structure | Used For |
|-------|---------------|----------|
| `crossbeam-skiplist` | Concurrent skip list | Index structures in SurrealKV and surrealmx |
| `crossbeam-deque` | Work-stealing deque | Task distribution |
| `crossbeam-queue` | Lock-free queue | Message passing |
| `papaya` | Concurrent HashMap | In-memory engine (surrealmx) |
| `dashmap` | Sharded concurrent map | Pub/sub routing for live queries |
| `arc-swap` | Atomic pointer swap | Hot-swapping data structures without locks |
| `parking_lot` | Fast mutex/rwlock | Where locking is needed (faster than std) |
## No Fibers
SurrealDB does **not** use fibers, green threads, or any custom scheduling mechanism. The concurrency model is:
```
Tokio tasks (async/await) — for I/O-bound work
Rayon threads (par_iter) — for CPU-bound work
affinitypool threads (pinned) — for blocking storage I/O
Lock-free structures — for concurrent data access
```
This is pragmatic — each concurrency mechanism is used where it fits, rather than forcing everything through one model.
## Live Queries / Real-Time Subscriptions
SurrealDB's live query system pushes changes to connected clients in real-time:
1. Client registers a live query via WebSocket: `LIVE SELECT * FROM person WHERE age > 21`
2. Server tracks the query in a `dashmap` (concurrent map)
3. When a transaction commits changes to `person`, the engine evaluates which live queries are affected
4. Matching subscribers receive the diff via their WebSocket connection
5. Transport: `tokio-tungstenite` for WebSocket, `async-channel` for internal pub/sub routing
### Comparison with writeonce Subscriptions
| Aspect | SurrealDB | writeonce |
|--------|-----------|-----------|
| Transport | WebSocket (tokio-tungstenite) | Raw socket fd (kernel-level write) |
| Query registration | SQL-like live query over WebSocket | `register!` macro binding fd to content pattern |
| Change detection | Transaction commit triggers evaluation | inotify detects file change |
| Notification routing | `dashmap` + `async-channel` | `SubscriptionManager` HashMap + direct `write(fd)` |
| Runtime | Tokio multi-threaded executor | Single-threaded epoll event loop |
| Protocol framing | WebSocket frames | Length-prefixed payloads (no protocol) |
SurrealDB's live queries are the architectural inspiration for writeonce's subscription model (as noted in [03-data.md](../03-data.md)), but the implementation is fundamentally different — SurrealDB uses a full async runtime with WebSocket transport, while writeonce uses kernel fd notifications with no protocol layer.
## Storage Engine Architecture
SurrealDB supports 5 backends:
| Backend | Type | Concurrency |
|---------|------|------------|
| **surrealmx** | In-memory | Lock-free (papaya, crossbeam-skiplist, arc-swap) |
| **surrealkv** | Embedded persistent | Tokio async + crossbeam + parking_lot |
| **RocksDB** | Embedded persistent | affinitypool (CPU-pinned blocking threads) |
| **TiKV** | Distributed | Async TiKV client over gRPC |
| **IndxDB** | Browser/WASM | IndexedDB via wasm-bindgen-futures |
### Comparison with writeonce Storage
| Aspect | SurrealDB | writeonce |
|--------|-----------|-----------|
| Storage format | Key-value entries in LSM trees (RocksDB) or custom B-trees (SurrealKV) | `.seg` files with length-prefixed bincode records |
| Index | Built into storage engine | Separate `.idx` files (title hash, date sorted, tags inverted) |
| Concurrency | Multi-threaded with locks/lock-free structures | Single-threaded, positional I/O (pread/pwrite) |
| Transaction | ACID with MVCC | Full rebuild on change (article count is small) |
| Complexity | ~100K+ lines across storage crates | ~300 lines (wo-seg + wo-index) |
writeonce's storage is intentionally simple — the dataset is small (hundreds of articles, not millions of rows), so a full rebuild on change is fast enough and avoids the complexity of concurrent transactions.
## Key Takeaways
1. **Async/await alone is not enough for a database.** SurrealDB uses four concurrency mechanisms, each for a different workload profile.
2. **Blocking I/O needs its own thread pool.** The affinitypool pattern — CPU-pinned threads for storage operations — keeps blocking work off the async executor. writeonce avoids this entirely by using `pread` (non-blocking positional reads) in a single-threaded loop.
3. **Lock-free data structures matter at scale.** SurrealDB's hot path avoids mutexes. writeonce doesn't need this — single-threaded access means no contention.
4. **Live queries are the hard problem.** Both SurrealDB and writeonce solve "push changes to subscribers," but at vastly different scales. SurrealDB handles arbitrary SQL predicates over millions of rows. writeonce handles content queries over hundreds of articles.
5. **The right amount of complexity depends on the problem.** SurrealDB is a general-purpose database — it needs the complexity. writeonce is a content platform — the event loop model is sufficient and far simpler.
## Reference
Add SurrealDB as a submodule for code reference:
```bash
git submodule add https://github.com/surrealdb/surrealdb.git references/surrealdb
```
Key files to study:
- `crates/core/src/kvs/` — storage engine abstraction and transaction handling
- `crates/sdk/src/api/engine/` — live query subscription routing
- `lib/affinitypool/` — CPU-pinned thread pool for blocking I/O
- `crates/core/src/sql/` — query parser and execution engine

34
infra/deploy.sh Executable file
View file

@ -0,0 +1,34 @@
#!/bin/bash
# deploy.sh — run from the development machine to build, copy, and restart
set -e
SERVER="writeonce.de"
REMOTE_DIR="/opt/writeonce"
echo "Building release binary..."
cargo build --release -p wo-rt --bin writeonce
echo "Copying binary to server..."
scp target/release/writeonce "$SERVER:$REMOTE_DIR/writeonce.new"
echo "Syncing content..."
rsync -az --delete content/ "$SERVER:$REMOTE_DIR/content/"
echo "Syncing templates..."
rsync -az --delete templates/ "$SERVER:$REMOTE_DIR/templates/"
echo "Syncing static assets..."
rsync -az --delete static/ "$SERVER:$REMOTE_DIR/static/"
echo "Swapping binary and restarting service..."
ssh "$SERVER" "
sudo mv $REMOTE_DIR/writeonce.new $REMOTE_DIR/writeonce
sudo chmod +x $REMOTE_DIR/writeonce
sudo chown writeonce:writeonce $REMOTE_DIR/writeonce
sudo systemctl restart writeonce
"
echo "Checking status..."
ssh "$SERVER" "sudo systemctl status writeonce --no-pager"
echo "Deploy complete."

36
infra/setup.sh Executable file
View file

@ -0,0 +1,36 @@
#!/bin/bash
# setup.sh — run once on the server to create user, directory, service, nginx, and SSL
set -e
echo "Creating writeonce user..."
sudo useradd -r -s /bin/false writeonce 2>/dev/null || echo "User already exists"
echo "Creating directory structure..."
sudo mkdir -p /opt/writeonce/{content,data,templates,static}
sudo chown -R writeonce:writeonce /opt/writeonce
echo "Installing systemd service..."
sudo cp writeonce.service /etc/systemd/system/
sudo systemctl daemon-reload
sudo systemctl enable writeonce
echo "Configuring nginx..."
sudo cp writeonce.de.nginx /etc/nginx/sites-available/writeonce.de
sudo ln -sf /etc/nginx/sites-available/writeonce.de /etc/nginx/sites-enabled/
sudo nginx -t
sudo systemctl reload nginx
echo "Setting up SSL with Let's Encrypt..."
sudo apt install -y certbot python3-certbot-nginx
sudo certbot --nginx -d writeonce.de -d www.writeonce.de
echo "Installing certbot reload hook..."
sudo mkdir -p /etc/letsencrypt/renewal-hooks/deploy
sudo tee /etc/letsencrypt/renewal-hooks/deploy/reload-nginx.sh > /dev/null << 'HOOK'
#!/bin/bash
systemctl reload nginx
HOOK
sudo chmod +x /etc/letsencrypt/renewal-hooks/deploy/reload-nginx.sh
echo "Setup complete."
echo "Run deploy.sh from your development machine to deploy the binary."

40
infra/writeonce.de.nginx Normal file
View file

@ -0,0 +1,40 @@
server {
listen 80;
server_name writeonce.de www.writeonce.de;
return 301 https://$server_name$request_uri;
}
server {
listen 443 ssl http2;
server_name writeonce.de www.writeonce.de;
ssl_certificate /etc/letsencrypt/live/writeonce.de/fullchain.pem;
ssl_certificate_key /etc/letsencrypt/live/writeonce.de/privkey.pem;
ssl_protocols TLSv1.2 TLSv1.3;
ssl_ciphers HIGH:!aNULL:!MD5;
ssl_prefer_server_ciphers on;
# HSTS
add_header Strict-Transport-Security "max-age=31536000; includeSubDomains" always;
location / {
proxy_pass http://127.0.0.1:3000;
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $scheme;
# Keep connections open for database subscriptions
proxy_http_version 1.1;
proxy_set_header Connection "";
proxy_read_timeout 86400s;
proxy_send_timeout 86400s;
}
# Static assets — let nginx serve directly for better caching
location /static/ {
alias /opt/writeonce/static/;
expires 1y;
add_header Cache-Control "public, immutable";
}
}

24
infra/writeonce.service Normal file
View file

@ -0,0 +1,24 @@
[Unit]
Description=writeonce content platform
After=network.target
[Service]
Type=simple
User=writeonce
Group=writeonce
WorkingDirectory=/opt/writeonce
ExecStart=/opt/writeonce/writeonce
Restart=on-failure
RestartSec=5
StandardOutput=journal
StandardError=journal
# Security hardening
NoNewPrivileges=true
ProtectSystem=strict
ProtectHome=true
ReadWritePaths=/opt/writeonce/data
PrivateTmp=true
[Install]
WantedBy=multi-user.target

View file

@ -0,0 +1,38 @@
# Ignore target directory (where build artifacts are stored)
target/
# Ignore Cargo.lock if you want to use the latest dependencies
# Uncomment the line below if you want to ignore it
# Cargo.lock
# Ignore build files
**/*.rs.bk
**/*.rlib
# Ignore any local configuration files
*.env
*.local
# Ignore any files that are not part of the Rust project
*.md
*.txt
# Ignore version control directories
.git/
.gitignore
# Ignore IDE/editor configuration files
.vscode/
.idea/
*.sublime-workspace
*.sublime-project
*.swp
*.DS_Store
# Ignore any backup files
*~
# Diesel CLI generated schema.rs
diesel.toml
**/migrations/**/diesel_schema.lock

View file

@ -0,0 +1,4 @@
PORT=
DATABASE_URL=
API_ACCESS_TOKEN=
AWS_INFRA_BASE_URL=

50
reference/writeonce-api/.gitignore vendored Normal file
View file

@ -0,0 +1,50 @@
# Ignore target directory (generated build files)
target/
# Ignore intermediate files from rustc compiler
*.rs.bk
# Cargo-specific files
Cargo.lock
/.cargo/
# Output files and logs from Cargo
**/*.log
# Ignore binaries and libraries (compiled executables or libraries)
*.exe
*.exe~
*.out
*.o
*.a
*.so
*.dylib
*.dll
# Ignore dependency build artifacts
/.rustup/
/.git/target/
/.git/target/**
# macOS specific files
.DS_Store
# Windows specific files
Thumbs.db
ehthumbs.db
desktop.ini
$RECYCLE.BIN/
# Linux specific files
*~
# IDE/editor specific files (optional)
/.idea/
/.vscode/
/*.iml
# Other tooling-specific files (optional)
/node_modules/
.env
!.env.sample

View file

@ -0,0 +1,44 @@
stages:
- build
- push
- deploy
build:
stage: build
image: docker:20.10.7
services:
- docker:20.10.7-dind
before_script:
- echo "Global setup before script"
- chmod +x ./infra/00-build-image.sh
script:
- ./infra/00-build-image.sh
push:
stage: push
image: docker:20.10.7
services:
- docker:20.10.7-dind
before_script:
- chmod +x ./infra/02-docker-login.sh
- chmod +x ./infra/03-upload-image.sh
script:
- infra/02-docker-login.sh
- infra/03-upload-image.sh
only:
- main
deploy:
stage: deploy
image: docker:20.10.7
services:
- docker:20.10.7-dind
before_script:
- chmod +x ./infra/04-ssh-setup.sh
- chmod +x ./infra/05-deploy.sh
script:
- echo "Docker set docker auth config"
- infra/04-ssh-setup.sh
- infra/05-deploy.sh
only:
- main

View file

@ -0,0 +1,51 @@
#[target.aarch64-unknown-linux-gnu]
#image = "ghcr.io/cross-rs/aarch64-unknown-linux-gnu:main"
#dockerfile = "Dockerfile.cross"
#linker = "aarch64-linux-gnu-gcc"
[env]
#OPENSSL_STATIC = "1"
#OPENSSL_DIR = "/usr/aarch64-linux-gnu/"
[package]
name = "writeonce-manage-article-api"
version = "0.1.0"
edition = "2021"
[[bin]]
name = "writeonce-manage-article-api"
path = "src/main.rs"
[[bin]]
name = "lambda-post-article"
path = "src/lambda/post_article.rs"
[dependencies]
actix-web = "4.0"
warp = "0.3"
serde = { version = "1.0", features = ["derive"] }
serde_json = "1.0"
dotenv = "0.15"
reqwest = { version = "0.11", default-features = false, features = ["json","rustls-tls"] }
url = "2"
chrono = "0.4"
actix-files = "0.6"
############ Diesel #################
diesel = { version = "2.2.0", default-features = false, features = ["postgres","serde_json","r2d2"] }
diesel_migrations = "2.2.0"
r2d2 = "0.8"
thiserror = "1.0"
lazy_static = "1.4"
#openssl = { version = "0.10", features = ["vendored"] }
##### Lambda ##########
tokio = "1.43.0"
lambda_runtime = "0.13.0"
#### AWS ####
aws-config = { version = "1", features = ["behavior-version-latest"] }
aws-sdk-s3 = "1"
aws-sdk-ec2 = "1"
aws-smithy-types = "1"

View file

@ -0,0 +1,5 @@
[build]
default-target = "aarch64-unknown-linux-gnu"
[target.aarch64-unknown-linux-gnu]
image = "cross-aarch64:latest"

View file

@ -0,0 +1,66 @@
# Declare ARG in global scope
ARG VERSION=1.82.0
FROM rust:${VERSION} AS base
WORKDIR /usr/src/app
FROM base as build
# Install build dependencies
RUN apt-get update && apt-get install -y \
curl \
build-essential \
pkg-config \
libssl-dev \
&& rm -rf /var/lib/apt/lists/*
# Install and configure sccache to speed up compilation
RUN cargo install sccache
ENV RUSTC_WRAPPER="sccache"
# Cache dependencies to speed up builds
COPY Cargo.toml ./
RUN mkdir src && echo "fn main() {}" > src/main.rs
RUN cargo build --release && rm -rf target/release/deps/*
COPY . .
RUN cargo build --release --bin writeonce-manage-article-api
# Stage 2: Create a minimal runtime image
FROM debian:bookworm-slim as runtime
# Install necessary runtime dependencies
RUN apt-get update && apt-get install -y \
ca-certificates \
libc6 \
libpq5 \
&& rm -rf /var/lib/apt/lists/*
COPY --from=build /usr/src/app/target/release/writeonce-manage-article-api /usr/local/bin/app
COPY doc /usr/local/share/doc
ARG PORT
ARG AWS_INFRA_BASE_URL
ARG DATABASE_URL
ARG API_ACCESS_TOKEN
ARG API_ACCESS_ADMIN_TOKEN
RUN echo $PORT
ENV PORT=$PORT
ENV AWS_INFRA_BASE_URL=$AWS_INFRA_BASE_URL
ENV API_ACCESS_TOKEN=$API_ACCESS_TOKEN
ENV API_ACCESS_ADMIN_TOKEN=$API_ACCESS_ADMIN_TOKEN
ENV DATABASE_URL=$DATABASE_URL
# Expose the application port
EXPOSE ${PORT}
WORKDIR /usr/local/share
# Run the application
CMD ["app"]

View file

@ -0,0 +1,16 @@
FROM ghcr.io/cross-rs/aarch64-unknown-linux-gnu:main
ENV PKG_CONFIG_ALLOW_CROSS=1
ENV PKG_CONFIG_PATH=/usr/lib/aarch64-linux-gnu/pkgconfig
RUN dpkg --add-architecture arm64 && \
apt-get update && \
apt-get install -y \
curl \
build-essential \
pkg-config \
libssl-dev \
libpq-dev:arm64 \
&& rm -rf /var/lib/apt/lists/*

View file

@ -0,0 +1,38 @@
# Stage 1: Build the Rust application using the official Rust image
FROM rust:bullseye AS builder
# Install required dependencies for cargo-lambda and Zig (if needed)
RUN apt-get update && apt-get install -y \
curl \
build-essential \
gcc \
libc-dev \
pkg-config \
libssl-dev \
gcc-aarch64-linux-gnu \
&& rm -rf /var/lib/apt/lists/*
# Install Zig (only if needed for cross-compiling or complex setups)
RUN curl -sSf https://ziglang.org/builds/zig-linux-x86_64-0.15.0-dev.377+f01833e03.tar.xz | tar -xJf - -C /usr/local/bin --strip-components=1
RUN rustup target add aarch64-unknown-linux-gnu
# Install cargo-lambda via cargo for building AWS Lambda-compatible binaries
RUN cargo install cargo-lambda
# Create a working directory for the project
WORKDIR /app
# Copy the entire Rust project into the container
COPY . .
# Build the Rust application with cargo-lambda, targeting Lambda-compatible format
RUN cargo lambda build --release --bin lambda-post-article --arm64
# Copy the Lambda-compatible binary from the build stage into the Lambda runtime directory
RUN cd target/lambda/lambda-post-article && zip function.zip bootstrap
# Stage 2: Extract just the zip artifact
FROM alpine:latest AS artifact
# Copy only the zip file out of the builder
COPY --from=builder /app/target/lambda/lambda-post-article/function.zip /function.zip

View file

@ -0,0 +1,27 @@
# Variables
CARGO=cargo
BUILD_DIR=target/lambda
BINARIES=lambda-post-article
# Default target
all: build
# Build all Lambda functions
build:
@for bin in $(BINARIES); do \
$(CARGO) lambda build --release --bin $$bin --output-format zip; \
done
# Clean up build
clean:
$(CARGO) clean
start-emulator:
$(CARGO) lambda start &
run-post:
$(CARGO) lambda invoke lambda-post-article --data-file events/post.json
dev-run-post: start-emulator run-post

View file

@ -0,0 +1,25 @@
# Manage Articles API
<img src="https://writeonce.de/assets/images/logo.png " width="200" alt="writeonce">
The **Manage Articles API** allows you to create, retrieve, and manage blog articles. This API is built to help with managing articles on a blog platform, allowing operations like fetching articles by title, pagination, and article creation.
## Base URL
api.writeonce.de
## Tech Stack
The **Manage Articles API** is built using the following technologies:
- **Rust**: The backend of the API is built with the Rust programming language for high performance and reliability.
- **Diesel**: Diesel is used as the ORM (Object Relational Mapper) to interact with the database.
- **PostgreSQL**: The database used for storing articles and related data.
## Cross build
CROSS_CONTAINER_ENGINE=docker cross build \
--bin writeonce-manage-article-api \
--release \
--target aarch64-unknown-linux-gnu \
--verbose

View file

@ -0,0 +1,9 @@
# For documentation on how to configure this file,
# see https://diesel.rs/guides/configuring-diesel-cli
[print_schema]
file = "src/schema.rs"
custom_type_derives = ["diesel::query_builder::QueryId", "Clone"]
[migrations_directory]
dir = "/home/shoney/manage-articles-api/migrations"

View file

@ -0,0 +1,42 @@
<!DOCTYPE html>
<html lang="en">
<head>
<meta charset="UTF-8">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Swagger UI</title>
<link rel="stylesheet" type="text/css"
href="https://cdnjs.cloudflare.com/ajax/libs/swagger-ui/4.5.0/swagger-ui.css">
<script src="https://cdnjs.cloudflare.com/ajax/libs/swagger-ui/4.5.0/swagger-ui-bundle.js"></script>
</head>
<body>
<div id="swagger-ui"></div>
<script>
const ui = SwaggerUIBundle({
url: "./open-api-spec", // Path to the OpenAPI spec
dom_id: '#swagger-ui',
deepLinking: true,
presets: [
SwaggerUIBundle.presets.apis,
SwaggerUIBundle.presets.topbar
],
layout: "BaseLayout",
requestInterceptor: (request) => {
const token = localStorage.getItem("auth_token"); // Optionally save token in localStorage
if (token) {
request.headers['Authorization'] = 'Bearer ' + token;
}
return request;
},
// Add an "Authorize" button to handle token input
authActions: {
authorize: (authorize) => {
// You can handle custom logic for the "Authorize" button here if needed
}
}
});
</script>
</body>
</html>

View file

@ -0,0 +1,520 @@
{
"openapi": "3.0.3",
"info": {
"title": "Manage Articles API",
"termsOfService": "",
"contact": {
"email": "shoney.john@outlook.com"
},
"host": "api.writeonce.de",
"version": "0.0.1",
"summary": "Blog manage API",
"description": "🚀 **WriteOnce API Server**\n\nSource code available on [GitLab](https://git.writeonce.de/shoney/writeonce-manage-article-api).\n\n🔐 **Authorization:**\nYou can use the API key `REDACTED_TOKEN` to test the authorization filters (except for POST requests).\n\n📦 **S3 + AWS Lambda Integration:**\nWhen JSON documents are saved to an S3 bucket, an AWS Lambda function is triggered automatically. This function then makes a POST request to this API endpoint to create a new article.\n\n💡 View the [AWS Lambda function code](https://git.writeonce.de/shoney/writeonce-function)."
},
"security": [
{
"BearerAuth": []
}
],
"paths": {
"/article/title/{sys_title}": {
"get": {
"summary": "Fetch Articles by page route blog title name",
"operationId": "getArticleByTitle",
"security": [
{
"BearerAuth": []
}
],
"parameters": [
{
"name": "sys_title",
"in": "path",
"required": true,
"description": "Specify the system title of the article",
"schema": {
"type": "string",
"example": "gitlab-runner-with-kubernetes-executor"
}
}
],
"responses": {
"200": {
"description": "Articles found",
"content": {
"application/json": {
"schema": {
"$ref": "#/components/schemas/Article"
}
}
}
},
"404": {
"description": "Article not found",
"content": {
"text/plain": {
"example": "Article not found"
}
}
},
"401": {
"description": "Unauthorized",
"content": {
"text/plain": {
"example": "Unauthorized"
}
}
}
}
}
},
"/articles/{skip}/{limit}": {
"get": {
"summary": "Fetch Articles Pagination",
"operationId": "getArticleByPage",
"security": [
{
"BearerAuth": []
}
],
"parameters": [
{
"name": "skip",
"in": "path",
"required": true,
"description": "The lower range of Article pages",
"schema": {
"type": "integer",
"example": 0
}
},
{
"name": "limit",
"in": "path",
"required": true,
"description": "The limit range of Article pages",
"schema": {
"type": "integer",
"example": 0
}
}
],
"responses": {
"200": {
"description": "Articles found",
"content": {
"application/json": {
"schema": {
"$ref": "#/components/schemas/Article"
}
}
}
},
"404": {
"description": "Article not found",
"content": {
"text/plain": {
"example": "Article not found"
}
}
},
"401": {
"description": "Unauthorized",
"content": {
"text/plain": {
"example": "Unauthorized"
}
}
}
}
}
},
"/article/{id}": {
"get": {
"summary": "Get article by ID",
"operationId": "getArticleById",
"security": [
{
"BearerAuth": []
}
],
"parameters": [
{
"name": "id",
"in": "path",
"required": true,
"description": "The ID of the article to retrieve.",
"schema": {
"type": "integer",
"example": 1
}
}
],
"responses": {
"200": {
"description": "Article found",
"content": {
"application/json": {
"schema": {
"$ref": "#/components/schemas/Article"
}
}
}
},
"404": {
"description": "Article not found",
"content": {
"text/plain": {
"example": "Article not found"
}
}
},
"401": {
"description": "Unauthorized",
"content": {
"text/plain": {
"example": "Unauthorized"
}
}
}
}
}
},
"/article/": {
"post": {
"summary": "Create a new Article",
"operationId": "saveArticle",
"requestBody": {
"description": "Save a new article",
"required": true,
"content": {
"application/json": {
"schema": {
"type": "object",
"properties": {
"title": {
"type": "string",
"example": "Gitlab Runner with Kubernetes Executor"
},
"sys_title": {
"type": "string",
"example": "gitlab-runner-with-kubernetes-executor"
},
"published": {
"type": "boolean",
"example": true
},
"content": {
"type": "object",
"properties": {
"author": {
"type": "string",
"example": "Shoney Arickathil"
},
"content": {
"type": "object",
"properties": {
"codes": {
"type": "array",
"items": {
"type": "object",
"properties": {
"language": {
"type": "string",
"example": "yaml"
},
"paragraphIndex": {
"type": "integer",
"example": 0
},
"sectionIndex": {
"type": "integer",
"example": 1
},
"snippet": {
"type": "string",
"example": "gitlab-runnergitlab-runner-deploy/gitlab-runner-with-kubernetes-executor"
},
"title": {
"type": "string",
"example": "gitlab runner deploy"
}
}
}
},
"images": {
"type": "array",
"items": {
"type": "object",
"properties": {}
}
},
"img": {
"type": "object",
"properties": {
"caption": {
"type": "string",
"example": "Gitlab runner with kubernetes executor."
},
"path": {
"type": "string",
"example": "https://writeonce-filecloud-bucket.s3.eu-central-1.amazonaws.com/images/kubernetes.png"
}
}
},
"sections": {
"type": "array",
"items": {
"type": "object",
"properties": {
"heading": {
"type": "string",
"example": "Introduction"
},
"paragraphs": {
"type": "array",
"items": {
"type": "string",
"example": "Deploying a Gitlab runner using kubernetes is a great option to overcome the limitations of other gitlab runner executor such as docker and docker machine."
}
}
}
}
}
}
},
"publishedOn": {
"type": "integer",
"example": 1737414886
},
"references": {
"type": "array",
"items": {
"type": "object",
"properties": {
"dateAccessed": {
"type": "integer",
"example": 1737414886
},
"title": {
"type": "string",
"example": "Gitlab Kubernetes executor"
},
"url": {
"type": "string",
"example": "https://docs.gitlab.com/runner/executors/kubernetes/"
}
}
}
},
"systitle": {
"type": "string",
"example": "gitlab-runner-with-kubernetes-executor"
},
"tags": {
"type": "array",
"items": {
"type": "string",
"example": ["Kubernetes", "Gitlab runner"]
}
},
"title": {
"type": "string",
"example": "Gitlab Runner with Kubernetes Executor"
}
}
},
"do_aws_sync": {
"type": "boolean",
"example": false
}
}
}
}
}
},
"responses": {
"201": {
"description": "created successfully",
"content": {
"application/json": {
"schema": {
"type": "object",
"properties": {
"message": {
"type": "string",
"example": ""
},
"articleId": {
"type": "integer",
"example": "2"
}
}
}
}
}
},
"400": {
"description": "Bad request"
},
"500": {
"description": "Internal server error"
}
}
}
}
},
"components": {
"securitySchemes": {
"BearerAuth": {
"type": "http",
"scheme": "bearer",
"bearerFormat": "JWT"
}
},
"schemas": {
"Article": {
"type": "object",
"properties": {
"title": {
"type": "string",
"example": "Gitlab Runner with Kubernetes Executor"
},
"sys_title": {
"type": "string",
"example": "gitlab-runner-with-kubernetes-executor"
},
"published": {
"type": "boolean",
"example": true
},
"content": {
"type": "object",
"properties": {
"author": {
"type": "string",
"example": "Shoney Arickathil"
},
"content": {
"type": "object",
"properties": {
"codes": {
"type": "array",
"items": {
"type": "object",
"properties": {
"language": {
"type": "string",
"example": "yaml"
},
"paragraphIndex": {
"type": "integer",
"example": 0
},
"sectionIndex": {
"type": "integer",
"example": 1
},
"snippet": {
"type": "string",
"example": "gitlab-runnergitlab-runner-deploy/gitlab-runner-with-kubernetes-executor"
},
"title": {
"type": "string",
"example": "gitlab runner deploy"
}
}
}
},
"images": {
"type": "array",
"items": {
"type": "object",
"properties": {}
}
},
"img": {
"type": "object",
"properties": {
"caption": {
"type": "string",
"example": "Gitlab runner with kubernetes executor."
},
"path": {
"type": "string",
"example": "https://writeonce-filecloud-bucket.s3.eu-central-1.amazonaws.com/images/kubernetes.png"
}
}
},
"sections": {
"type": "array",
"items": {
"type": "object",
"properties": {
"heading": {
"type": "string",
"example": "Introduction"
},
"paragraphs": {
"type": "array",
"items": {
"type": "string",
"example": "Deploying a Gitlab runner using kubernetes is a great option to overcome the limitations of other gitlab runner executor such as docker and docker machine."
}
}
}
}
}
}
},
"publishedOn": {
"type": "integer",
"example": 1737414886
},
"references": {
"type": "array",
"items": {
"type": "object",
"properties": {
"dateAccessed": {
"type": "integer",
"example": 1737414886
},
"title": {
"type": "string",
"example": "Gitlab Kubernetes executor"
},
"url": {
"type": "string",
"example": "https://docs.gitlab.com/runner/executors/kubernetes/"
}
}
}
},
"systitle": {
"type": "string",
"example": "gitlab-runner-with-kubernetes-executor"
},
"tags": {
"type": "array",
"items": {
"type": "string",
"example": ["Kubernetes", "Gitlab runner"]
}
},
"title": {
"type": "string",
"example": "Gitlab Runner with Kubernetes Executor"
}
}
},
"do_aws_sync": {
"type": "boolean",
"example": false
}
}
}
}
}
}

View file

@ -0,0 +1,53 @@
services:
writeonce-postgress-db:
image: postgres:17.0
container_name: writeonce-postgress-db
restart: always
# set shared memory limit when using docker-compose
shm_size: 128mb
# or set shared memory limit when deploy via swarm stack
#volumes:
# - type: tmpfs
# target: /dev/shm
# tmpfs:
# size: 134217728 # 128*2^20 bytes = 128Mb
environment:
POSTGRES_USER: ${POSTGRES_USER}
POSTGRES_PASSWORD: ${POSTGRES_PASSWORD}
POSTGRES_DB: ${POSTGRES_DB}
DB_USER: ${DB_USER}
DB_PASSWORD: ${DB_PASSWORD}
DB_NAME: ${DB_NAME}
volumes:
- ./postgress/:/docker-entrypoint-initdb.d
healthcheck:
test: ["CMD-SHELL", "pg_isready -U ${POSTGRES_USER}"]
interval: 10s
timeout: 5s
retries: 5
networks:
- writeonce-network
writeonce-manage-article-api:
image: registry.writeonce.de/shoney/writeonce-manage-article-api
container_name: writeonce-manage-article-api
ports:
- ${PORT}:${PORT}
depends_on:
writeonce-postgress-db:
condition: service_healthy
environment:
- AWS_INFRA_BASE_URL=${AWS_INFRA_BASE_URL}
- PORT=${PORT}
- API_ACCESS_TOKEN=${API_ACCESS_TOKEN}
- API_ACCESS_ADMIN_TOKEN=${API_ACCESS_ADMIN_TOKEN}
- DATABASE_URL=${DATABASE_URL}
restart: always
networks:
- writeonce-network
networks:
writeonce-network:
external: true

View file

@ -0,0 +1,51 @@
{
"content": {
"author": "Shoney Arickathil",
"content": {
"codes": [
{
"language": "yaml",
"paragraphIndex": 0,
"sectionIndex": 1,
"snippet": "gitlab-runnergitlab-runner-deploy/gitlab-runner-with-kubernetes-executor",
"title": "gitlab runner deploy"
}
],
"images": [
{}
],
"img": {
"caption": "Gitlab runner with kubernetes executor.",
"path": "https://writeonce-filecloud-bucket.s3.eu-central-1.amazonaws.com/images/kubernetes.png"
},
"sections": [
{
"heading": "Introduction",
"paragraphs": [
"Deploying a Gitlab runner using kubernetes is a great option to overcome the limitations of other gitlab runner executor such as docker and docker machine."
]
}
]
},
"publishedOn": 1737414886,
"references": [
{
"dateAccessed": 1737414886,
"title": "Gitlab Kubernetes executor",
"url": "https://docs.gitlab.com/runner/executors/kubernetes/"
}
],
"systitle": "gitlab-runner-with-kubernetes-executor",
"tags": [
[
"Kubernetes",
"Gitlab runner"
]
],
"title": "Gitlab Runner with Kubernetes Executor"
},
"do_aws_sync": false,
"published": true,
"sys_title": "gitlab-runner-with-kubernetes-executor",
"title": "Gitlab Runner with Kubernetes Executor"
}

View file

@ -0,0 +1,16 @@
#!/bin/sh
if [ -z "$API_PORT" ] || [ -z "$AWS_INFRA_BASE_URL" ] || [ -z "$DATABASE_URL" ] || [ -z "$API_ACCESS_TOKEN" ] || [ -z "$API_ACCESS_ADMIN_TOKEN" ]; then
echo "Error: Missing required environment variables"
exit 1
fi
echo PORT="$API_PORT"
docker build \
--build-arg PORT="$API_PORT" \
--build-arg AWS_INFRA_BASE_URL="$AWS_INFRA_BASE_URL" \
--build-arg DATABASE_URL="$DATABASE_URL" \
--build-arg API_ACCESS_TOKEN="$API_ACCESS_TOKEN" \
--build-arg API_ACCESS_ADMIN_TOKEN="$API_ACCESS_ADMIN_TOKEN" \
-t "$CI_REGISTRY_IMAGE:$CI_COMMIT_SHORT_SHA" .

View file

@ -0,0 +1,3 @@
#!/bin/sh
docker login -u $CI_REGISTRY_USER -p $CI_REGISTRY_PASSWORD $CI_REGISTRY

View file

@ -0,0 +1,2 @@
#!/bin/sh
docker push "$CI_REGISTRY_IMAGE:$CI_COMMIT_SHORT_SHA"

Some files were not shown because too many files have changed in this diff Show more