begin
This commit is contained in:
commit
c32e482a58
271 changed files with 15357 additions and 0 deletions
3
.gitignore
vendored
Normal file
3
.gitignore
vendored
Normal file
|
|
@ -0,0 +1,3 @@
|
|||
/target
|
||||
/data
|
||||
/content
|
||||
6
.gitmodules
vendored
Normal file
6
.gitmodules
vendored
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
[submodule "reference/writeonce-app"]
|
||||
path = reference/writeonce-app
|
||||
url = https://github.com/shoneyJ/writeonce-app
|
||||
[submodule "reference/writeonce-api"]
|
||||
path = reference/writeonce-api
|
||||
url = https://github.com/shoneyJ/writeonce-api
|
||||
579
Cargo.lock
generated
Normal file
579
Cargo.lock
generated
Normal file
|
|
@ -0,0 +1,579 @@
|
|||
# This file is automatically @generated by Cargo.
|
||||
# It is not intended for manual editing.
|
||||
version = 4
|
||||
|
||||
[[package]]
|
||||
name = "anyhow"
|
||||
version = "1.0.102"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "7f202df86484c868dbad7eaa557ef785d5c66295e41b460ef922eca0723b842c"
|
||||
|
||||
[[package]]
|
||||
name = "bincode"
|
||||
version = "1.3.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b1f45e9417d87227c7a56d22e471c6206462cba514c7590c09aff4cf6d1ddcad"
|
||||
dependencies = [
|
||||
"serde",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "bitflags"
|
||||
version = "2.11.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "843867be96c8daad0d758b57df9392b6d8d271134fce549de6ce169ff98a92af"
|
||||
|
||||
[[package]]
|
||||
name = "cfg-if"
|
||||
version = "1.0.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801"
|
||||
|
||||
[[package]]
|
||||
name = "equivalent"
|
||||
version = "1.0.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "877a4ace8713b0bcf2a4e7eec82529c029f1d0619886d18145fea96c3ffe5c0f"
|
||||
|
||||
[[package]]
|
||||
name = "errno"
|
||||
version = "0.3.14"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "39cab71617ae0d63f51a36d69f866391735b51691dbda63cf6f96d042b63efeb"
|
||||
dependencies = [
|
||||
"libc",
|
||||
"windows-sys",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "fastrand"
|
||||
version = "2.3.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "37909eebbb50d72f9059c3b6d82c0463f2ff062c9e95845c43a6c9c0355411be"
|
||||
|
||||
[[package]]
|
||||
name = "foldhash"
|
||||
version = "0.1.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d9c4f5dac5e15c24eb999c26181a6ca40b39fe946cbe4c263c7209467bc83af2"
|
||||
|
||||
[[package]]
|
||||
name = "getrandom"
|
||||
version = "0.4.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0de51e6874e94e7bf76d726fc5d13ba782deca734ff60d5bb2fb2607c7406555"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"libc",
|
||||
"r-efi",
|
||||
"wasip2",
|
||||
"wasip3",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "hashbrown"
|
||||
version = "0.15.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9229cfe53dfd69f0609a49f65461bd93001ea1ef889cd5529dd176593f5338a1"
|
||||
dependencies = [
|
||||
"foldhash",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "hashbrown"
|
||||
version = "0.16.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "841d1cc9bed7f9236f321df977030373f4a4163ae1a7dbfe1a51a2c1a51d9100"
|
||||
|
||||
[[package]]
|
||||
name = "heck"
|
||||
version = "0.5.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "2304e00983f87ffb38b55b444b5e3b60a884b5d30c0fca7d82fe33449bbe55ea"
|
||||
|
||||
[[package]]
|
||||
name = "id-arena"
|
||||
version = "2.3.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3d3067d79b975e8844ca9eb072e16b31c3c1c36928edf9c6789548c524d0d954"
|
||||
|
||||
[[package]]
|
||||
name = "indexmap"
|
||||
version = "2.13.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "7714e70437a7dc3ac8eb7e6f8df75fd8eb422675fc7678aff7364301092b1017"
|
||||
dependencies = [
|
||||
"equivalent",
|
||||
"hashbrown 0.16.1",
|
||||
"serde",
|
||||
"serde_core",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "itoa"
|
||||
version = "1.0.18"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682"
|
||||
|
||||
[[package]]
|
||||
name = "leb128fmt"
|
||||
version = "0.1.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "09edd9e8b54e49e587e4f6295a7d29c3ea94d469cb40ab8ca70b288248a81db2"
|
||||
|
||||
[[package]]
|
||||
name = "libc"
|
||||
version = "0.2.183"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b5b646652bf6661599e1da8901b3b9522896f01e736bad5f723fe7a3a27f899d"
|
||||
|
||||
[[package]]
|
||||
name = "linux-raw-sys"
|
||||
version = "0.12.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "32a66949e030da00e8c7d4434b251670a91556f4144941d37452769c25d58a53"
|
||||
|
||||
[[package]]
|
||||
name = "log"
|
||||
version = "0.4.29"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "5e5032e24019045c762d3c0f28f5b6b8bbf38563a65908389bf7978758920897"
|
||||
|
||||
[[package]]
|
||||
name = "memchr"
|
||||
version = "2.8.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f8ca58f447f06ed17d5fc4043ce1b10dd205e060fb3ce5b979b8ed8e59ff3f79"
|
||||
|
||||
[[package]]
|
||||
name = "once_cell"
|
||||
version = "1.21.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50"
|
||||
|
||||
[[package]]
|
||||
name = "prettyplease"
|
||||
version = "0.2.37"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "479ca8adacdd7ce8f1fb39ce9ecccbfe93a3f1344b3d0d97f20bc0196208f62b"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"syn",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "proc-macro2"
|
||||
version = "1.0.106"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "8fd00f0bb2e90d81d1044c2b32617f68fcb9fa3bb7640c23e9c748e53fb30934"
|
||||
dependencies = [
|
||||
"unicode-ident",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "quote"
|
||||
version = "1.0.45"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "41f2619966050689382d2b44f664f4bc593e129785a36d6ee376ddf37259b924"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "r-efi"
|
||||
version = "6.0.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf"
|
||||
|
||||
[[package]]
|
||||
name = "rustix"
|
||||
version = "1.1.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b6fe4565b9518b83ef4f91bb47ce29620ca828bd32cb7e408f0062e9930ba190"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"errno",
|
||||
"libc",
|
||||
"linux-raw-sys",
|
||||
"windows-sys",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "semver"
|
||||
version = "1.0.27"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d767eb0aabc880b29956c35734170f26ed551a859dbd361d140cdbeca61ab1e2"
|
||||
|
||||
[[package]]
|
||||
name = "serde"
|
||||
version = "1.0.228"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9a8e94ea7f378bd32cbbd37198a4a91436180c5bb472411e48b5ec2e2124ae9e"
|
||||
dependencies = [
|
||||
"serde_core",
|
||||
"serde_derive",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "serde_core"
|
||||
version = "1.0.228"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "41d385c7d4ca58e59fc732af25c3983b67ac852c1a25000afe1175de458b67ad"
|
||||
dependencies = [
|
||||
"serde_derive",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "serde_derive"
|
||||
version = "1.0.228"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d540f220d3187173da220f885ab66608367b6574e925011a9353e4badda91d79"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "serde_json"
|
||||
version = "1.0.149"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "83fc039473c5595ace860d8c4fafa220ff474b3fc6bfdb4293327f1a37e94d86"
|
||||
dependencies = [
|
||||
"itoa",
|
||||
"memchr",
|
||||
"serde",
|
||||
"serde_core",
|
||||
"zmij",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "syn"
|
||||
version = "2.0.117"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e665b8803e7b1d2a727f4023456bbbbe74da67099c585258af0ad9c5013b9b99"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"unicode-ident",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "tempfile"
|
||||
version = "3.27.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd"
|
||||
dependencies = [
|
||||
"fastrand",
|
||||
"getrandom",
|
||||
"once_cell",
|
||||
"rustix",
|
||||
"windows-sys",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "unicode-ident"
|
||||
version = "1.0.24"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e6e4313cd5fcd3dad5cafa179702e2b244f760991f45397d14d4ebf38247da75"
|
||||
|
||||
[[package]]
|
||||
name = "unicode-xid"
|
||||
version = "0.2.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ebc1c04c71510c7f702b52b7c350734c9ff1295c464a03335b00bb84fc54f853"
|
||||
|
||||
[[package]]
|
||||
name = "wasip2"
|
||||
version = "1.0.2+wasi-0.2.9"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9517f9239f02c069db75e65f174b3da828fe5f5b945c4dd26bd25d89c03ebcf5"
|
||||
dependencies = [
|
||||
"wit-bindgen",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wasip3"
|
||||
version = "0.4.0+wasi-0.3.0-rc-2026-01-06"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "5428f8bf88ea5ddc08faddef2ac4a67e390b88186c703ce6dbd955e1c145aca5"
|
||||
dependencies = [
|
||||
"wit-bindgen",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wasm-encoder"
|
||||
version = "0.244.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "990065f2fe63003fe337b932cfb5e3b80e0b4d0f5ff650e6985b1048f62c8319"
|
||||
dependencies = [
|
||||
"leb128fmt",
|
||||
"wasmparser",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wasm-metadata"
|
||||
version = "0.244.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "bb0e353e6a2fbdc176932bbaab493762eb1255a7900fe0fea1a2f96c296cc909"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"indexmap",
|
||||
"wasm-encoder",
|
||||
"wasmparser",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wasmparser"
|
||||
version = "0.244.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "47b807c72e1bac69382b3a6fb3dbe8ea4c0ed87ff5629b8685ae6b9a611028fe"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"hashbrown 0.15.5",
|
||||
"indexmap",
|
||||
"semver",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "windows-link"
|
||||
version = "0.2.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f0805222e57f7521d6a62e36fa9163bc891acd422f971defe97d64e70d0a4fe5"
|
||||
|
||||
[[package]]
|
||||
name = "windows-sys"
|
||||
version = "0.61.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ae137229bcbd6cdf0f7b80a31df61766145077ddf49416a728b02cb3921ff3fc"
|
||||
dependencies = [
|
||||
"windows-link",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wit-bindgen"
|
||||
version = "0.51.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d7249219f66ced02969388cf2bb044a09756a083d0fab1e566056b04d9fbcaa5"
|
||||
dependencies = [
|
||||
"wit-bindgen-rust-macro",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wit-bindgen-core"
|
||||
version = "0.51.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ea61de684c3ea68cb082b7a88508a8b27fcc8b797d738bfc99a82facf1d752dc"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"heck",
|
||||
"wit-parser",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wit-bindgen-rust"
|
||||
version = "0.51.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b7c566e0f4b284dd6561c786d9cb0142da491f46a9fbed79ea69cdad5db17f21"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"heck",
|
||||
"indexmap",
|
||||
"prettyplease",
|
||||
"syn",
|
||||
"wasm-metadata",
|
||||
"wit-bindgen-core",
|
||||
"wit-component",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wit-bindgen-rust-macro"
|
||||
version = "0.51.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0c0f9bfd77e6a48eccf51359e3ae77140a7f50b1e2ebfe62422d8afdaffab17a"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"prettyplease",
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn",
|
||||
"wit-bindgen-core",
|
||||
"wit-bindgen-rust",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wit-component"
|
||||
version = "0.244.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9d66ea20e9553b30172b5e831994e35fbde2d165325bec84fc43dbf6f4eb9cb2"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"bitflags",
|
||||
"indexmap",
|
||||
"log",
|
||||
"serde",
|
||||
"serde_derive",
|
||||
"serde_json",
|
||||
"wasm-encoder",
|
||||
"wasm-metadata",
|
||||
"wasmparser",
|
||||
"wit-parser",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wit-parser"
|
||||
version = "0.244.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ecc8ac4bc1dc3381b7f59c34f00b67e18f910c2c0f50015669dde7def656a736"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"id-arena",
|
||||
"indexmap",
|
||||
"log",
|
||||
"semver",
|
||||
"serde",
|
||||
"serde_derive",
|
||||
"serde_json",
|
||||
"unicode-xid",
|
||||
"wasmparser",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wo-event"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"libc",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wo-htmlx"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"serde",
|
||||
"serde_json",
|
||||
"tempfile",
|
||||
"wo-md",
|
||||
"wo-model",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wo-http"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"libc",
|
||||
"tempfile",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wo-index"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"bincode",
|
||||
"libc",
|
||||
"serde",
|
||||
"tempfile",
|
||||
"wo-model",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wo-md"
|
||||
version = "0.1.0"
|
||||
|
||||
[[package]]
|
||||
name = "wo-model"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"bincode",
|
||||
"serde",
|
||||
"serde_json",
|
||||
"tempfile",
|
||||
"wo-md",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wo-route"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"wo-http",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wo-rt"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"bincode",
|
||||
"libc",
|
||||
"serde",
|
||||
"serde_json",
|
||||
"tempfile",
|
||||
"wo-event",
|
||||
"wo-htmlx",
|
||||
"wo-http",
|
||||
"wo-md",
|
||||
"wo-model",
|
||||
"wo-route",
|
||||
"wo-serve",
|
||||
"wo-store",
|
||||
"wo-sub",
|
||||
"wo-watch",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wo-seg"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"bincode",
|
||||
"libc",
|
||||
"serde",
|
||||
"tempfile",
|
||||
"wo-model",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wo-serve"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"libc",
|
||||
"tempfile",
|
||||
"wo-http",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wo-store"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"serde_json",
|
||||
"tempfile",
|
||||
"wo-index",
|
||||
"wo-model",
|
||||
"wo-seg",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wo-sub"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"bincode",
|
||||
"libc",
|
||||
"serde",
|
||||
"serde_json",
|
||||
"tempfile",
|
||||
"wo-event",
|
||||
"wo-model",
|
||||
"wo-store",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wo-watch"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"libc",
|
||||
"tempfile",
|
||||
"wo-store",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "zmij"
|
||||
version = "1.0.21"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b8848ee67ecc8aedbaf3e4122217aff892639231befc6a1b58d29fff4c2cabaa"
|
||||
23
Cargo.toml
Normal file
23
Cargo.toml
Normal file
|
|
@ -0,0 +1,23 @@
|
|||
[workspace]
|
||||
resolver = "2"
|
||||
members = [
|
||||
"crates/wo-model",
|
||||
"crates/wo-seg",
|
||||
"crates/wo-index",
|
||||
"crates/wo-store",
|
||||
"crates/wo-watch",
|
||||
"crates/wo-event",
|
||||
"crates/wo-sub",
|
||||
"crates/wo-rt",
|
||||
"crates/wo-http",
|
||||
"crates/wo-route",
|
||||
"crates/wo-htmlx",
|
||||
"crates/wo-md",
|
||||
"crates/wo-serve",
|
||||
]
|
||||
|
||||
[workspace.dependencies]
|
||||
serde = { version = "1", features = ["derive"] }
|
||||
serde_json = "1"
|
||||
bincode = "1"
|
||||
libc = "0.2"
|
||||
64
README.md
Normal file
64
README.md
Normal file
|
|
@ -0,0 +1,64 @@
|
|||
# writeonce
|
||||
|
||||
A single self-contained binary that serves a content platform — no external database, no cloud pipeline, no JavaScript framework. Built in Rust on raw Linux kernel primitives.
|
||||
|
||||
## Why
|
||||
|
||||
The original writeonce system spread across five repositories, four languages, AWS infrastructure (S3, Lambda, API Gateway), PostgreSQL, and an Angular frontend. All of that to serve articles from local files. This project collapses everything into one process that owns storage, serves content, and pushes real-time updates.
|
||||
|
||||
## Architecture
|
||||
|
||||
- **Single process** — one binary replaces S3 + Lambda + Rust API + PostgreSQL + Angular
|
||||
- **Embedded storage** — custom `.seg` segment files with positional indexing, no external database
|
||||
- **Real-time subscriptions** — route-based SSE streams push content diffs to connected clients
|
||||
- **Server-rendered HTML** — `.htmlx` templates with data bindings, minimal client-side JS
|
||||
- **Markdown-first content** — `.md` files are the source of truth, JSON holds only metadata
|
||||
- **Linux kernel I/O** — `epoll`, `inotify`, `eventfd`, `timerfd`, `sendfile` — no tokio, no async runtime
|
||||
|
||||
## Workspace Crates
|
||||
|
||||
| Crate | Purpose |
|
||||
|-------|---------|
|
||||
| `wo-model` | Article and metadata types |
|
||||
| `wo-seg` | Segment file reader/writer (.seg format) |
|
||||
| `wo-index` | Title hash map, date sorted array, tags inverted index |
|
||||
| `wo-store` | Query API over segments and indexes |
|
||||
| `wo-watch` | `inotify`-based content directory watcher |
|
||||
| `wo-event` | `epoll` event loop, `eventfd`, `timerfd`, `signalfd` |
|
||||
| `wo-sub` | Subscription manager and diff delivery |
|
||||
| `wo-rt` | Single-threaded runtime tying I/O sources together |
|
||||
| `wo-http` | HTTP request parsing and response writing |
|
||||
| `wo-route` | URL routing and handler dispatch |
|
||||
| `wo-htmlx` | Template engine for `.htmlx` files |
|
||||
| `wo-md` | Markdown to HTML rendering |
|
||||
| `wo-serve` | Binary entry point — wires everything together |
|
||||
|
||||
## Build
|
||||
|
||||
```sh
|
||||
cargo build --release
|
||||
```
|
||||
|
||||
## Deploy
|
||||
|
||||
The binary runs behind nginx with Let's Encrypt SSL. See `infra/setup.sh` for first-time server setup and `docs/07-ssl.md` for the full deployment walkthrough.
|
||||
|
||||
```sh
|
||||
# Build, copy binary, sync content, restart service
|
||||
./infra/deploy.sh
|
||||
```
|
||||
|
||||
## Documentation
|
||||
|
||||
Design documents live in `docs/`:
|
||||
|
||||
- `00-linux.md` — Linux kernel primitives used
|
||||
- `01-problem.md` — Problem statement and motivation
|
||||
- `02-recovery.md` — Target architecture
|
||||
- `03-data.md` — Embedded storage and subscription model
|
||||
- `04-ui.md` — Server-rendered HTMLX templates
|
||||
- `05-datalayer.md` — Data layer implementation status
|
||||
- `06-markdown-render.md` — Markdown-first content model
|
||||
- `07-ssl.md` — SSL, nginx, and deployment
|
||||
- `runtime/` — Deep dives on async runtimes, fibers, and Rust's ownership model
|
||||
- `future-scope/` — Planned features including AI agent content management
|
||||
7
crates/wo-event/Cargo.toml
Normal file
7
crates/wo-event/Cargo.toml
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
[package]
|
||||
name = "wo-event"
|
||||
version = "0.1.0"
|
||||
edition = "2021"
|
||||
|
||||
[dependencies]
|
||||
libc = { workspace = true }
|
||||
183
crates/wo-event/src/epoll.rs
Normal file
183
crates/wo-event/src/epoll.rs
Normal file
|
|
@ -0,0 +1,183 @@
|
|||
use std::io;
|
||||
use std::os::unix::io::RawFd;
|
||||
use std::time::Duration;
|
||||
|
||||
/// Caller-assigned identifier for a registered file descriptor.
|
||||
pub type Token = u64;
|
||||
|
||||
/// Interest flags for epoll registration.
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub enum Interest {
|
||||
Readable,
|
||||
Writable,
|
||||
ReadWrite,
|
||||
}
|
||||
|
||||
impl Interest {
|
||||
fn to_epoll_flags(self) -> u32 {
|
||||
match self {
|
||||
Interest::Readable => libc::EPOLLIN as u32,
|
||||
Interest::Writable => libc::EPOLLOUT as u32,
|
||||
Interest::ReadWrite => (libc::EPOLLIN | libc::EPOLLOUT) as u32,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// An event delivered by the event loop.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct Event {
|
||||
pub token: Token,
|
||||
pub readable: bool,
|
||||
pub writable: bool,
|
||||
pub error: bool,
|
||||
pub hangup: bool,
|
||||
}
|
||||
|
||||
/// Single-threaded event loop built on epoll.
|
||||
pub struct EventLoop {
|
||||
epoll_fd: RawFd,
|
||||
}
|
||||
|
||||
impl EventLoop {
|
||||
/// Create a new event loop.
|
||||
pub fn new() -> io::Result<Self> {
|
||||
let fd = unsafe { libc::epoll_create1(libc::EPOLL_CLOEXEC) };
|
||||
if fd < 0 {
|
||||
return Err(io::Error::last_os_error());
|
||||
}
|
||||
Ok(Self { epoll_fd: fd })
|
||||
}
|
||||
|
||||
/// Register a file descriptor for the given interest.
|
||||
pub fn register(&self, fd: RawFd, interest: Interest, token: Token) -> io::Result<()> {
|
||||
let mut event = libc::epoll_event {
|
||||
events: interest.to_epoll_flags() | libc::EPOLLRDHUP as u32,
|
||||
u64: token,
|
||||
};
|
||||
let ret = unsafe { libc::epoll_ctl(self.epoll_fd, libc::EPOLL_CTL_ADD, fd, &mut event) };
|
||||
if ret < 0 {
|
||||
Err(io::Error::last_os_error())
|
||||
} else {
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
/// Modify interest for an already-registered fd.
|
||||
pub fn modify(&self, fd: RawFd, interest: Interest, token: Token) -> io::Result<()> {
|
||||
let mut event = libc::epoll_event {
|
||||
events: interest.to_epoll_flags() | libc::EPOLLRDHUP as u32,
|
||||
u64: token,
|
||||
};
|
||||
let ret = unsafe { libc::epoll_ctl(self.epoll_fd, libc::EPOLL_CTL_MOD, fd, &mut event) };
|
||||
if ret < 0 {
|
||||
Err(io::Error::last_os_error())
|
||||
} else {
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
/// Remove a file descriptor from the event loop.
|
||||
pub fn deregister(&self, fd: RawFd) -> io::Result<()> {
|
||||
let ret = unsafe {
|
||||
libc::epoll_ctl(self.epoll_fd, libc::EPOLL_CTL_DEL, fd, std::ptr::null_mut())
|
||||
};
|
||||
if ret < 0 {
|
||||
Err(io::Error::last_os_error())
|
||||
} else {
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
/// Wait for events. Returns when at least one event is ready or timeout expires.
|
||||
///
|
||||
/// `timeout`: `None` blocks indefinitely, `Some(duration)` sets a timeout.
|
||||
/// Returns up to 64 events per call.
|
||||
pub fn poll(&self, timeout: Option<Duration>) -> io::Result<Vec<Event>> {
|
||||
let timeout_ms = match timeout {
|
||||
None => -1i32,
|
||||
Some(d) => d.as_millis() as i32,
|
||||
};
|
||||
|
||||
let mut events = [libc::epoll_event { events: 0, u64: 0 }; 64];
|
||||
let n = unsafe {
|
||||
libc::epoll_wait(self.epoll_fd, events.as_mut_ptr(), events.len() as i32, timeout_ms)
|
||||
};
|
||||
|
||||
if n < 0 {
|
||||
let err = io::Error::last_os_error();
|
||||
// EINTR is not an error — just return empty.
|
||||
if err.raw_os_error() == Some(libc::EINTR) {
|
||||
return Ok(vec![]);
|
||||
}
|
||||
return Err(err);
|
||||
}
|
||||
|
||||
let result = (0..n as usize)
|
||||
.map(|i| {
|
||||
let e = events[i].events;
|
||||
Event {
|
||||
token: events[i].u64,
|
||||
readable: (e & libc::EPOLLIN as u32) != 0,
|
||||
writable: (e & libc::EPOLLOUT as u32) != 0,
|
||||
error: (e & libc::EPOLLERR as u32) != 0,
|
||||
hangup: (e & (libc::EPOLLHUP | libc::EPOLLRDHUP) as u32) != 0,
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
/// Get the raw epoll file descriptor.
|
||||
pub fn fd(&self) -> RawFd {
|
||||
self.epoll_fd
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for EventLoop {
|
||||
fn drop(&mut self) {
|
||||
unsafe { libc::close(self.epoll_fd) };
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::EventFd;
|
||||
|
||||
#[test]
|
||||
fn register_and_poll_eventfd() {
|
||||
let eloop = EventLoop::new().unwrap();
|
||||
let efd = EventFd::new().unwrap();
|
||||
|
||||
eloop.register(efd.fd(), Interest::Readable, 42).unwrap();
|
||||
|
||||
// Write to eventfd from the same thread.
|
||||
efd.write(1).unwrap();
|
||||
|
||||
let events = eloop.poll(Some(Duration::from_millis(100))).unwrap();
|
||||
assert_eq!(events.len(), 1);
|
||||
assert_eq!(events[0].token, 42);
|
||||
assert!(events[0].readable);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn poll_timeout_no_events() {
|
||||
let eloop = EventLoop::new().unwrap();
|
||||
let events = eloop.poll(Some(Duration::from_millis(10))).unwrap();
|
||||
assert!(events.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn deregister() {
|
||||
let eloop = EventLoop::new().unwrap();
|
||||
let efd = EventFd::new().unwrap();
|
||||
|
||||
eloop.register(efd.fd(), Interest::Readable, 1).unwrap();
|
||||
eloop.deregister(efd.fd()).unwrap();
|
||||
|
||||
efd.write(1).unwrap();
|
||||
let events = eloop.poll(Some(Duration::from_millis(10))).unwrap();
|
||||
assert!(events.is_empty());
|
||||
}
|
||||
}
|
||||
66
crates/wo-event/src/eventfd.rs
Normal file
66
crates/wo-event/src/eventfd.rs
Normal file
|
|
@ -0,0 +1,66 @@
|
|||
use std::io;
|
||||
use std::os::unix::io::RawFd;
|
||||
|
||||
/// Wrapper around Linux eventfd(2) for lightweight signaling.
|
||||
pub struct EventFd {
|
||||
fd: RawFd,
|
||||
}
|
||||
|
||||
impl EventFd {
|
||||
/// Create a new eventfd with initial value 0.
|
||||
pub fn new() -> io::Result<Self> {
|
||||
let fd = unsafe { libc::eventfd(0, libc::EFD_NONBLOCK | libc::EFD_CLOEXEC) };
|
||||
if fd < 0 {
|
||||
return Err(io::Error::last_os_error());
|
||||
}
|
||||
Ok(Self { fd })
|
||||
}
|
||||
|
||||
/// Write a value to the eventfd (signals waiters).
|
||||
pub fn write(&self, val: u64) -> io::Result<()> {
|
||||
let buf = val.to_ne_bytes();
|
||||
let ret = unsafe { libc::write(self.fd, buf.as_ptr() as *const libc::c_void, 8) };
|
||||
if ret < 0 {
|
||||
Err(io::Error::last_os_error())
|
||||
} else {
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
/// Read the current value (resets to 0).
|
||||
pub fn read(&self) -> io::Result<u64> {
|
||||
let mut buf = [0u8; 8];
|
||||
let ret = unsafe { libc::read(self.fd, buf.as_mut_ptr() as *mut libc::c_void, 8) };
|
||||
if ret < 0 {
|
||||
Err(io::Error::last_os_error())
|
||||
} else {
|
||||
Ok(u64::from_ne_bytes(buf))
|
||||
}
|
||||
}
|
||||
|
||||
/// Get the raw file descriptor for epoll registration.
|
||||
pub fn fd(&self) -> RawFd {
|
||||
self.fd
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for EventFd {
|
||||
fn drop(&mut self) {
|
||||
unsafe { libc::close(self.fd) };
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn write_and_read() {
|
||||
let efd = EventFd::new().unwrap();
|
||||
efd.write(5).unwrap();
|
||||
efd.write(3).unwrap();
|
||||
// eventfd accumulates writes.
|
||||
let val = efd.read().unwrap();
|
||||
assert_eq!(val, 8);
|
||||
}
|
||||
}
|
||||
9
crates/wo-event/src/lib.rs
Normal file
9
crates/wo-event/src/lib.rs
Normal file
|
|
@ -0,0 +1,9 @@
|
|||
mod epoll;
|
||||
mod eventfd;
|
||||
mod timerfd;
|
||||
mod signalfd;
|
||||
|
||||
pub use epoll::{EventLoop, Event, Interest, Token};
|
||||
pub use eventfd::EventFd;
|
||||
pub use timerfd::TimerFd;
|
||||
pub use signalfd::SignalFd;
|
||||
62
crates/wo-event/src/signalfd.rs
Normal file
62
crates/wo-event/src/signalfd.rs
Normal file
|
|
@ -0,0 +1,62 @@
|
|||
use std::io;
|
||||
use std::os::unix::io::RawFd;
|
||||
|
||||
/// Wrapper around Linux signalfd(2) for handling signals via fd.
|
||||
///
|
||||
/// Blocks SIGINT and SIGTERM in the process signal mask and delivers
|
||||
/// them as readable events on the fd instead.
|
||||
pub struct SignalFd {
|
||||
fd: RawFd,
|
||||
}
|
||||
|
||||
impl SignalFd {
|
||||
/// Create a signalfd that catches SIGINT and SIGTERM.
|
||||
///
|
||||
/// Also blocks these signals in the process signal mask so they
|
||||
/// don't trigger default handlers.
|
||||
pub fn new() -> io::Result<Self> {
|
||||
let mut mask: libc::sigset_t = unsafe { std::mem::zeroed() };
|
||||
unsafe {
|
||||
libc::sigemptyset(&mut mask);
|
||||
libc::sigaddset(&mut mask, libc::SIGINT);
|
||||
libc::sigaddset(&mut mask, libc::SIGTERM);
|
||||
|
||||
// Block these signals so they go to signalfd instead.
|
||||
let ret = libc::pthread_sigmask(libc::SIG_BLOCK, &mask, std::ptr::null_mut());
|
||||
if ret != 0 {
|
||||
return Err(io::Error::from_raw_os_error(ret));
|
||||
}
|
||||
}
|
||||
|
||||
let fd = unsafe { libc::signalfd(-1, &mask, libc::SFD_NONBLOCK | libc::SFD_CLOEXEC) };
|
||||
if fd < 0 {
|
||||
return Err(io::Error::last_os_error());
|
||||
}
|
||||
|
||||
Ok(Self { fd })
|
||||
}
|
||||
|
||||
/// Read a pending signal. Returns the signal number (e.g., SIGINT = 2).
|
||||
pub fn read(&self) -> io::Result<i32> {
|
||||
let mut info: libc::signalfd_siginfo = unsafe { std::mem::zeroed() };
|
||||
let size = std::mem::size_of::<libc::signalfd_siginfo>();
|
||||
let ret = unsafe {
|
||||
libc::read(self.fd, &mut info as *mut _ as *mut libc::c_void, size)
|
||||
};
|
||||
if ret < 0 {
|
||||
Err(io::Error::last_os_error())
|
||||
} else {
|
||||
Ok(info.ssi_signo as i32)
|
||||
}
|
||||
}
|
||||
|
||||
pub fn fd(&self) -> RawFd {
|
||||
self.fd
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for SignalFd {
|
||||
fn drop(&mut self) {
|
||||
unsafe { libc::close(self.fd) };
|
||||
}
|
||||
}
|
||||
91
crates/wo-event/src/timerfd.rs
Normal file
91
crates/wo-event/src/timerfd.rs
Normal file
|
|
@ -0,0 +1,91 @@
|
|||
use std::io;
|
||||
use std::os::unix::io::RawFd;
|
||||
use std::time::Duration;
|
||||
|
||||
/// Wrapper around Linux timerfd_create(2) for timer-as-fd.
|
||||
pub struct TimerFd {
|
||||
fd: RawFd,
|
||||
}
|
||||
|
||||
impl TimerFd {
|
||||
/// Create a new monotonic timer fd.
|
||||
pub fn new() -> io::Result<Self> {
|
||||
let fd = unsafe {
|
||||
libc::timerfd_create(
|
||||
libc::CLOCK_MONOTONIC,
|
||||
libc::TFD_NONBLOCK | libc::TFD_CLOEXEC,
|
||||
)
|
||||
};
|
||||
if fd < 0 {
|
||||
return Err(io::Error::last_os_error());
|
||||
}
|
||||
Ok(Self { fd })
|
||||
}
|
||||
|
||||
/// Arm the timer to fire once after `initial` and then repeat every `interval`.
|
||||
///
|
||||
/// Pass `Duration::ZERO` for `interval` for a one-shot timer.
|
||||
pub fn set(&self, initial: Duration, interval: Duration) -> io::Result<()> {
|
||||
let spec = libc::itimerspec {
|
||||
it_interval: duration_to_timespec(interval),
|
||||
it_value: duration_to_timespec(initial),
|
||||
};
|
||||
let ret =
|
||||
unsafe { libc::timerfd_settime(self.fd, 0, &spec, std::ptr::null_mut()) };
|
||||
if ret < 0 {
|
||||
Err(io::Error::last_os_error())
|
||||
} else {
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
/// Read the number of expirations since last read.
|
||||
pub fn read(&self) -> io::Result<u64> {
|
||||
let mut buf = [0u8; 8];
|
||||
let ret = unsafe { libc::read(self.fd, buf.as_mut_ptr() as *mut libc::c_void, 8) };
|
||||
if ret < 0 {
|
||||
Err(io::Error::last_os_error())
|
||||
} else {
|
||||
Ok(u64::from_ne_bytes(buf))
|
||||
}
|
||||
}
|
||||
|
||||
pub fn fd(&self) -> RawFd {
|
||||
self.fd
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for TimerFd {
|
||||
fn drop(&mut self) {
|
||||
unsafe { libc::close(self.fd) };
|
||||
}
|
||||
}
|
||||
|
||||
fn duration_to_timespec(d: Duration) -> libc::timespec {
|
||||
libc::timespec {
|
||||
tv_sec: d.as_secs() as libc::time_t,
|
||||
tv_nsec: d.subsec_nanos() as libc::c_long,
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::{EventLoop, Interest};
|
||||
|
||||
#[test]
|
||||
fn timer_fires() {
|
||||
let eloop = EventLoop::new().unwrap();
|
||||
let timer = TimerFd::new().unwrap();
|
||||
|
||||
timer.set(Duration::from_millis(20), Duration::ZERO).unwrap();
|
||||
eloop.register(timer.fd(), Interest::Readable, 99).unwrap();
|
||||
|
||||
let events = eloop.poll(Some(Duration::from_millis(100))).unwrap();
|
||||
assert!(!events.is_empty());
|
||||
assert_eq!(events[0].token, 99);
|
||||
|
||||
let expirations = timer.read().unwrap();
|
||||
assert!(expirations >= 1);
|
||||
}
|
||||
}
|
||||
13
crates/wo-htmlx/Cargo.toml
Normal file
13
crates/wo-htmlx/Cargo.toml
Normal file
|
|
@ -0,0 +1,13 @@
|
|||
[package]
|
||||
name = "wo-htmlx"
|
||||
version = "0.1.0"
|
||||
edition = "2021"
|
||||
|
||||
[dependencies]
|
||||
wo-model = { path = "../wo-model" }
|
||||
wo-md = { path = "../wo-md" }
|
||||
serde = { workspace = true }
|
||||
serde_json = { workspace = true }
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile = "3"
|
||||
18
crates/wo-htmlx/src/ast.rs
Normal file
18
crates/wo-htmlx/src/ast.rs
Normal file
|
|
@ -0,0 +1,18 @@
|
|||
/// A parsed template node.
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub enum Node {
|
||||
/// Raw HTML passthrough.
|
||||
Literal(String),
|
||||
/// Variable binding: `{{path.to.value}}`
|
||||
Binding(Vec<String>),
|
||||
/// Iteration: `{{#each path}} ... {{/each}}`
|
||||
Each {
|
||||
path: Vec<String>,
|
||||
body: Vec<Node>,
|
||||
},
|
||||
/// Partial inclusion: `{{> name arg=value}}`
|
||||
Partial {
|
||||
name: String,
|
||||
args: Vec<(String, Vec<String>)>,
|
||||
},
|
||||
}
|
||||
11
crates/wo-htmlx/src/lib.rs
Normal file
11
crates/wo-htmlx/src/lib.rs
Normal file
|
|
@ -0,0 +1,11 @@
|
|||
mod ast;
|
||||
mod parser;
|
||||
mod render;
|
||||
mod registry;
|
||||
mod value;
|
||||
|
||||
pub use ast::Node;
|
||||
pub use parser::parse;
|
||||
pub use render::render;
|
||||
pub use registry::TemplateRegistry;
|
||||
pub use value::Value;
|
||||
173
crates/wo-htmlx/src/parser.rs
Normal file
173
crates/wo-htmlx/src/parser.rs
Normal file
|
|
@ -0,0 +1,173 @@
|
|||
use crate::ast::Node;
|
||||
|
||||
/// Parse a template string into a list of AST nodes.
|
||||
pub fn parse(template: &str) -> Vec<Node> {
|
||||
let mut nodes = Vec::new();
|
||||
let mut rest = template;
|
||||
|
||||
while !rest.is_empty() {
|
||||
if let Some(pos) = rest.find("{{") {
|
||||
// Literal before the opening `{{`.
|
||||
if pos > 0 {
|
||||
nodes.push(Node::Literal(rest[..pos].to_string()));
|
||||
}
|
||||
|
||||
let after_open = &rest[pos + 2..];
|
||||
|
||||
if let Some(close) = after_open.find("}}") {
|
||||
let expr = after_open[..close].trim();
|
||||
rest = &after_open[close + 2..];
|
||||
|
||||
if let Some(each_path) = expr.strip_prefix("#each ") {
|
||||
// Block: {{#each path}} ... {{/each}}
|
||||
let path = parse_path(each_path.trim());
|
||||
let (body, remaining) = parse_until_end_each(rest);
|
||||
nodes.push(Node::Each { path, body });
|
||||
rest = remaining;
|
||||
} else if let Some(partial_expr) = expr.strip_prefix("> ") {
|
||||
// Partial: {{> name arg=value}}
|
||||
let (name, args) = parse_partial_expr(partial_expr.trim());
|
||||
nodes.push(Node::Partial { name, args });
|
||||
} else if !expr.starts_with('/') {
|
||||
// Binding: {{path.to.value}}
|
||||
let path = parse_path(expr);
|
||||
nodes.push(Node::Binding(path));
|
||||
}
|
||||
// {{/each}} handled by parse_until_end_each
|
||||
} else {
|
||||
// No closing `}}` — treat rest as literal.
|
||||
nodes.push(Node::Literal(rest.to_string()));
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
// No more `{{` — rest is literal.
|
||||
nodes.push(Node::Literal(rest.to_string()));
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
nodes
|
||||
}
|
||||
|
||||
fn parse_path(s: &str) -> Vec<String> {
|
||||
s.split('.').map(|p| p.trim().to_string()).collect()
|
||||
}
|
||||
|
||||
fn parse_until_end_each(input: &str) -> (Vec<Node>, &str) {
|
||||
// Find the matching {{/each}}.
|
||||
let mut depth = 1;
|
||||
let mut pos = 0;
|
||||
let bytes = input.as_bytes();
|
||||
|
||||
while pos < bytes.len() {
|
||||
if let Some(open) = input[pos..].find("{{") {
|
||||
let abs = pos + open;
|
||||
let after = &input[abs + 2..];
|
||||
if let Some(close) = after.find("}}") {
|
||||
let expr = after[..close].trim();
|
||||
if expr.starts_with("#each ") {
|
||||
depth += 1;
|
||||
} else if expr == "/each" {
|
||||
depth -= 1;
|
||||
if depth == 0 {
|
||||
let body_str = &input[..abs];
|
||||
let rest = &after[close + 2..];
|
||||
return (parse(body_str), rest);
|
||||
}
|
||||
}
|
||||
pos = abs + 2 + close + 2;
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Unmatched {{#each}} — return rest as literal body.
|
||||
(vec![Node::Literal(input.to_string())], "")
|
||||
}
|
||||
|
||||
fn parse_partial_expr(expr: &str) -> (String, Vec<(String, Vec<String>)>) {
|
||||
let parts: Vec<&str> = expr.splitn(2, ' ').collect();
|
||||
let name = parts[0].to_string();
|
||||
let mut args = Vec::new();
|
||||
|
||||
if parts.len() > 1 {
|
||||
for arg in parts[1].split_whitespace() {
|
||||
if let Some((key, val)) = arg.split_once('=') {
|
||||
args.push((key.to_string(), parse_path(val)));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
(name, args)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::ast::Node;
|
||||
|
||||
#[test]
|
||||
fn parse_literal() {
|
||||
let nodes = parse("<h1>Hello</h1>");
|
||||
assert_eq!(nodes, vec![Node::Literal("<h1>Hello</h1>".into())]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_binding() {
|
||||
let nodes = parse("{{article.title}}");
|
||||
assert_eq!(nodes, vec![Node::Binding(vec!["article".into(), "title".into()])]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_mixed() {
|
||||
let nodes = parse("<h1>{{title}}</h1>");
|
||||
assert_eq!(nodes.len(), 3);
|
||||
assert_eq!(nodes[0], Node::Literal("<h1>".into()));
|
||||
assert_eq!(nodes[1], Node::Binding(vec!["title".into()]));
|
||||
assert_eq!(nodes[2], Node::Literal("</h1>".into()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_each() {
|
||||
let nodes = parse("{{#each items}}<li>{{this}}</li>{{/each}}");
|
||||
assert_eq!(nodes.len(), 1);
|
||||
match &nodes[0] {
|
||||
Node::Each { path, body } => {
|
||||
assert_eq!(path, &vec!["items".to_string()]);
|
||||
assert_eq!(body.len(), 3);
|
||||
}
|
||||
_ => panic!("expected Each"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_partial() {
|
||||
let nodes = parse("{{> article-card article=this}}");
|
||||
assert_eq!(nodes.len(), 1);
|
||||
match &nodes[0] {
|
||||
Node::Partial { name, args } => {
|
||||
assert_eq!(name, "article-card");
|
||||
assert_eq!(args.len(), 1);
|
||||
assert_eq!(args[0].0, "article");
|
||||
assert_eq!(args[0].1, vec!["this".to_string()]);
|
||||
}
|
||||
_ => panic!("expected Partial"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_nested_each() {
|
||||
let nodes = parse("{{#each sections}}{{#each paragraphs}}{{this}}{{/each}}{{/each}}");
|
||||
assert_eq!(nodes.len(), 1);
|
||||
match &nodes[0] {
|
||||
Node::Each { body, .. } => {
|
||||
assert_eq!(body.len(), 1);
|
||||
assert!(matches!(&body[0], Node::Each { .. }));
|
||||
}
|
||||
_ => panic!("expected nested Each"),
|
||||
}
|
||||
}
|
||||
}
|
||||
121
crates/wo-htmlx/src/registry.rs
Normal file
121
crates/wo-htmlx/src/registry.rs
Normal file
|
|
@ -0,0 +1,121 @@
|
|||
use std::collections::HashMap;
|
||||
use std::fs;
|
||||
use std::io;
|
||||
use std::path::Path;
|
||||
|
||||
use crate::ast::Node;
|
||||
use crate::parser;
|
||||
|
||||
/// Holds parsed templates and partials, loaded from a templates directory.
|
||||
pub struct TemplateRegistry {
|
||||
/// Page templates (e.g., "home" -> parsed nodes from home.htmlx).
|
||||
templates: HashMap<String, Vec<Node>>,
|
||||
/// Component partials (e.g., "article-card" -> parsed nodes).
|
||||
partials: HashMap<String, Vec<Node>>,
|
||||
}
|
||||
|
||||
impl TemplateRegistry {
|
||||
/// Load all .htmlx files from the given templates directory.
|
||||
///
|
||||
/// Templates in the root are page templates.
|
||||
/// Templates in `components/` are partials.
|
||||
pub fn load(templates_dir: &Path) -> io::Result<Self> {
|
||||
let mut templates = HashMap::new();
|
||||
let mut partials = HashMap::new();
|
||||
|
||||
// Load page templates from root.
|
||||
if templates_dir.exists() {
|
||||
for entry in fs::read_dir(templates_dir)? {
|
||||
let entry = entry?;
|
||||
let path = entry.path();
|
||||
if path.is_file() && has_htmlx_ext(&path) {
|
||||
let name = stem(&path);
|
||||
let content = fs::read_to_string(&path)?;
|
||||
templates.insert(name, parser::parse(&content));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Load partials from components/.
|
||||
let components_dir = templates_dir.join("components");
|
||||
if components_dir.exists() {
|
||||
for entry in fs::read_dir(&components_dir)? {
|
||||
let entry = entry?;
|
||||
let path = entry.path();
|
||||
if path.is_file() && has_htmlx_ext(&path) {
|
||||
let name = stem(&path);
|
||||
let content = fs::read_to_string(&path)?;
|
||||
partials.insert(name, parser::parse(&content));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Ok(Self { templates, partials })
|
||||
}
|
||||
|
||||
/// Get a page template by name.
|
||||
pub fn get(&self, name: &str) -> Option<&Vec<Node>> {
|
||||
self.templates.get(name)
|
||||
}
|
||||
|
||||
/// Get the partials map (for passing to render()).
|
||||
pub fn partials(&self) -> &HashMap<String, Vec<Node>> {
|
||||
&self.partials
|
||||
}
|
||||
|
||||
/// Number of loaded templates.
|
||||
pub fn template_count(&self) -> usize {
|
||||
self.templates.len()
|
||||
}
|
||||
|
||||
/// Number of loaded partials.
|
||||
pub fn partial_count(&self) -> usize {
|
||||
self.partials.len()
|
||||
}
|
||||
}
|
||||
|
||||
fn has_htmlx_ext(path: &Path) -> bool {
|
||||
path.extension().and_then(|e| e.to_str()) == Some("htmlx")
|
||||
}
|
||||
|
||||
fn stem(path: &Path) -> String {
|
||||
path.file_stem()
|
||||
.and_then(|s| s.to_str())
|
||||
.unwrap_or("")
|
||||
.to_string()
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn load_templates() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let dir = tmp.path();
|
||||
|
||||
fs::write(dir.join("home.htmlx"), "<h1>{{title}}</h1>").unwrap();
|
||||
fs::write(dir.join("article.htmlx"), "<article>{{body}}</article>").unwrap();
|
||||
|
||||
fs::create_dir_all(dir.join("components")).unwrap();
|
||||
fs::write(
|
||||
dir.join("components/card.htmlx"),
|
||||
"<div>{{name}}</div>",
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let reg = TemplateRegistry::load(dir).unwrap();
|
||||
assert_eq!(reg.template_count(), 2);
|
||||
assert_eq!(reg.partial_count(), 1);
|
||||
assert!(reg.get("home").is_some());
|
||||
assert!(reg.get("article").is_some());
|
||||
assert!(reg.partials().contains_key("card"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn empty_directory() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let reg = TemplateRegistry::load(tmp.path()).unwrap();
|
||||
assert_eq!(reg.template_count(), 0);
|
||||
}
|
||||
}
|
||||
140
crates/wo-htmlx/src/render.rs
Normal file
140
crates/wo-htmlx/src/render.rs
Normal file
|
|
@ -0,0 +1,140 @@
|
|||
use std::collections::BTreeMap;
|
||||
|
||||
use crate::ast::Node;
|
||||
use crate::value::Value;
|
||||
|
||||
/// Render a list of AST nodes against a context value.
|
||||
///
|
||||
/// `partials` maps partial names to their parsed node lists.
|
||||
pub fn render(
|
||||
nodes: &[Node],
|
||||
context: &Value,
|
||||
partials: &std::collections::HashMap<String, Vec<Node>>,
|
||||
) -> String {
|
||||
let mut output = String::new();
|
||||
|
||||
for node in nodes {
|
||||
match node {
|
||||
Node::Literal(text) => {
|
||||
output.push_str(text);
|
||||
}
|
||||
Node::Binding(path) => {
|
||||
if path.len() == 1 && path[0] == "this" {
|
||||
output.push_str(&context.to_display());
|
||||
} else {
|
||||
let val = context.resolve(path);
|
||||
output.push_str(&val.to_display());
|
||||
}
|
||||
}
|
||||
Node::Each { path, body } => {
|
||||
let list_val = if path.len() == 1 && path[0] == "this" {
|
||||
context
|
||||
} else {
|
||||
context.resolve(path)
|
||||
};
|
||||
|
||||
for item in list_val.as_list() {
|
||||
output.push_str(&render(body, item, partials));
|
||||
}
|
||||
}
|
||||
Node::Partial { name, args } => {
|
||||
if let Some(partial_nodes) = partials.get(name.as_str()) {
|
||||
// Build the partial context from args.
|
||||
let partial_ctx = if args.is_empty() {
|
||||
context.clone()
|
||||
} else {
|
||||
let mut map = BTreeMap::new();
|
||||
for (key, path) in args {
|
||||
let val = if path.len() == 1 && path[0] == "this" {
|
||||
context.clone()
|
||||
} else {
|
||||
context.resolve(path).clone()
|
||||
};
|
||||
map.insert(key.clone(), val);
|
||||
}
|
||||
Value::Object(map)
|
||||
};
|
||||
output.push_str(&render(partial_nodes, &partial_ctx, partials));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
output
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::parser;
|
||||
use std::collections::HashMap;
|
||||
|
||||
fn ctx(json: serde_json::Value) -> Value {
|
||||
Value::from_json(&json)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn render_literal() {
|
||||
let nodes = parser::parse("<h1>Hello</h1>");
|
||||
let result = render(&nodes, &Value::Null, &HashMap::new());
|
||||
assert_eq!(result, "<h1>Hello</h1>");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn render_binding() {
|
||||
let nodes = parser::parse("<h1>{{title}}</h1>");
|
||||
let context = ctx(serde_json::json!({"title": "My Article"}));
|
||||
let result = render(&nodes, &context, &HashMap::new());
|
||||
assert_eq!(result, "<h1>My Article</h1>");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn render_nested_binding() {
|
||||
let nodes = parser::parse("by {{article.author}}");
|
||||
let context = ctx(serde_json::json!({"article": {"author": "Shoney"}}));
|
||||
let result = render(&nodes, &context, &HashMap::new());
|
||||
assert_eq!(result, "by Shoney");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn render_each() {
|
||||
let nodes = parser::parse("<ul>{{#each tags}}<li>{{this}}</li>{{/each}}</ul>");
|
||||
let context = ctx(serde_json::json!({"tags": ["rust", "linux"]}));
|
||||
let result = render(&nodes, &context, &HashMap::new());
|
||||
assert_eq!(result, "<ul><li>rust</li><li>linux</li></ul>");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn render_each_objects() {
|
||||
let nodes = parser::parse("{{#each sections}}<h2>{{heading}}</h2>{{/each}}");
|
||||
let context = ctx(serde_json::json!({
|
||||
"sections": [
|
||||
{"heading": "Intro"},
|
||||
{"heading": "Details"}
|
||||
]
|
||||
}));
|
||||
let result = render(&nodes, &context, &HashMap::new());
|
||||
assert_eq!(result, "<h2>Intro</h2><h2>Details</h2>");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn render_partial() {
|
||||
let nodes = parser::parse("{{> card item=this}}");
|
||||
let mut partials = HashMap::new();
|
||||
partials.insert(
|
||||
"card".to_string(),
|
||||
parser::parse("<div>{{item.name}}</div>"),
|
||||
);
|
||||
let context = ctx(serde_json::json!({"name": "Test"}));
|
||||
let result = render(&nodes, &context, &partials);
|
||||
assert_eq!(result, "<div>Test</div>");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn render_missing_binding() {
|
||||
let nodes = parser::parse("{{nonexistent}}");
|
||||
let context = ctx(serde_json::json!({}));
|
||||
let result = render(&nodes, &context, &HashMap::new());
|
||||
assert_eq!(result, "");
|
||||
}
|
||||
}
|
||||
122
crates/wo-htmlx/src/value.rs
Normal file
122
crates/wo-htmlx/src/value.rs
Normal file
|
|
@ -0,0 +1,122 @@
|
|||
use std::collections::BTreeMap;
|
||||
|
||||
/// Template context value — the data that template bindings resolve against.
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub enum Value {
|
||||
String(String),
|
||||
Number(f64),
|
||||
Bool(bool),
|
||||
List(Vec<Value>),
|
||||
Object(BTreeMap<String, Value>),
|
||||
Null,
|
||||
}
|
||||
|
||||
impl Value {
|
||||
/// Resolve a dotted path like ["article", "title"] against this value.
|
||||
pub fn resolve(&self, path: &[String]) -> &Value {
|
||||
let mut current = self;
|
||||
for key in path {
|
||||
match current {
|
||||
Value::Object(map) => {
|
||||
current = map.get(key.as_str()).unwrap_or(&Value::Null);
|
||||
}
|
||||
_ => return &Value::Null,
|
||||
}
|
||||
}
|
||||
current
|
||||
}
|
||||
|
||||
/// Convert to a display string for template output.
|
||||
pub fn to_display(&self) -> String {
|
||||
match self {
|
||||
Value::String(s) => s.clone(),
|
||||
Value::Number(n) => {
|
||||
if *n == (*n as i64) as f64 {
|
||||
format!("{}", *n as i64)
|
||||
} else {
|
||||
format!("{}", n)
|
||||
}
|
||||
}
|
||||
Value::Bool(b) => b.to_string(),
|
||||
Value::Null => String::new(),
|
||||
Value::List(items) => {
|
||||
let parts: Vec<String> = items.iter().map(|v| v.to_display()).collect();
|
||||
parts.join(", ")
|
||||
}
|
||||
Value::Object(_) => "[object]".to_string(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Convert to a list for iteration (returns empty vec if not a list).
|
||||
pub fn as_list(&self) -> &[Value] {
|
||||
match self {
|
||||
Value::List(items) => items,
|
||||
_ => &[],
|
||||
}
|
||||
}
|
||||
|
||||
/// Convert a serde_json::Value to a template Value.
|
||||
pub fn from_json(json: &serde_json::Value) -> Self {
|
||||
match json {
|
||||
serde_json::Value::Null => Value::Null,
|
||||
serde_json::Value::Bool(b) => Value::Bool(*b),
|
||||
serde_json::Value::Number(n) => Value::Number(n.as_f64().unwrap_or(0.0)),
|
||||
serde_json::Value::String(s) => Value::String(s.clone()),
|
||||
serde_json::Value::Array(arr) => {
|
||||
Value::List(arr.iter().map(Value::from_json).collect())
|
||||
}
|
||||
serde_json::Value::Object(map) => {
|
||||
let btree = map
|
||||
.iter()
|
||||
.map(|(k, v)| (k.clone(), Value::from_json(v)))
|
||||
.collect();
|
||||
Value::Object(btree)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Convert an Article to a template Value using serde_json as intermediary.
|
||||
pub fn article_to_value(article: &wo_model::Article) -> Value {
|
||||
let json = serde_json::to_value(article).unwrap_or(serde_json::Value::Null);
|
||||
Value::from_json(&json)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn resolve_path() {
|
||||
let mut inner = BTreeMap::new();
|
||||
inner.insert("title".into(), Value::String("Hello".into()));
|
||||
let mut root = BTreeMap::new();
|
||||
root.insert("article".into(), Value::Object(inner));
|
||||
let val = Value::Object(root);
|
||||
|
||||
let result = val.resolve(&["article".into(), "title".into()]);
|
||||
assert_eq!(result, &Value::String("Hello".into()));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn resolve_missing() {
|
||||
let val = Value::Object(BTreeMap::new());
|
||||
assert_eq!(val.resolve(&["nope".into()]), &Value::Null);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn from_json() {
|
||||
let json: serde_json::Value = serde_json::json!({
|
||||
"name": "test",
|
||||
"count": 42,
|
||||
"active": true,
|
||||
"tags": ["a", "b"]
|
||||
});
|
||||
let val = Value::from_json(&json);
|
||||
assert_eq!(
|
||||
val.resolve(&["name".into()]),
|
||||
&Value::String("test".into())
|
||||
);
|
||||
assert_eq!(val.resolve(&["count".into()]), &Value::Number(42.0));
|
||||
}
|
||||
}
|
||||
10
crates/wo-http/Cargo.toml
Normal file
10
crates/wo-http/Cargo.toml
Normal file
|
|
@ -0,0 +1,10 @@
|
|||
[package]
|
||||
name = "wo-http"
|
||||
version = "0.1.0"
|
||||
edition = "2021"
|
||||
|
||||
[dependencies]
|
||||
libc = { workspace = true }
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile = "3"
|
||||
202
crates/wo-http/src/connection.rs
Normal file
202
crates/wo-http/src/connection.rs
Normal file
|
|
@ -0,0 +1,202 @@
|
|||
use std::io;
|
||||
use std::os::unix::io::RawFd;
|
||||
|
||||
use crate::request::{self, ParseResult, Request};
|
||||
use crate::response::Response;
|
||||
|
||||
/// Connection state machine.
|
||||
#[derive(Debug, Clone, Copy, PartialEq)]
|
||||
pub enum State {
|
||||
/// Accumulating request bytes.
|
||||
ReadingRequest,
|
||||
/// Writing response bytes.
|
||||
WritingResponse,
|
||||
/// Initial response sent, fd handed to subscription manager.
|
||||
/// Connection stays open for diff payloads.
|
||||
Subscribed,
|
||||
/// Ready to close.
|
||||
Done,
|
||||
}
|
||||
|
||||
/// A single HTTP connection.
|
||||
pub struct Connection {
|
||||
pub fd: RawFd,
|
||||
pub state: State,
|
||||
read_buf: Vec<u8>,
|
||||
write_buf: Vec<u8>,
|
||||
write_offset: usize,
|
||||
parsed_request: Option<Request>,
|
||||
}
|
||||
|
||||
impl Connection {
|
||||
pub fn new(fd: RawFd) -> Self {
|
||||
Self {
|
||||
fd,
|
||||
state: State::ReadingRequest,
|
||||
read_buf: Vec::with_capacity(4096),
|
||||
write_buf: Vec::new(),
|
||||
write_offset: 0,
|
||||
parsed_request: None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Read available data from the socket into the read buffer.
|
||||
/// Returns Ok(true) if data was read, Ok(false) if connection closed.
|
||||
pub fn read(&mut self) -> io::Result<bool> {
|
||||
let mut buf = [0u8; 4096];
|
||||
let n = unsafe {
|
||||
libc::read(self.fd, buf.as_mut_ptr() as *mut libc::c_void, buf.len())
|
||||
};
|
||||
if n < 0 {
|
||||
let err = io::Error::last_os_error();
|
||||
if err.raw_os_error() == Some(libc::EAGAIN) {
|
||||
return Ok(true); // no data yet, but connection still open
|
||||
}
|
||||
return Err(err);
|
||||
}
|
||||
if n == 0 {
|
||||
return Ok(false); // connection closed
|
||||
}
|
||||
self.read_buf.extend_from_slice(&buf[..n as usize]);
|
||||
Ok(true)
|
||||
}
|
||||
|
||||
/// Try to parse the request from the read buffer.
|
||||
/// Returns Some(Request) if a full request was parsed.
|
||||
pub fn try_parse(&mut self) -> Option<Request> {
|
||||
if self.parsed_request.is_some() {
|
||||
return self.parsed_request.clone();
|
||||
}
|
||||
match request::parse(&self.read_buf) {
|
||||
ParseResult::Complete(req, _consumed) => {
|
||||
self.parsed_request = Some(req.clone());
|
||||
Some(req)
|
||||
}
|
||||
ParseResult::Incomplete => None,
|
||||
ParseResult::Error(_) => {
|
||||
self.state = State::Done;
|
||||
None
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Set the response to write back to the client.
|
||||
pub fn set_response(&mut self, response: &Response) {
|
||||
self.write_buf = response.to_bytes();
|
||||
self.write_offset = 0;
|
||||
self.state = State::WritingResponse;
|
||||
}
|
||||
|
||||
/// Write pending response bytes to the socket.
|
||||
/// Returns Ok(true) if all bytes written, Ok(false) if partial (need EPOLLOUT).
|
||||
pub fn write(&mut self) -> io::Result<bool> {
|
||||
let remaining = &self.write_buf[self.write_offset..];
|
||||
if remaining.is_empty() {
|
||||
return Ok(true);
|
||||
}
|
||||
|
||||
let n = unsafe {
|
||||
libc::write(
|
||||
self.fd,
|
||||
remaining.as_ptr() as *const libc::c_void,
|
||||
remaining.len(),
|
||||
)
|
||||
};
|
||||
if n < 0 {
|
||||
let err = io::Error::last_os_error();
|
||||
if err.raw_os_error() == Some(libc::EAGAIN) {
|
||||
return Ok(false);
|
||||
}
|
||||
return Err(err);
|
||||
}
|
||||
self.write_offset += n as usize;
|
||||
|
||||
if self.write_offset >= self.write_buf.len() {
|
||||
Ok(true) // all written
|
||||
} else {
|
||||
Ok(false) // partial write
|
||||
}
|
||||
}
|
||||
|
||||
/// Transition to subscribed state (connection stays open).
|
||||
pub fn set_subscribed(&mut self) {
|
||||
self.state = State::Subscribed;
|
||||
}
|
||||
|
||||
/// Mark connection as done (will be closed).
|
||||
pub fn set_done(&mut self) {
|
||||
self.state = State::Done;
|
||||
}
|
||||
|
||||
/// Take the parsed request out of the connection.
|
||||
pub fn take_request(&mut self) -> Option<Request> {
|
||||
self.parsed_request.take()
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for Connection {
|
||||
fn drop(&mut self) {
|
||||
unsafe { libc::close(self.fd) };
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn connection_lifecycle() {
|
||||
// Use a socketpair to simulate a connection.
|
||||
let mut fds = [0i32; 2];
|
||||
assert_eq!(
|
||||
unsafe { libc::socketpair(libc::AF_UNIX, libc::SOCK_STREAM | libc::SOCK_NONBLOCK, 0, fds.as_mut_ptr()) },
|
||||
0
|
||||
);
|
||||
let (server_fd, client_fd) = (fds[0], fds[1]);
|
||||
|
||||
// Write a request from the "client" side.
|
||||
let request_bytes = b"GET /blog/test HTTP/1.1\r\nHost: localhost\r\n\r\n";
|
||||
unsafe {
|
||||
libc::write(
|
||||
client_fd,
|
||||
request_bytes.as_ptr() as *const libc::c_void,
|
||||
request_bytes.len(),
|
||||
);
|
||||
}
|
||||
|
||||
// Create connection on the "server" side.
|
||||
// We need to manage the fd manually since Connection will close it on drop.
|
||||
let dup_fd = unsafe { libc::dup(server_fd) };
|
||||
let mut conn = Connection::new(dup_fd);
|
||||
assert_eq!(conn.state, State::ReadingRequest);
|
||||
|
||||
// Read and parse.
|
||||
assert!(conn.read().unwrap());
|
||||
let req = conn.try_parse().unwrap();
|
||||
assert_eq!(req.path, "/blog/test");
|
||||
|
||||
// Set response.
|
||||
let response = Response::html("<h1>Test</h1>".into());
|
||||
conn.set_response(&response);
|
||||
assert_eq!(conn.state, State::WritingResponse);
|
||||
|
||||
// Write response.
|
||||
let complete = conn.write().unwrap();
|
||||
assert!(complete);
|
||||
|
||||
// Read from client side to verify.
|
||||
let mut buf = [0u8; 4096];
|
||||
let n = unsafe {
|
||||
libc::read(client_fd, buf.as_mut_ptr() as *mut libc::c_void, buf.len())
|
||||
};
|
||||
assert!(n > 0);
|
||||
let response_str = std::str::from_utf8(&buf[..n as usize]).unwrap();
|
||||
assert!(response_str.contains("HTTP/1.1 200 OK"));
|
||||
assert!(response_str.contains("<h1>Test</h1>"));
|
||||
|
||||
unsafe {
|
||||
libc::close(server_fd);
|
||||
libc::close(client_fd);
|
||||
}
|
||||
}
|
||||
}
|
||||
4
crates/wo-http/src/lib.rs
Normal file
4
crates/wo-http/src/lib.rs
Normal file
|
|
@ -0,0 +1,4 @@
|
|||
pub mod listener;
|
||||
pub mod request;
|
||||
pub mod response;
|
||||
pub mod connection;
|
||||
202
crates/wo-http/src/listener.rs
Normal file
202
crates/wo-http/src/listener.rs
Normal file
|
|
@ -0,0 +1,202 @@
|
|||
use std::io;
|
||||
use std::os::unix::io::RawFd;
|
||||
|
||||
/// A non-blocking TCP listener wrapping raw socket syscalls.
|
||||
pub struct TcpListener {
|
||||
fd: RawFd,
|
||||
}
|
||||
|
||||
impl TcpListener {
|
||||
/// Bind to the given address (e.g., "0.0.0.0:3000") and start listening.
|
||||
pub fn bind(addr: &str) -> io::Result<Self> {
|
||||
let (ip, port) = parse_addr(addr)?;
|
||||
|
||||
let fd = unsafe {
|
||||
libc::socket(
|
||||
libc::AF_INET,
|
||||
libc::SOCK_STREAM | libc::SOCK_NONBLOCK | libc::SOCK_CLOEXEC,
|
||||
0,
|
||||
)
|
||||
};
|
||||
if fd < 0 {
|
||||
return Err(io::Error::last_os_error());
|
||||
}
|
||||
|
||||
// SO_REUSEADDR to allow quick restart.
|
||||
let optval: libc::c_int = 1;
|
||||
let ret = unsafe {
|
||||
libc::setsockopt(
|
||||
fd,
|
||||
libc::SOL_SOCKET,
|
||||
libc::SO_REUSEADDR,
|
||||
&optval as *const _ as *const libc::c_void,
|
||||
std::mem::size_of::<libc::c_int>() as libc::socklen_t,
|
||||
)
|
||||
};
|
||||
if ret < 0 {
|
||||
unsafe { libc::close(fd) };
|
||||
return Err(io::Error::last_os_error());
|
||||
}
|
||||
|
||||
let sockaddr = libc::sockaddr_in {
|
||||
sin_family: libc::AF_INET as libc::sa_family_t,
|
||||
sin_port: port.to_be(),
|
||||
sin_addr: libc::in_addr { s_addr: ip },
|
||||
sin_zero: [0; 8],
|
||||
};
|
||||
|
||||
let ret = unsafe {
|
||||
libc::bind(
|
||||
fd,
|
||||
&sockaddr as *const _ as *const libc::sockaddr,
|
||||
std::mem::size_of::<libc::sockaddr_in>() as libc::socklen_t,
|
||||
)
|
||||
};
|
||||
if ret < 0 {
|
||||
unsafe { libc::close(fd) };
|
||||
return Err(io::Error::last_os_error());
|
||||
}
|
||||
|
||||
let ret = unsafe { libc::listen(fd, 128) };
|
||||
if ret < 0 {
|
||||
unsafe { libc::close(fd) };
|
||||
return Err(io::Error::last_os_error());
|
||||
}
|
||||
|
||||
Ok(Self { fd })
|
||||
}
|
||||
|
||||
/// Accept a new connection. Returns the client fd or EAGAIN if none pending.
|
||||
pub fn accept(&self) -> io::Result<Option<RawFd>> {
|
||||
let client_fd = unsafe {
|
||||
libc::accept4(
|
||||
self.fd,
|
||||
std::ptr::null_mut(),
|
||||
std::ptr::null_mut(),
|
||||
libc::SOCK_NONBLOCK | libc::SOCK_CLOEXEC,
|
||||
)
|
||||
};
|
||||
if client_fd < 0 {
|
||||
let err = io::Error::last_os_error();
|
||||
if err.raw_os_error() == Some(libc::EAGAIN)
|
||||
|| err.raw_os_error() == Some(libc::EWOULDBLOCK)
|
||||
{
|
||||
return Ok(None);
|
||||
}
|
||||
return Err(err);
|
||||
}
|
||||
Ok(Some(client_fd))
|
||||
}
|
||||
|
||||
/// The listener file descriptor for epoll registration.
|
||||
pub fn fd(&self) -> RawFd {
|
||||
self.fd
|
||||
}
|
||||
|
||||
/// Get the actual bound address (useful when binding to port 0).
|
||||
pub fn local_addr(&self) -> io::Result<(String, u16)> {
|
||||
let mut addr: libc::sockaddr_in = unsafe { std::mem::zeroed() };
|
||||
let mut len = std::mem::size_of::<libc::sockaddr_in>() as libc::socklen_t;
|
||||
let ret = unsafe {
|
||||
libc::getsockname(
|
||||
self.fd,
|
||||
&mut addr as *mut _ as *mut libc::sockaddr,
|
||||
&mut len,
|
||||
)
|
||||
};
|
||||
if ret < 0 {
|
||||
return Err(io::Error::last_os_error());
|
||||
}
|
||||
let ip = u32::from_be(addr.sin_addr.s_addr);
|
||||
let port = u16::from_be(addr.sin_port);
|
||||
let ip_str = format!(
|
||||
"{}.{}.{}.{}",
|
||||
(ip >> 24) & 0xFF,
|
||||
(ip >> 16) & 0xFF,
|
||||
(ip >> 8) & 0xFF,
|
||||
ip & 0xFF,
|
||||
);
|
||||
Ok((ip_str, port))
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for TcpListener {
|
||||
fn drop(&mut self) {
|
||||
unsafe { libc::close(self.fd) };
|
||||
}
|
||||
}
|
||||
|
||||
fn parse_addr(addr: &str) -> io::Result<(u32, u16)> {
|
||||
let parts: Vec<&str> = addr.rsplitn(2, ':').collect();
|
||||
if parts.len() != 2 {
|
||||
return Err(io::Error::new(io::ErrorKind::InvalidInput, "expected host:port"));
|
||||
}
|
||||
let port: u16 = parts[0]
|
||||
.parse()
|
||||
.map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "invalid port"))?;
|
||||
let ip_str = parts[1];
|
||||
|
||||
let ip = if ip_str == "0.0.0.0" {
|
||||
0u32
|
||||
} else if ip_str == "127.0.0.1" {
|
||||
0x7F000001u32.to_be()
|
||||
} else {
|
||||
let octets: Vec<u8> = ip_str
|
||||
.split('.')
|
||||
.map(|s| s.parse().unwrap_or(0))
|
||||
.collect();
|
||||
if octets.len() != 4 {
|
||||
return Err(io::Error::new(io::ErrorKind::InvalidInput, "invalid IP"));
|
||||
}
|
||||
u32::from_ne_bytes([octets[0], octets[1], octets[2], octets[3]])
|
||||
};
|
||||
|
||||
Ok((ip, port))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn bind_and_accept() {
|
||||
let listener = TcpListener::bind("127.0.0.1:0").unwrap();
|
||||
let (_, port) = listener.local_addr().unwrap();
|
||||
assert!(port > 0);
|
||||
|
||||
// No pending connections → None.
|
||||
assert!(listener.accept().unwrap().is_none());
|
||||
|
||||
// Connect a client.
|
||||
let client_fd = unsafe {
|
||||
libc::socket(libc::AF_INET, libc::SOCK_STREAM, 0)
|
||||
};
|
||||
assert!(client_fd >= 0);
|
||||
|
||||
let addr = libc::sockaddr_in {
|
||||
sin_family: libc::AF_INET as libc::sa_family_t,
|
||||
sin_port: port.to_be(),
|
||||
sin_addr: libc::in_addr {
|
||||
s_addr: 0x7F000001u32.to_be(),
|
||||
},
|
||||
sin_zero: [0; 8],
|
||||
};
|
||||
let ret = unsafe {
|
||||
libc::connect(
|
||||
client_fd,
|
||||
&addr as *const _ as *const libc::sockaddr,
|
||||
std::mem::size_of::<libc::sockaddr_in>() as libc::socklen_t,
|
||||
)
|
||||
};
|
||||
assert_eq!(ret, 0);
|
||||
|
||||
// Now accept should return a fd.
|
||||
let accepted = listener.accept().unwrap();
|
||||
assert!(accepted.is_some());
|
||||
|
||||
unsafe {
|
||||
libc::close(client_fd);
|
||||
libc::close(accepted.unwrap());
|
||||
}
|
||||
}
|
||||
}
|
||||
158
crates/wo-http/src/request.rs
Normal file
158
crates/wo-http/src/request.rs
Normal file
|
|
@ -0,0 +1,158 @@
|
|||
use std::collections::HashMap;
|
||||
|
||||
/// HTTP method.
|
||||
#[derive(Debug, Clone, Copy, PartialEq)]
|
||||
pub enum Method {
|
||||
Get,
|
||||
Post,
|
||||
Head,
|
||||
Unknown,
|
||||
}
|
||||
|
||||
/// A parsed HTTP/1.1 request.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct Request {
|
||||
pub method: Method,
|
||||
pub path: String,
|
||||
pub query: Option<String>,
|
||||
pub headers: HashMap<String, String>,
|
||||
}
|
||||
|
||||
/// Result of attempting to parse a request from a byte buffer.
|
||||
pub enum ParseResult {
|
||||
/// Request fully parsed; returns the request and number of bytes consumed.
|
||||
Complete(Request, usize),
|
||||
/// Need more data.
|
||||
Incomplete,
|
||||
/// Malformed request.
|
||||
Error(String),
|
||||
}
|
||||
|
||||
/// Parse an HTTP/1.1 request from a byte buffer.
|
||||
///
|
||||
/// Returns `ParseResult::Complete` when the full header has been received
|
||||
/// (delimited by `\r\n\r\n`).
|
||||
pub fn parse(buf: &[u8]) -> ParseResult {
|
||||
// Find the end of headers.
|
||||
let header_end = match find_header_end(buf) {
|
||||
Some(pos) => pos,
|
||||
None => {
|
||||
if buf.len() > 8192 {
|
||||
return ParseResult::Error("request too large".into());
|
||||
}
|
||||
return ParseResult::Incomplete;
|
||||
}
|
||||
};
|
||||
|
||||
let header_str = match std::str::from_utf8(&buf[..header_end]) {
|
||||
Ok(s) => s,
|
||||
Err(_) => return ParseResult::Error("invalid UTF-8 in headers".into()),
|
||||
};
|
||||
|
||||
let mut lines = header_str.lines();
|
||||
|
||||
// Request line: "GET /path HTTP/1.1"
|
||||
let request_line = match lines.next() {
|
||||
Some(l) => l,
|
||||
None => return ParseResult::Error("empty request".into()),
|
||||
};
|
||||
|
||||
let mut parts = request_line.split_whitespace();
|
||||
let method = match parts.next() {
|
||||
Some("GET") => Method::Get,
|
||||
Some("POST") => Method::Post,
|
||||
Some("HEAD") => Method::Head,
|
||||
Some(_) => Method::Unknown,
|
||||
None => return ParseResult::Error("missing method".into()),
|
||||
};
|
||||
|
||||
let raw_path = match parts.next() {
|
||||
Some(p) => p,
|
||||
None => return ParseResult::Error("missing path".into()),
|
||||
};
|
||||
|
||||
// Split path and query string.
|
||||
let (path, query) = match raw_path.split_once('?') {
|
||||
Some((p, q)) => (p.to_string(), Some(q.to_string())),
|
||||
None => (raw_path.to_string(), None),
|
||||
};
|
||||
|
||||
// Parse headers.
|
||||
let mut headers = HashMap::new();
|
||||
for line in lines {
|
||||
if line.is_empty() {
|
||||
break;
|
||||
}
|
||||
if let Some((key, value)) = line.split_once(':') {
|
||||
headers.insert(
|
||||
key.trim().to_lowercase(),
|
||||
value.trim().to_string(),
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// Bytes consumed: header + \r\n\r\n delimiter.
|
||||
let consumed = header_end + 4;
|
||||
|
||||
ParseResult::Complete(Request { method, path, query, headers }, consumed)
|
||||
}
|
||||
|
||||
fn find_header_end(buf: &[u8]) -> Option<usize> {
|
||||
for i in 0..buf.len().saturating_sub(3) {
|
||||
if &buf[i..i + 4] == b"\r\n\r\n" {
|
||||
return Some(i);
|
||||
}
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn parse_simple_get() {
|
||||
let raw = b"GET /blog/linux-misc HTTP/1.1\r\nHost: localhost\r\n\r\n";
|
||||
match parse(raw) {
|
||||
ParseResult::Complete(req, consumed) => {
|
||||
assert_eq!(req.method, Method::Get);
|
||||
assert_eq!(req.path, "/blog/linux-misc");
|
||||
assert!(req.query.is_none());
|
||||
assert_eq!(req.headers.get("host").unwrap(), "localhost");
|
||||
assert_eq!(consumed, raw.len());
|
||||
}
|
||||
_ => panic!("expected Complete"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_with_query_string() {
|
||||
let raw = b"GET /tag/rust?page=2 HTTP/1.1\r\n\r\n";
|
||||
match parse(raw) {
|
||||
ParseResult::Complete(req, _) => {
|
||||
assert_eq!(req.path, "/tag/rust");
|
||||
assert_eq!(req.query.as_deref(), Some("page=2"));
|
||||
}
|
||||
_ => panic!("expected Complete"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_incomplete() {
|
||||
let raw = b"GET / HTTP/1.1\r\nHost: local";
|
||||
assert!(matches!(parse(raw), ParseResult::Incomplete));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_multiple_headers() {
|
||||
let raw = b"GET / HTTP/1.1\r\nHost: localhost\r\nAccept: text/html\r\nConnection: keep-alive\r\n\r\n";
|
||||
match parse(raw) {
|
||||
ParseResult::Complete(req, _) => {
|
||||
assert_eq!(req.headers.len(), 3);
|
||||
assert_eq!(req.headers.get("accept").unwrap(), "text/html");
|
||||
assert_eq!(req.headers.get("connection").unwrap(), "keep-alive");
|
||||
}
|
||||
_ => panic!("expected Complete"),
|
||||
}
|
||||
}
|
||||
}
|
||||
110
crates/wo-http/src/response.rs
Normal file
110
crates/wo-http/src/response.rs
Normal file
|
|
@ -0,0 +1,110 @@
|
|||
/// HTTP response builder.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct Response {
|
||||
pub status: u16,
|
||||
pub status_text: String,
|
||||
pub headers: Vec<(String, String)>,
|
||||
pub body: Vec<u8>,
|
||||
}
|
||||
|
||||
impl Response {
|
||||
pub fn new(status: u16, status_text: &str) -> Self {
|
||||
Self {
|
||||
status,
|
||||
status_text: status_text.to_string(),
|
||||
headers: vec![],
|
||||
body: vec![],
|
||||
}
|
||||
}
|
||||
|
||||
/// 200 OK with HTML body.
|
||||
pub fn html(body: String) -> Self {
|
||||
let mut r = Self::new(200, "OK");
|
||||
r.header("Content-Type", "text/html; charset=utf-8");
|
||||
r.header("Content-Length", &body.len().to_string());
|
||||
r.body = body.into_bytes();
|
||||
r
|
||||
}
|
||||
|
||||
/// 200 OK with a raw body and content type.
|
||||
pub fn ok(body: Vec<u8>, content_type: &str) -> Self {
|
||||
let mut r = Self::new(200, "OK");
|
||||
r.header("Content-Type", content_type);
|
||||
r.header("Content-Length", &body.len().to_string());
|
||||
r.body = body;
|
||||
r
|
||||
}
|
||||
|
||||
/// 404 Not Found.
|
||||
pub fn not_found() -> Self {
|
||||
let body = "<h1>404 Not Found</h1>";
|
||||
let mut r = Self::new(404, "Not Found");
|
||||
r.header("Content-Type", "text/html; charset=utf-8");
|
||||
r.header("Content-Length", &body.len().to_string());
|
||||
r.body = body.as_bytes().to_vec();
|
||||
r
|
||||
}
|
||||
|
||||
/// 500 Internal Server Error.
|
||||
pub fn internal_error(msg: &str) -> Self {
|
||||
let body = format!("<h1>500 Internal Server Error</h1><p>{}</p>", msg);
|
||||
let mut r = Self::new(500, "Internal Server Error");
|
||||
r.header("Content-Type", "text/html; charset=utf-8");
|
||||
r.header("Content-Length", &body.len().to_string());
|
||||
r.body = body.into_bytes();
|
||||
r
|
||||
}
|
||||
|
||||
/// Add a header.
|
||||
pub fn header(&mut self, key: &str, value: &str) -> &mut Self {
|
||||
self.headers.push((key.to_string(), value.to_string()));
|
||||
self
|
||||
}
|
||||
|
||||
/// Serialize the response to bytes for writing to a socket.
|
||||
pub fn to_bytes(&self) -> Vec<u8> {
|
||||
let mut buf = Vec::with_capacity(256 + self.body.len());
|
||||
|
||||
// Status line.
|
||||
buf.extend_from_slice(
|
||||
format!("HTTP/1.1 {} {}\r\n", self.status, self.status_text).as_bytes(),
|
||||
);
|
||||
|
||||
// Headers.
|
||||
for (key, value) in &self.headers {
|
||||
buf.extend_from_slice(format!("{}: {}\r\n", key, value).as_bytes());
|
||||
}
|
||||
|
||||
// End of headers.
|
||||
buf.extend_from_slice(b"\r\n");
|
||||
|
||||
// Body.
|
||||
buf.extend_from_slice(&self.body);
|
||||
|
||||
buf
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn html_response() {
|
||||
let r = Response::html("<h1>Hello</h1>".into());
|
||||
let bytes = r.to_bytes();
|
||||
let s = String::from_utf8(bytes).unwrap();
|
||||
assert!(s.starts_with("HTTP/1.1 200 OK\r\n"));
|
||||
assert!(s.contains("Content-Type: text/html"));
|
||||
assert!(s.contains("<h1>Hello</h1>"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn not_found_response() {
|
||||
let r = Response::not_found();
|
||||
assert_eq!(r.status, 404);
|
||||
let bytes = r.to_bytes();
|
||||
let s = String::from_utf8(bytes).unwrap();
|
||||
assert!(s.contains("404 Not Found"));
|
||||
}
|
||||
}
|
||||
13
crates/wo-index/Cargo.toml
Normal file
13
crates/wo-index/Cargo.toml
Normal file
|
|
@ -0,0 +1,13 @@
|
|||
[package]
|
||||
name = "wo-index"
|
||||
version = "0.1.0"
|
||||
edition = "2021"
|
||||
|
||||
[dependencies]
|
||||
wo-model = { path = "../wo-model" }
|
||||
libc = { workspace = true }
|
||||
bincode = { workspace = true }
|
||||
serde = { workspace = true }
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile = "3"
|
||||
194
crates/wo-index/src/date.rs
Normal file
194
crates/wo-index/src/date.rs
Normal file
|
|
@ -0,0 +1,194 @@
|
|||
use std::io;
|
||||
use std::path::Path;
|
||||
|
||||
/// Sorted index mapping timestamps to byte offsets in a .seg file.
|
||||
///
|
||||
/// On-disk format:
|
||||
/// ```text
|
||||
/// [8 bytes] entry_count: u64 LE
|
||||
/// For each entry (sorted by timestamp ascending):
|
||||
/// [8 bytes] timestamp: i64 LE
|
||||
/// [8 bytes] offset: u64 LE
|
||||
/// ```
|
||||
const ENTRY_SIZE: usize = 16; // i64 + u64
|
||||
const HEADER_SIZE: usize = 8;
|
||||
|
||||
pub struct DateIndex {
|
||||
data: Vec<u8>,
|
||||
count: usize,
|
||||
}
|
||||
|
||||
impl DateIndex {
|
||||
/// Build a date index. Entries are sorted by timestamp before writing.
|
||||
pub fn build(path: &Path, entries: &mut [(i64, u64)]) -> io::Result<()> {
|
||||
entries.sort_by_key(|e| e.0);
|
||||
|
||||
let file_size = HEADER_SIZE + entries.len() * ENTRY_SIZE;
|
||||
let mut data = vec![0u8; file_size];
|
||||
|
||||
data[0..8].copy_from_slice(&(entries.len() as u64).to_le_bytes());
|
||||
|
||||
for (i, &(ts, offset)) in entries.iter().enumerate() {
|
||||
let base = HEADER_SIZE + i * ENTRY_SIZE;
|
||||
data[base..base + 8].copy_from_slice(&ts.to_le_bytes());
|
||||
data[base + 8..base + 16].copy_from_slice(&offset.to_le_bytes());
|
||||
}
|
||||
|
||||
std::fs::write(path, &data)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Open a date index from disk.
|
||||
pub fn open(path: &Path) -> io::Result<Self> {
|
||||
let data = std::fs::read(path)?;
|
||||
if data.len() < HEADER_SIZE {
|
||||
return Err(io::Error::new(io::ErrorKind::InvalidData, "date.idx too small"));
|
||||
}
|
||||
let count = u64::from_le_bytes(data[0..8].try_into().unwrap()) as usize;
|
||||
Ok(Self { data, count })
|
||||
}
|
||||
|
||||
/// Get offsets for all entries with timestamps in [start, end].
|
||||
pub fn range(&self, start: i64, end: i64) -> Vec<u64> {
|
||||
let lo = self.lower_bound(start);
|
||||
let hi = self.upper_bound(end);
|
||||
|
||||
(lo..hi).map(|i| self.offset_at(i)).collect()
|
||||
}
|
||||
|
||||
/// Get offsets for the N most recent entries (highest timestamps).
|
||||
pub fn latest(&self, n: usize) -> Vec<u64> {
|
||||
let start = self.count.saturating_sub(n);
|
||||
(start..self.count)
|
||||
.rev()
|
||||
.map(|i| self.offset_at(i))
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Number of entries.
|
||||
pub fn len(&self) -> usize {
|
||||
self.count
|
||||
}
|
||||
|
||||
pub fn is_empty(&self) -> bool {
|
||||
self.count == 0
|
||||
}
|
||||
|
||||
fn timestamp_at(&self, i: usize) -> i64 {
|
||||
let base = HEADER_SIZE + i * ENTRY_SIZE;
|
||||
i64::from_le_bytes(self.data[base..base + 8].try_into().unwrap())
|
||||
}
|
||||
|
||||
fn offset_at(&self, i: usize) -> u64 {
|
||||
let base = HEADER_SIZE + i * ENTRY_SIZE;
|
||||
u64::from_le_bytes(self.data[base + 8..base + 16].try_into().unwrap())
|
||||
}
|
||||
|
||||
/// Binary search: first index where timestamp >= target.
|
||||
fn lower_bound(&self, target: i64) -> usize {
|
||||
let (mut lo, mut hi) = (0usize, self.count);
|
||||
while lo < hi {
|
||||
let mid = lo + (hi - lo) / 2;
|
||||
if self.timestamp_at(mid) < target {
|
||||
lo = mid + 1;
|
||||
} else {
|
||||
hi = mid;
|
||||
}
|
||||
}
|
||||
lo
|
||||
}
|
||||
|
||||
/// Binary search: first index where timestamp > target.
|
||||
fn upper_bound(&self, target: i64) -> usize {
|
||||
let (mut lo, mut hi) = (0usize, self.count);
|
||||
while lo < hi {
|
||||
let mid = lo + (hi - lo) / 2;
|
||||
if self.timestamp_at(mid) <= target {
|
||||
lo = mid + 1;
|
||||
} else {
|
||||
hi = mid;
|
||||
}
|
||||
}
|
||||
lo
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn build_and_range_query() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let path = tmp.path().join("date.idx");
|
||||
|
||||
let mut entries = vec![
|
||||
(1000i64, 100u64),
|
||||
(2000, 200),
|
||||
(3000, 300),
|
||||
(4000, 400),
|
||||
(5000, 500),
|
||||
];
|
||||
|
||||
DateIndex::build(&path, &mut entries).unwrap();
|
||||
let idx = DateIndex::open(&path).unwrap();
|
||||
|
||||
assert_eq!(idx.len(), 5);
|
||||
|
||||
// Full range.
|
||||
let all = idx.range(0, 9999);
|
||||
assert_eq!(all, vec![100, 200, 300, 400, 500]);
|
||||
|
||||
// Partial range.
|
||||
let mid = idx.range(2000, 4000);
|
||||
assert_eq!(mid, vec![200, 300, 400]);
|
||||
|
||||
// Single.
|
||||
let one = idx.range(3000, 3000);
|
||||
assert_eq!(one, vec![300]);
|
||||
|
||||
// Empty range.
|
||||
let none = idx.range(6000, 9000);
|
||||
assert!(none.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn latest_entries() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let path = tmp.path().join("date.idx");
|
||||
|
||||
let mut entries = vec![
|
||||
(1000i64, 100u64),
|
||||
(2000, 200),
|
||||
(3000, 300),
|
||||
];
|
||||
|
||||
DateIndex::build(&path, &mut entries).unwrap();
|
||||
let idx = DateIndex::open(&path).unwrap();
|
||||
|
||||
let top2 = idx.latest(2);
|
||||
assert_eq!(top2, vec![300, 200]);
|
||||
|
||||
let top10 = idx.latest(10);
|
||||
assert_eq!(top10, vec![300, 200, 100]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unsorted_input() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let path = tmp.path().join("date.idx");
|
||||
|
||||
let mut entries = vec![
|
||||
(5000i64, 500u64),
|
||||
(1000, 100),
|
||||
(3000, 300),
|
||||
];
|
||||
|
||||
DateIndex::build(&path, &mut entries).unwrap();
|
||||
let idx = DateIndex::open(&path).unwrap();
|
||||
|
||||
// Should be sorted on disk.
|
||||
let all = idx.range(0, 9999);
|
||||
assert_eq!(all, vec![100, 300, 500]);
|
||||
}
|
||||
}
|
||||
40
crates/wo-index/src/lib.rs
Normal file
40
crates/wo-index/src/lib.rs
Normal file
|
|
@ -0,0 +1,40 @@
|
|||
mod title;
|
||||
mod date;
|
||||
mod tags;
|
||||
|
||||
pub use title::TitleIndex;
|
||||
pub use date::DateIndex;
|
||||
pub use tags::TagIndex;
|
||||
|
||||
use std::io;
|
||||
use std::path::Path;
|
||||
|
||||
/// Rebuild all three index files from an iterator of article metadata.
|
||||
///
|
||||
/// Each entry is `(sys_title, published_on_timestamp, tags, seg_offset)`.
|
||||
pub fn rebuild_all(
|
||||
index_dir: &Path,
|
||||
entries: &[(String, i64, Vec<String>, u64)],
|
||||
) -> io::Result<()> {
|
||||
std::fs::create_dir_all(index_dir)?;
|
||||
|
||||
// Build title index.
|
||||
let title_entries: Vec<(&str, u64)> = entries.iter().map(|e| (e.0.as_str(), e.3)).collect();
|
||||
TitleIndex::build(&index_dir.join("title.idx"), &title_entries)?;
|
||||
|
||||
// Build date index.
|
||||
let mut date_entries: Vec<(i64, u64)> = entries.iter().map(|e| (e.1, e.3)).collect();
|
||||
DateIndex::build(&index_dir.join("date.idx"), &mut date_entries)?;
|
||||
|
||||
// Build tags index.
|
||||
let mut tag_map: std::collections::HashMap<String, Vec<u64>> = std::collections::HashMap::new();
|
||||
for entry in entries {
|
||||
for tag in &entry.2 {
|
||||
tag_map.entry(tag.clone()).or_default().push(entry.3);
|
||||
}
|
||||
}
|
||||
let tag_entries: Vec<(String, Vec<u64>)> = tag_map.into_iter().collect();
|
||||
TagIndex::build(&index_dir.join("tags.idx"), &tag_entries)?;
|
||||
|
||||
Ok(())
|
||||
}
|
||||
97
crates/wo-index/src/tags.rs
Normal file
97
crates/wo-index/src/tags.rs
Normal file
|
|
@ -0,0 +1,97 @@
|
|||
use std::collections::HashMap;
|
||||
use std::io;
|
||||
use std::path::Path;
|
||||
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
/// Inverted index mapping tag strings to lists of .seg byte offsets.
|
||||
///
|
||||
/// Serialized to disk using bincode for simplicity (article count is
|
||||
/// small enough that loading the full index into memory is fine).
|
||||
#[derive(Debug, Serialize, Deserialize)]
|
||||
struct TagIndexData {
|
||||
tags: HashMap<String, Vec<u64>>,
|
||||
}
|
||||
|
||||
pub struct TagIndex {
|
||||
tags: HashMap<String, Vec<u64>>,
|
||||
}
|
||||
|
||||
impl TagIndex {
|
||||
/// Build a tags index from a list of (tag, offsets) pairs.
|
||||
pub fn build(path: &Path, entries: &[(String, Vec<u64>)]) -> io::Result<()> {
|
||||
let tags: HashMap<String, Vec<u64>> = entries.iter().cloned().collect();
|
||||
let data = TagIndexData { tags };
|
||||
let bytes = bincode::serialize(&data).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::InvalidData, format!("bincode: {}", e))
|
||||
})?;
|
||||
std::fs::write(path, bytes)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Open a tags index from disk.
|
||||
pub fn open(path: &Path) -> io::Result<Self> {
|
||||
let bytes = std::fs::read(path)?;
|
||||
let data: TagIndexData = bincode::deserialize(&bytes).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::InvalidData, format!("bincode: {}", e))
|
||||
})?;
|
||||
Ok(Self { tags: data.tags })
|
||||
}
|
||||
|
||||
/// Get all offsets for articles tagged with the given tag.
|
||||
pub fn get(&self, tag: &str) -> Option<&[u64]> {
|
||||
self.tags.get(tag).map(|v| v.as_slice())
|
||||
}
|
||||
|
||||
/// All tag names in the index.
|
||||
pub fn tags(&self) -> Vec<&str> {
|
||||
self.tags.keys().map(|s| s.as_str()).collect()
|
||||
}
|
||||
|
||||
/// Number of distinct tags.
|
||||
pub fn len(&self) -> usize {
|
||||
self.tags.len()
|
||||
}
|
||||
|
||||
pub fn is_empty(&self) -> bool {
|
||||
self.tags.is_empty()
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn build_and_lookup() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let path = tmp.path().join("tags.idx");
|
||||
|
||||
let entries = vec![
|
||||
("rust".to_string(), vec![100u64, 200]),
|
||||
("linux".to_string(), vec![300]),
|
||||
("tutorial".to_string(), vec![100, 300, 400]),
|
||||
];
|
||||
|
||||
TagIndex::build(&path, &entries).unwrap();
|
||||
let idx = TagIndex::open(&path).unwrap();
|
||||
|
||||
assert_eq!(idx.len(), 3);
|
||||
assert_eq!(idx.get("rust"), Some(vec![100u64, 200].as_slice()));
|
||||
assert_eq!(idx.get("linux"), Some(vec![300u64].as_slice()));
|
||||
assert_eq!(idx.get("tutorial"), Some(vec![100u64, 300, 400].as_slice()));
|
||||
assert_eq!(idx.get("nonexistent"), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn empty_index() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let path = tmp.path().join("tags.idx");
|
||||
|
||||
TagIndex::build(&path, &[]).unwrap();
|
||||
let idx = TagIndex::open(&path).unwrap();
|
||||
|
||||
assert!(idx.is_empty());
|
||||
assert_eq!(idx.get("anything"), None);
|
||||
}
|
||||
}
|
||||
183
crates/wo-index/src/title.rs
Normal file
183
crates/wo-index/src/title.rs
Normal file
|
|
@ -0,0 +1,183 @@
|
|||
use std::collections::hash_map::DefaultHasher;
|
||||
use std::hash::{Hash, Hasher};
|
||||
use std::io;
|
||||
use std::path::Path;
|
||||
|
||||
/// On-disk hash index mapping `sys_title` (string) to byte offset in a .seg file.
|
||||
///
|
||||
/// Uses open addressing with linear probing. The table is sized at 2x the entry
|
||||
/// count (load factor 0.5) to keep collision chains short.
|
||||
///
|
||||
/// On-disk format:
|
||||
/// ```text
|
||||
/// [8 bytes] table_size: u64 LE (number of slots)
|
||||
/// [8 bytes] entry_count: u64 LE
|
||||
/// For each slot (table_size slots):
|
||||
/// [8 bytes] offset: u64 LE (0 = empty, otherwise seg offset + 1)
|
||||
/// [2 bytes] key_len: u16 LE
|
||||
/// [128 bytes] key_data: zero-padded sys_title
|
||||
/// Total slot size: 138 bytes
|
||||
/// ```
|
||||
const SLOT_SIZE: usize = 8 + 2 + 128; // offset + key_len + key_data
|
||||
const HEADER_SIZE: usize = 16; // table_size + entry_count
|
||||
const MAX_KEY_LEN: usize = 128;
|
||||
|
||||
pub struct TitleIndex {
|
||||
data: Vec<u8>,
|
||||
table_size: u64,
|
||||
}
|
||||
|
||||
impl TitleIndex {
|
||||
/// Build a title index file from a list of (sys_title, seg_offset) pairs.
|
||||
pub fn build(path: &Path, entries: &[(&str, u64)]) -> io::Result<()> {
|
||||
let table_size = (entries.len() * 2).max(16) as u64;
|
||||
let file_size = HEADER_SIZE + (table_size as usize) * SLOT_SIZE;
|
||||
|
||||
let mut data = vec![0u8; file_size];
|
||||
|
||||
// Write header.
|
||||
data[0..8].copy_from_slice(&table_size.to_le_bytes());
|
||||
data[8..16].copy_from_slice(&(entries.len() as u64).to_le_bytes());
|
||||
|
||||
// Insert entries.
|
||||
for &(key, offset) in entries {
|
||||
let slot = Self::find_empty_slot(&data, table_size, key);
|
||||
Self::write_slot(&mut data, slot, key, offset);
|
||||
}
|
||||
|
||||
std::fs::write(path, &data)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Open a title index from disk (loads into memory).
|
||||
pub fn open(path: &Path) -> io::Result<Self> {
|
||||
let data = std::fs::read(path)?;
|
||||
if data.len() < HEADER_SIZE {
|
||||
return Err(io::Error::new(io::ErrorKind::InvalidData, "title.idx too small"));
|
||||
}
|
||||
let table_size = u64::from_le_bytes(data[0..8].try_into().unwrap());
|
||||
Ok(Self { data, table_size })
|
||||
}
|
||||
|
||||
/// Look up a sys_title, returning the seg file byte offset if found.
|
||||
pub fn get(&self, sys_title: &str) -> Option<u64> {
|
||||
let mut slot = Self::hash_key(sys_title) % self.table_size;
|
||||
|
||||
for _ in 0..self.table_size {
|
||||
let (stored_offset, stored_key) = self.read_slot(slot as usize);
|
||||
|
||||
if stored_offset == 0 {
|
||||
return None; // empty slot, key not found
|
||||
}
|
||||
|
||||
if stored_key == sys_title {
|
||||
return Some(stored_offset - 1); // stored as offset + 1
|
||||
}
|
||||
|
||||
slot = (slot + 1) % self.table_size;
|
||||
}
|
||||
|
||||
None
|
||||
}
|
||||
|
||||
/// Number of entries in the index.
|
||||
pub fn len(&self) -> usize {
|
||||
u64::from_le_bytes(self.data[8..16].try_into().unwrap()) as usize
|
||||
}
|
||||
|
||||
pub fn is_empty(&self) -> bool {
|
||||
self.len() == 0
|
||||
}
|
||||
|
||||
fn hash_key(key: &str) -> u64 {
|
||||
let mut hasher = DefaultHasher::new();
|
||||
key.hash(&mut hasher);
|
||||
hasher.finish()
|
||||
}
|
||||
|
||||
fn find_empty_slot(data: &[u8], table_size: u64, key: &str) -> usize {
|
||||
let mut slot = (Self::hash_key(key) % table_size) as usize;
|
||||
loop {
|
||||
let base = HEADER_SIZE + slot * SLOT_SIZE;
|
||||
let stored_offset = u64::from_le_bytes(data[base..base + 8].try_into().unwrap());
|
||||
if stored_offset == 0 {
|
||||
return slot;
|
||||
}
|
||||
slot = (slot + 1) % table_size as usize;
|
||||
}
|
||||
}
|
||||
|
||||
fn write_slot(data: &mut [u8], slot: usize, key: &str, offset: u64) {
|
||||
let base = HEADER_SIZE + slot * SLOT_SIZE;
|
||||
let stored_offset = offset + 1; // +1 so that 0 means empty
|
||||
data[base..base + 8].copy_from_slice(&stored_offset.to_le_bytes());
|
||||
|
||||
let key_bytes = key.as_bytes();
|
||||
let key_len = key_bytes.len().min(MAX_KEY_LEN) as u16;
|
||||
data[base + 8..base + 10].copy_from_slice(&key_len.to_le_bytes());
|
||||
data[base + 10..base + 10 + key_len as usize].copy_from_slice(&key_bytes[..key_len as usize]);
|
||||
}
|
||||
|
||||
fn read_slot(&self, slot: usize) -> (u64, String) {
|
||||
let base = HEADER_SIZE + slot * SLOT_SIZE;
|
||||
let stored_offset = u64::from_le_bytes(self.data[base..base + 8].try_into().unwrap());
|
||||
let key_len = u16::from_le_bytes(self.data[base + 8..base + 10].try_into().unwrap()) as usize;
|
||||
let key = String::from_utf8_lossy(&self.data[base + 10..base + 10 + key_len]).to_string();
|
||||
(stored_offset, key)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn build_and_lookup() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let path = tmp.path().join("title.idx");
|
||||
|
||||
let entries = vec![
|
||||
("linux-misc", 32u64),
|
||||
("aws-lambda-pulumi", 500),
|
||||
("rust-patterns", 1200),
|
||||
];
|
||||
|
||||
TitleIndex::build(&path, &entries).unwrap();
|
||||
let idx = TitleIndex::open(&path).unwrap();
|
||||
|
||||
assert_eq!(idx.len(), 3);
|
||||
assert_eq!(idx.get("linux-misc"), Some(32));
|
||||
assert_eq!(idx.get("aws-lambda-pulumi"), Some(500));
|
||||
assert_eq!(idx.get("rust-patterns"), Some(1200));
|
||||
assert_eq!(idx.get("nonexistent"), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn empty_index() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let path = tmp.path().join("title.idx");
|
||||
|
||||
TitleIndex::build(&path, &[]).unwrap();
|
||||
let idx = TitleIndex::open(&path).unwrap();
|
||||
|
||||
assert!(idx.is_empty());
|
||||
assert_eq!(idx.get("anything"), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn many_entries() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let path = tmp.path().join("title.idx");
|
||||
|
||||
let keys: Vec<String> = (0..100).map(|i| format!("article-{}", i)).collect();
|
||||
let entries: Vec<(&str, u64)> = keys.iter().enumerate().map(|(i, k)| (k.as_str(), i as u64 * 100)).collect();
|
||||
|
||||
TitleIndex::build(&path, &entries).unwrap();
|
||||
let idx = TitleIndex::open(&path).unwrap();
|
||||
|
||||
assert_eq!(idx.len(), 100);
|
||||
for (i, key) in keys.iter().enumerate() {
|
||||
assert_eq!(idx.get(key), Some(i as u64 * 100));
|
||||
}
|
||||
}
|
||||
}
|
||||
6
crates/wo-md/Cargo.toml
Normal file
6
crates/wo-md/Cargo.toml
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
[package]
|
||||
name = "wo-md"
|
||||
version = "0.1.0"
|
||||
edition = "2021"
|
||||
|
||||
[dependencies]
|
||||
324
crates/wo-md/src/block.rs
Normal file
324
crates/wo-md/src/block.rs
Normal file
|
|
@ -0,0 +1,324 @@
|
|||
use crate::highlight::highlight;
|
||||
use crate::inline::markdown_to_html;
|
||||
|
||||
/// Parse a full markdown document into HTML.
|
||||
///
|
||||
/// Handles block-level elements (headings, paragraphs, code fences, lists,
|
||||
/// blockquotes, images) and delegates inline formatting to `markdown_to_html`.
|
||||
pub fn markdown_to_html_block(md: &str) -> String {
|
||||
let lines: Vec<&str> = md.lines().collect();
|
||||
let len = lines.len();
|
||||
let mut html = String::with_capacity(md.len() * 2);
|
||||
let mut i = 0;
|
||||
|
||||
while i < len {
|
||||
let line = lines[i];
|
||||
let trimmed = line.trim();
|
||||
|
||||
// Skip blank lines.
|
||||
if trimmed.is_empty() {
|
||||
i += 1;
|
||||
continue;
|
||||
}
|
||||
|
||||
// Fenced code block: ```lang ... ```
|
||||
if trimmed.starts_with("```") {
|
||||
let lang = trimmed[3..].trim();
|
||||
i += 1;
|
||||
let mut code = String::new();
|
||||
while i < len && !lines[i].trim().starts_with("```") {
|
||||
if !code.is_empty() {
|
||||
code.push('\n');
|
||||
}
|
||||
code.push_str(lines[i]);
|
||||
i += 1;
|
||||
}
|
||||
if i < len {
|
||||
i += 1; // skip closing ```
|
||||
}
|
||||
|
||||
if lang.is_empty() {
|
||||
html.push_str("<pre><code>");
|
||||
html.push_str(&escape_html(&code));
|
||||
html.push_str("</code></pre>\n");
|
||||
} else {
|
||||
html.push_str(&format!("<pre><code class=\"language-{}\">", escape_html(lang)));
|
||||
html.push_str(&highlight(&code, lang));
|
||||
html.push_str("</code></pre>\n");
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
// Heading: # ... ######
|
||||
if trimmed.starts_with('#') {
|
||||
let level = trimmed.chars().take_while(|&c| c == '#').count().min(6);
|
||||
let text = trimmed[level..].trim();
|
||||
html.push_str(&format!(
|
||||
"<h{0}>{1}</h{0}>\n",
|
||||
level,
|
||||
markdown_to_html(text)
|
||||
));
|
||||
i += 1;
|
||||
continue;
|
||||
}
|
||||
|
||||
// Blockquote: > text
|
||||
if trimmed.starts_with('>') {
|
||||
let mut quote_lines = Vec::new();
|
||||
while i < len && lines[i].trim().starts_with('>') {
|
||||
let content = lines[i].trim().strip_prefix('>').unwrap_or("").trim();
|
||||
quote_lines.push(content);
|
||||
i += 1;
|
||||
}
|
||||
html.push_str("<blockquote>");
|
||||
html.push_str(&markdown_to_html("e_lines.join(" ")));
|
||||
html.push_str("</blockquote>\n");
|
||||
continue;
|
||||
}
|
||||
|
||||
// Unordered list: - item or * item
|
||||
if (trimmed.starts_with("- ") || trimmed.starts_with("* "))
|
||||
&& !trimmed.starts_with("---")
|
||||
{
|
||||
html.push_str("<ul>\n");
|
||||
while i < len {
|
||||
let lt = lines[i].trim();
|
||||
if lt.starts_with("- ") || lt.starts_with("* ") {
|
||||
let text = <[2..];
|
||||
html.push_str(&format!("<li>{}</li>\n", markdown_to_html(text)));
|
||||
i += 1;
|
||||
} else if lt.is_empty() {
|
||||
i += 1;
|
||||
break;
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
}
|
||||
html.push_str("</ul>\n");
|
||||
continue;
|
||||
}
|
||||
|
||||
// Ordered list: 1. item
|
||||
if trimmed.len() > 2 && trimmed.as_bytes()[0].is_ascii_digit() {
|
||||
if let Some(rest) = strip_ordered_prefix(trimmed) {
|
||||
html.push_str("<ol>\n");
|
||||
html.push_str(&format!("<li>{}</li>\n", markdown_to_html(rest)));
|
||||
i += 1;
|
||||
while i < len {
|
||||
let lt = lines[i].trim();
|
||||
if let Some(rest) = strip_ordered_prefix(lt) {
|
||||
html.push_str(&format!("<li>{}</li>\n", markdown_to_html(rest)));
|
||||
i += 1;
|
||||
} else if lt.is_empty() {
|
||||
i += 1;
|
||||
break;
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
}
|
||||
html.push_str("</ol>\n");
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
// Image: 
|
||||
if trimmed.starts_with("![") {
|
||||
if let Some((alt, url)) = parse_image(trimmed) {
|
||||
html.push_str(&format!(
|
||||
"<img src=\"{}\" alt=\"{}\">\n",
|
||||
escape_html(url),
|
||||
escape_html(alt)
|
||||
));
|
||||
i += 1;
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
// Horizontal rule: --- or ***
|
||||
if trimmed == "---" || trimmed == "***" || trimmed == "___" {
|
||||
html.push_str("<hr>\n");
|
||||
i += 1;
|
||||
continue;
|
||||
}
|
||||
|
||||
// Paragraph: collect consecutive non-blank, non-block lines.
|
||||
let mut para_lines = Vec::new();
|
||||
while i < len {
|
||||
let lt = lines[i].trim();
|
||||
if lt.is_empty()
|
||||
|| lt.starts_with('#')
|
||||
|| lt.starts_with("```")
|
||||
|| lt.starts_with('>')
|
||||
|| lt == "---"
|
||||
|| lt == "***"
|
||||
|| lt == "___"
|
||||
{
|
||||
break;
|
||||
}
|
||||
// Check if next line starts a list
|
||||
if (lt.starts_with("- ") || lt.starts_with("* ")) && !lt.starts_with("---") {
|
||||
break;
|
||||
}
|
||||
if strip_ordered_prefix(lt).is_some() && para_lines.is_empty() {
|
||||
break;
|
||||
}
|
||||
para_lines.push(lt);
|
||||
i += 1;
|
||||
}
|
||||
if !para_lines.is_empty() {
|
||||
html.push_str("<p>");
|
||||
html.push_str(&markdown_to_html(¶_lines.join(" ")));
|
||||
html.push_str("</p>\n");
|
||||
}
|
||||
}
|
||||
|
||||
html
|
||||
}
|
||||
|
||||
fn escape_html(s: &str) -> String {
|
||||
s.replace('&', "&")
|
||||
.replace('<', "<")
|
||||
.replace('>', ">")
|
||||
.replace('"', """)
|
||||
}
|
||||
|
||||
fn strip_ordered_prefix(s: &str) -> Option<&str> {
|
||||
let dot_pos = s.find(". ")?;
|
||||
if dot_pos > 0 && s[..dot_pos].chars().all(|c| c.is_ascii_digit()) {
|
||||
Some(&s[dot_pos + 2..])
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
fn parse_image(s: &str) -> Option<(&str, &str)> {
|
||||
// 
|
||||
let alt_start = s.find("![")? + 2;
|
||||
let alt_end = s[alt_start..].find(']')? + alt_start;
|
||||
let url_start = s[alt_end..].find('(')? + alt_end + 1;
|
||||
let url_end = s[url_start..].find(')')? + url_start;
|
||||
Some((&s[alt_start..alt_end], &s[url_start..url_end]))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn headings() {
|
||||
let md = "# Title\n\n## Subtitle\n\n### Third";
|
||||
let html = markdown_to_html_block(md);
|
||||
assert!(html.contains("<h1>Title</h1>"));
|
||||
assert!(html.contains("<h2>Subtitle</h2>"));
|
||||
assert!(html.contains("<h3>Third</h3>"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn paragraphs() {
|
||||
let md = "First paragraph.\n\nSecond paragraph.";
|
||||
let html = markdown_to_html_block(md);
|
||||
assert!(html.contains("<p>First paragraph.</p>"));
|
||||
assert!(html.contains("<p>Second paragraph.</p>"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn code_fence_with_language() {
|
||||
let md = "```rust\nfn main() {\n let x = 42;\n}\n```";
|
||||
let html = markdown_to_html_block(md);
|
||||
assert!(html.contains("<pre><code class=\"language-rust\">"));
|
||||
assert!(html.contains("<span class=\"kw\">fn</span>"));
|
||||
assert!(html.contains("<span class=\"num\">42</span>"));
|
||||
assert!(html.contains("</code></pre>"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn code_fence_no_language() {
|
||||
let md = "```\nplain code\n```";
|
||||
let html = markdown_to_html_block(md);
|
||||
assert!(html.contains("<pre><code>plain code</code></pre>"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unordered_list() {
|
||||
let md = "- Item one\n- Item two\n- Item three";
|
||||
let html = markdown_to_html_block(md);
|
||||
assert!(html.contains("<ul>"));
|
||||
assert!(html.contains("<li>Item one</li>"));
|
||||
assert!(html.contains("<li>Item two</li>"));
|
||||
assert!(html.contains("</ul>"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ordered_list() {
|
||||
let md = "1. First\n2. Second\n3. Third";
|
||||
let html = markdown_to_html_block(md);
|
||||
assert!(html.contains("<ol>"));
|
||||
assert!(html.contains("<li>First</li>"));
|
||||
assert!(html.contains("<li>Third</li>"));
|
||||
assert!(html.contains("</ol>"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn blockquote() {
|
||||
let md = "> This is a quote\n> spanning two lines";
|
||||
let html = markdown_to_html_block(md);
|
||||
assert!(html.contains("<blockquote>This is a quote spanning two lines</blockquote>"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn image() {
|
||||
let md = "";
|
||||
let html = markdown_to_html_block(md);
|
||||
assert!(html.contains("<img src=\"https://example.com/img.png\" alt=\"Alt text\">"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn horizontal_rule() {
|
||||
let md = "Before\n\n---\n\nAfter";
|
||||
let html = markdown_to_html_block(md);
|
||||
assert!(html.contains("<hr>"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn inline_formatting_in_paragraphs() {
|
||||
let md = "Use **bold** and `code` in a paragraph.";
|
||||
let html = markdown_to_html_block(md);
|
||||
assert!(html.contains("<strong>bold</strong>"));
|
||||
assert!(html.contains("<code>code</code>"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn full_article() {
|
||||
let md = "\
|
||||
# Getting Started
|
||||
|
||||
Welcome to the guide.
|
||||
|
||||
## Installation
|
||||
|
||||
Install with cargo:
|
||||
|
||||
```bash
|
||||
cargo install writeonce
|
||||
```
|
||||
|
||||
## Features
|
||||
|
||||
- Fast rendering
|
||||
- Zero dependencies
|
||||
- **Bold** feature
|
||||
|
||||
> Note: this is a quote.
|
||||
|
||||

|
||||
";
|
||||
let html = markdown_to_html_block(md);
|
||||
assert!(html.contains("<h1>Getting Started</h1>"));
|
||||
assert!(html.contains("<h2>Installation</h2>"));
|
||||
assert!(html.contains("<p>Install with cargo:</p>"));
|
||||
assert!(html.contains("<pre><code class=\"language-bash\">"));
|
||||
assert!(html.contains("<li>Fast rendering</li>"));
|
||||
assert!(html.contains("<blockquote>"));
|
||||
assert!(html.contains("<img src=\"logo.png\""));
|
||||
}
|
||||
}
|
||||
154
crates/wo-md/src/highlight.rs
Normal file
154
crates/wo-md/src/highlight.rs
Normal file
|
|
@ -0,0 +1,154 @@
|
|||
/// Server-side syntax highlighting for code blocks.
|
||||
///
|
||||
/// Returns HTML with `<span class="...">` tokens. Pair with code-theme.css.
|
||||
pub fn highlight(code: &str, language: &str) -> String {
|
||||
let keywords = keywords_for(language);
|
||||
let mut result = String::with_capacity(code.len() * 2);
|
||||
let chars: Vec<char> = code.chars().collect();
|
||||
let len = chars.len();
|
||||
let mut i = 0;
|
||||
|
||||
while i < len {
|
||||
// String literals (double or single quotes).
|
||||
if chars[i] == '"' || chars[i] == '\'' {
|
||||
let quote = chars[i];
|
||||
let start = i;
|
||||
i += 1;
|
||||
while i < len && chars[i] != quote {
|
||||
if chars[i] == '\\' { i += 1; } // skip escaped char
|
||||
i += 1;
|
||||
}
|
||||
if i < len { i += 1; } // closing quote
|
||||
let s: String = chars[start..i].iter().collect();
|
||||
result.push_str(&format!("<span class=\"str\">{}</span>", escape(&s)));
|
||||
continue;
|
||||
}
|
||||
|
||||
// Line comments.
|
||||
if i + 1 < len && chars[i] == '/' && chars[i + 1] == '/' {
|
||||
let start = i;
|
||||
while i < len && chars[i] != '\n' { i += 1; }
|
||||
let s: String = chars[start..i].iter().collect();
|
||||
result.push_str(&format!("<span class=\"cm\">{}</span>", escape(&s)));
|
||||
continue;
|
||||
}
|
||||
|
||||
// Hash comments (bash, yaml, etc.).
|
||||
if chars[i] == '#' && (language == "bash" || language == "yaml" || language == "python") {
|
||||
let start = i;
|
||||
while i < len && chars[i] != '\n' { i += 1; }
|
||||
let s: String = chars[start..i].iter().collect();
|
||||
result.push_str(&format!("<span class=\"cm\">{}</span>", escape(&s)));
|
||||
continue;
|
||||
}
|
||||
|
||||
// Numbers.
|
||||
if chars[i].is_ascii_digit() {
|
||||
let start = i;
|
||||
while i < len && (chars[i].is_ascii_alphanumeric() || chars[i] == '.') { i += 1; }
|
||||
let s: String = chars[start..i].iter().collect();
|
||||
result.push_str(&format!("<span class=\"num\">{}</span>", escape(&s)));
|
||||
continue;
|
||||
}
|
||||
|
||||
// Identifiers / keywords.
|
||||
if chars[i].is_ascii_alphabetic() || chars[i] == '_' {
|
||||
let start = i;
|
||||
while i < len && (chars[i].is_ascii_alphanumeric() || chars[i] == '_') { i += 1; }
|
||||
let word: String = chars[start..i].iter().collect();
|
||||
if keywords.contains(&word.as_str()) {
|
||||
result.push_str(&format!("<span class=\"kw\">{}</span>", escape(&word)));
|
||||
} else {
|
||||
result.push_str(&escape(&word));
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
// Everything else.
|
||||
result.push_str(&escape(&chars[i].to_string()));
|
||||
i += 1;
|
||||
}
|
||||
|
||||
result
|
||||
}
|
||||
|
||||
fn escape(s: &str) -> String {
|
||||
s.replace('&', "&")
|
||||
.replace('<', "<")
|
||||
.replace('>', ">")
|
||||
}
|
||||
|
||||
fn keywords_for(language: &str) -> &'static [&'static str] {
|
||||
match language {
|
||||
"rust" => &[
|
||||
"fn", "let", "mut", "const", "struct", "enum", "impl", "trait", "pub", "use",
|
||||
"mod", "crate", "self", "super", "return", "if", "else", "match", "for", "while",
|
||||
"loop", "break", "continue", "where", "type", "as", "in", "ref", "move",
|
||||
"async", "await", "unsafe", "extern", "dyn", "true", "false",
|
||||
],
|
||||
"go" => &[
|
||||
"func", "var", "const", "type", "struct", "interface", "map", "chan",
|
||||
"package", "import", "return", "if", "else", "for", "range", "switch",
|
||||
"case", "default", "go", "defer", "select", "true", "false", "nil",
|
||||
],
|
||||
"bash" | "sh" => &[
|
||||
"if", "then", "else", "elif", "fi", "for", "while", "do", "done",
|
||||
"case", "esac", "function", "return", "export", "local", "echo",
|
||||
"in", "true", "false",
|
||||
],
|
||||
"yaml" => &["true", "false", "null", "yes", "no"],
|
||||
"json" => &["true", "false", "null"],
|
||||
"python" => &[
|
||||
"def", "class", "if", "elif", "else", "for", "while", "return",
|
||||
"import", "from", "as", "with", "try", "except", "finally",
|
||||
"raise", "pass", "lambda", "yield", "True", "False", "None", "in",
|
||||
"not", "and", "or", "is", "async", "await",
|
||||
],
|
||||
_ => &[],
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn highlight_rust() {
|
||||
let code = "fn main() {\n let x = 42;\n}";
|
||||
let html = highlight(code, "rust");
|
||||
assert!(html.contains("<span class=\"kw\">fn</span>"));
|
||||
assert!(html.contains("<span class=\"kw\">let</span>"));
|
||||
assert!(html.contains("<span class=\"num\">42</span>"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn highlight_string() {
|
||||
let code = r#"let s = "hello";"#;
|
||||
let html = highlight(code, "rust");
|
||||
assert!(html.contains("<span class=\"str\">\"hello\"</span>"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn highlight_comment() {
|
||||
let code = "// this is a comment\nlet x = 1;";
|
||||
let html = highlight(code, "rust");
|
||||
assert!(html.contains("<span class=\"cm\">// this is a comment</span>"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn highlight_bash() {
|
||||
let code = "# comment\nexport PATH=/usr/bin";
|
||||
let html = highlight(code, "bash");
|
||||
assert!(html.contains("<span class=\"cm\"># comment</span>"));
|
||||
assert!(html.contains("<span class=\"kw\">export</span>"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn escapes_html() {
|
||||
let code = "fn compare<T>(a: T, b: T) {}";
|
||||
let html = highlight(code, "rust");
|
||||
assert!(html.contains("<"));
|
||||
assert!(html.contains(">"));
|
||||
assert!(!html.contains("<T>"));
|
||||
}
|
||||
}
|
||||
171
crates/wo-md/src/inline.rs
Normal file
171
crates/wo-md/src/inline.rs
Normal file
|
|
@ -0,0 +1,171 @@
|
|||
/// Convert inline markdown to HTML.
|
||||
///
|
||||
/// Handles: **bold**, *italic*, `code`, [text](url), and HTML entity escaping.
|
||||
/// Does not handle block-level elements (headers, lists, etc.) — those are
|
||||
/// already structured in the article JSON.
|
||||
pub fn markdown_to_html(text: &str) -> String {
|
||||
let mut result = String::with_capacity(text.len() * 2);
|
||||
let chars: Vec<char> = text.chars().collect();
|
||||
let len = chars.len();
|
||||
let mut i = 0;
|
||||
|
||||
while i < len {
|
||||
match chars[i] {
|
||||
// HTML entity escaping.
|
||||
'&' => { result.push_str("&"); i += 1; }
|
||||
'<' => { result.push_str("<"); i += 1; }
|
||||
'>' => { result.push_str(">"); i += 1; }
|
||||
|
||||
// **bold** or *italic*
|
||||
'*' => {
|
||||
if i + 1 < len && chars[i + 1] == '*' {
|
||||
// **bold**
|
||||
if let Some(end) = find_closing(&chars, i + 2, "**") {
|
||||
result.push_str("<strong>");
|
||||
let inner: String = chars[i + 2..end].iter().collect();
|
||||
result.push_str(&escape_html(&inner));
|
||||
result.push_str("</strong>");
|
||||
i = end + 2;
|
||||
} else {
|
||||
result.push('*');
|
||||
i += 1;
|
||||
}
|
||||
} else {
|
||||
// *italic*
|
||||
if let Some(end) = find_closing_char(&chars, i + 1, '*') {
|
||||
result.push_str("<em>");
|
||||
let inner: String = chars[i + 1..end].iter().collect();
|
||||
result.push_str(&escape_html(&inner));
|
||||
result.push_str("</em>");
|
||||
i = end + 1;
|
||||
} else {
|
||||
result.push('*');
|
||||
i += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// `inline code`
|
||||
'`' => {
|
||||
if let Some(end) = find_closing_char(&chars, i + 1, '`') {
|
||||
result.push_str("<code>");
|
||||
let inner: String = chars[i + 1..end].iter().collect();
|
||||
result.push_str(&escape_html(&inner));
|
||||
result.push_str("</code>");
|
||||
i = end + 1;
|
||||
} else {
|
||||
result.push('`');
|
||||
i += 1;
|
||||
}
|
||||
}
|
||||
|
||||
// [text](url)
|
||||
'[' => {
|
||||
if let Some((text_end, url_start, url_end)) = parse_link(&chars, i) {
|
||||
let link_text: String = chars[i + 1..text_end].iter().collect();
|
||||
let url: String = chars[url_start..url_end].iter().collect();
|
||||
result.push_str(&format!(
|
||||
"<a href=\"{}\">{}</a>",
|
||||
escape_html(&url),
|
||||
escape_html(&link_text)
|
||||
));
|
||||
i = url_end + 1;
|
||||
} else {
|
||||
result.push('[');
|
||||
i += 1;
|
||||
}
|
||||
}
|
||||
|
||||
c => {
|
||||
result.push(c);
|
||||
i += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
result
|
||||
}
|
||||
|
||||
fn escape_html(s: &str) -> String {
|
||||
s.replace('&', "&")
|
||||
.replace('<', "<")
|
||||
.replace('>', ">")
|
||||
.replace('"', """)
|
||||
}
|
||||
|
||||
fn find_closing(chars: &[char], start: usize, marker: &str) -> Option<usize> {
|
||||
let marker_chars: Vec<char> = marker.chars().collect();
|
||||
let mlen = marker_chars.len();
|
||||
for i in start..chars.len().saturating_sub(mlen - 1) {
|
||||
if chars[i..i + mlen] == marker_chars[..] {
|
||||
return Some(i);
|
||||
}
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
fn find_closing_char(chars: &[char], start: usize, marker: char) -> Option<usize> {
|
||||
for i in start..chars.len() {
|
||||
if chars[i] == marker {
|
||||
return Some(i);
|
||||
}
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
fn parse_link(chars: &[char], start: usize) -> Option<(usize, usize, usize)> {
|
||||
// [text](url)
|
||||
let text_end = find_closing_char(chars, start + 1, ']')?;
|
||||
if text_end + 1 >= chars.len() || chars[text_end + 1] != '(' {
|
||||
return None;
|
||||
}
|
||||
let url_start = text_end + 2;
|
||||
let url_end = find_closing_char(chars, url_start, ')')?;
|
||||
Some((text_end, url_start, url_end))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn plain_text() {
|
||||
assert_eq!(markdown_to_html("hello world"), "hello world");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn bold() {
|
||||
assert_eq!(markdown_to_html("**bold**"), "<strong>bold</strong>");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn italic() {
|
||||
assert_eq!(markdown_to_html("*italic*"), "<em>italic</em>");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn inline_code() {
|
||||
assert_eq!(markdown_to_html("`code`"), "<code>code</code>");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn link() {
|
||||
assert_eq!(
|
||||
markdown_to_html("[click](https://example.com)"),
|
||||
"<a href=\"https://example.com\">click</a>"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn mixed() {
|
||||
assert_eq!(
|
||||
markdown_to_html("Use **Rust** with `cargo` for [docs](https://doc.rust-lang.org)"),
|
||||
"Use <strong>Rust</strong> with <code>cargo</code> for <a href=\"https://doc.rust-lang.org\">docs</a>"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn html_escaping() {
|
||||
assert_eq!(markdown_to_html("<script>alert('xss')</script>"), "<script>alert('xss')</script>");
|
||||
}
|
||||
}
|
||||
7
crates/wo-md/src/lib.rs
Normal file
7
crates/wo-md/src/lib.rs
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
mod block;
|
||||
mod inline;
|
||||
mod highlight;
|
||||
|
||||
pub use block::markdown_to_html_block;
|
||||
pub use inline::markdown_to_html;
|
||||
pub use highlight::highlight;
|
||||
13
crates/wo-model/Cargo.toml
Normal file
13
crates/wo-model/Cargo.toml
Normal file
|
|
@ -0,0 +1,13 @@
|
|||
[package]
|
||||
name = "wo-model"
|
||||
version = "0.1.0"
|
||||
edition = "2021"
|
||||
|
||||
[dependencies]
|
||||
serde = { workspace = true }
|
||||
serde_json = { workspace = true }
|
||||
bincode = { workspace = true }
|
||||
wo-md = { path = "../wo-md" }
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile = "3"
|
||||
177
crates/wo-model/src/article.rs
Normal file
177
crates/wo-model/src/article.rs
Normal file
|
|
@ -0,0 +1,177 @@
|
|||
use serde::{Deserialize, Serialize};
|
||||
|
||||
/// Markdown-first article: minimal metadata + pre-rendered HTML content.
|
||||
///
|
||||
/// The `sys_title` field is the primary key — a URL-safe slug
|
||||
/// used for lookups and routing (e.g., "linux-misc").
|
||||
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
|
||||
pub struct Article {
|
||||
pub sys_title: String,
|
||||
pub title: String,
|
||||
pub published: bool,
|
||||
pub author: String,
|
||||
#[serde(default)]
|
||||
pub tags: Vec<String>,
|
||||
#[serde(default)]
|
||||
pub published_on: Option<i64>,
|
||||
/// Pre-rendered HTML from the .md file. Populated by ContentLoader.
|
||||
#[serde(default)]
|
||||
pub content_html: String,
|
||||
}
|
||||
|
||||
// --- Legacy types for backwards compatibility with old JSON format ---
|
||||
|
||||
/// Legacy article format (nested JSON with sections/codes/etc.).
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct LegacyArticle {
|
||||
pub title: String,
|
||||
pub sys_title: String,
|
||||
pub published: bool,
|
||||
pub content: LegacyArticleContent,
|
||||
#[serde(default)]
|
||||
pub do_aws_sync: Option<bool>,
|
||||
#[serde(default)]
|
||||
pub published_on: Option<i64>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct LegacyArticleContent {
|
||||
pub author: String,
|
||||
pub content: LegacyArticleBody,
|
||||
#[serde(default, rename = "publishedOn")]
|
||||
pub published_on: Option<i64>,
|
||||
#[serde(default)]
|
||||
pub references: Vec<LegacyReference>,
|
||||
#[serde(default)]
|
||||
pub tags: Vec<String>,
|
||||
#[serde(default)]
|
||||
pub title: Option<String>,
|
||||
#[serde(default)]
|
||||
pub systitle: Option<String>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct LegacyArticleBody {
|
||||
#[serde(default)]
|
||||
pub sections: Vec<LegacySection>,
|
||||
#[serde(default)]
|
||||
pub codes: Vec<serde_json::Value>,
|
||||
#[serde(default)]
|
||||
pub images: Vec<serde_json::Value>,
|
||||
#[serde(default)]
|
||||
pub img: Option<serde_json::Value>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct LegacySection {
|
||||
pub heading: String,
|
||||
#[serde(default)]
|
||||
pub paragraphs: Vec<String>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
pub struct LegacyReference {
|
||||
#[serde(default, rename = "dateAccessed")]
|
||||
pub date_accessed: Option<i64>,
|
||||
#[serde(default)]
|
||||
pub title: Option<String>,
|
||||
#[serde(default)]
|
||||
pub url: Option<String>,
|
||||
}
|
||||
|
||||
impl LegacyArticle {
|
||||
/// Convert a legacy article to the new format by building HTML from sections.
|
||||
pub fn to_article(&self) -> Article {
|
||||
let mut html = String::new();
|
||||
for section in &self.content.content.sections {
|
||||
html.push_str(&format!("<h2>{}</h2>\n", section.heading));
|
||||
for para in §ion.paragraphs {
|
||||
if !para.is_empty() {
|
||||
html.push_str(&format!("<p>{}</p>\n", wo_md::markdown_to_html(para)));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Article {
|
||||
sys_title: self.sys_title.clone(),
|
||||
title: self.title.clone(),
|
||||
published: self.published,
|
||||
author: self.content.author.clone(),
|
||||
tags: self.content.tags.clone(),
|
||||
published_on: self.published_on.or(self.content.published_on),
|
||||
content_html: html,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn deserialize_new_format() {
|
||||
let json = r#"{
|
||||
"sys_title": "test-article",
|
||||
"title": "Test Article",
|
||||
"published": true,
|
||||
"author": "Author",
|
||||
"tags": ["test"],
|
||||
"published_on": 1000
|
||||
}"#;
|
||||
|
||||
let article: Article = serde_json::from_str(json).unwrap();
|
||||
assert_eq!(article.sys_title, "test-article");
|
||||
assert_eq!(article.author, "Author");
|
||||
assert_eq!(article.tags, vec!["test"]);
|
||||
assert!(article.published);
|
||||
assert_eq!(article.content_html, "");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn round_trip_bincode() {
|
||||
let article = Article {
|
||||
sys_title: "test".into(),
|
||||
title: "Test".into(),
|
||||
published: true,
|
||||
author: "Author".into(),
|
||||
tags: vec!["rust".into()],
|
||||
published_on: Some(1000),
|
||||
content_html: "<h1>Hello</h1>".into(),
|
||||
};
|
||||
|
||||
let bytes = bincode::serialize(&article).unwrap();
|
||||
let deserialized: Article = bincode::deserialize(&bytes).unwrap();
|
||||
assert_eq!(article, deserialized);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn legacy_to_article() {
|
||||
let json = r#"{
|
||||
"title": "Legacy Article",
|
||||
"sys_title": "legacy",
|
||||
"published": true,
|
||||
"content": {
|
||||
"author": "Author",
|
||||
"content": {
|
||||
"sections": [
|
||||
{"heading": "Intro", "paragraphs": ["Hello **world**."]}
|
||||
],
|
||||
"codes": [],
|
||||
"images": []
|
||||
},
|
||||
"tags": ["test"],
|
||||
"publishedOn": 1000
|
||||
},
|
||||
"published_on": 2000
|
||||
}"#;
|
||||
|
||||
let legacy: LegacyArticle = serde_json::from_str(json).unwrap();
|
||||
let article = legacy.to_article();
|
||||
assert_eq!(article.sys_title, "legacy");
|
||||
assert_eq!(article.author, "Author");
|
||||
assert_eq!(article.tags, vec!["test"]);
|
||||
assert_eq!(article.published_on, Some(2000));
|
||||
assert!(article.content_html.contains("<h2>Intro</h2>"));
|
||||
assert!(article.content_html.contains("<strong>world</strong>"));
|
||||
}
|
||||
}
|
||||
5
crates/wo-model/src/lib.rs
Normal file
5
crates/wo-model/src/lib.rs
Normal file
|
|
@ -0,0 +1,5 @@
|
|||
mod article;
|
||||
mod loader;
|
||||
|
||||
pub use article::*;
|
||||
pub use loader::ContentLoader;
|
||||
249
crates/wo-model/src/loader.rs
Normal file
249
crates/wo-model/src/loader.rs
Normal file
|
|
@ -0,0 +1,249 @@
|
|||
use std::fs;
|
||||
use std::io;
|
||||
use std::path::Path;
|
||||
|
||||
use crate::{Article, LegacyArticle};
|
||||
|
||||
/// Walks a content directory and loads all articles.
|
||||
///
|
||||
/// Supports two formats:
|
||||
/// - **New**: minimal `.json` metadata + `.md` file for content
|
||||
/// - **Legacy**: full JSON with nested sections/codes (falls back if no `.md`)
|
||||
///
|
||||
/// ```text
|
||||
/// content_dir/
|
||||
/// article-slug/
|
||||
/// article-slug.json # metadata
|
||||
/// article-slug.md # content (new format)
|
||||
/// ```
|
||||
pub struct ContentLoader;
|
||||
|
||||
impl ContentLoader {
|
||||
/// Load all articles from the given content directory.
|
||||
pub fn load_all(content_dir: &Path) -> io::Result<Vec<Article>> {
|
||||
let mut articles = Vec::new();
|
||||
|
||||
let entries = fs::read_dir(content_dir)?;
|
||||
for entry in entries {
|
||||
let entry = entry?;
|
||||
let path = entry.path();
|
||||
|
||||
if !path.is_dir() {
|
||||
continue;
|
||||
}
|
||||
|
||||
match Self::load_from_dir(&path) {
|
||||
Ok(article) => articles.push(article),
|
||||
Err(e) => {
|
||||
eprintln!(
|
||||
"wo-model: skipping {:?}: {}",
|
||||
path.file_name().unwrap_or_default(),
|
||||
e
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
articles.sort_by(|a, b| {
|
||||
b.published_on
|
||||
.unwrap_or(0)
|
||||
.cmp(&a.published_on.unwrap_or(0))
|
||||
});
|
||||
|
||||
Ok(articles)
|
||||
}
|
||||
|
||||
/// Load a single article from a directory.
|
||||
///
|
||||
/// Tries new format first (minimal JSON + .md), falls back to legacy JSON.
|
||||
pub fn load_from_dir(dir: &Path) -> io::Result<Article> {
|
||||
let json_file = Self::find_json_file(dir)?;
|
||||
let json_contents = fs::read_to_string(&json_file)?;
|
||||
|
||||
// Check if a .md file exists alongside the JSON.
|
||||
let md_file = Self::find_md_file(dir);
|
||||
|
||||
if let Some(md_path) = md_file {
|
||||
// New format: minimal JSON + markdown file.
|
||||
// Try parsing as new format first, then fall back to legacy.
|
||||
let mut article: Article =
|
||||
if let Ok(a) = serde_json::from_str::<Article>(&json_contents) {
|
||||
a
|
||||
} else {
|
||||
let legacy: LegacyArticle =
|
||||
serde_json::from_str(&json_contents).map_err(|e| {
|
||||
io::Error::new(
|
||||
io::ErrorKind::InvalidData,
|
||||
format!("{}: {}", json_file.display(), e),
|
||||
)
|
||||
})?;
|
||||
legacy.to_article()
|
||||
};
|
||||
|
||||
let md_content = fs::read_to_string(&md_path)?;
|
||||
article.content_html = wo_md::markdown_to_html_block(&md_content);
|
||||
Ok(article)
|
||||
} else {
|
||||
// Legacy format: full JSON with sections, no .md file.
|
||||
let legacy: LegacyArticle = serde_json::from_str(&json_contents).map_err(|e| {
|
||||
io::Error::new(
|
||||
io::ErrorKind::InvalidData,
|
||||
format!("{}: {}", json_file.display(), e),
|
||||
)
|
||||
})?;
|
||||
Ok(legacy.to_article())
|
||||
}
|
||||
}
|
||||
|
||||
/// Load a single article from a specific JSON file path.
|
||||
pub fn load_from_file(path: &Path) -> io::Result<Article> {
|
||||
let dir = path.parent().unwrap_or(Path::new("."));
|
||||
Self::load_from_dir(dir)
|
||||
}
|
||||
|
||||
fn find_json_file(dir: &Path) -> io::Result<std::path::PathBuf> {
|
||||
for entry in fs::read_dir(dir)? {
|
||||
let entry = entry?;
|
||||
let path = entry.path();
|
||||
if path.extension().and_then(|e| e.to_str()) == Some("json") {
|
||||
return Ok(path);
|
||||
}
|
||||
}
|
||||
Err(io::Error::new(
|
||||
io::ErrorKind::NotFound,
|
||||
format!("no .json file in {}", dir.display()),
|
||||
))
|
||||
}
|
||||
|
||||
fn find_md_file(dir: &Path) -> Option<std::path::PathBuf> {
|
||||
let entries = fs::read_dir(dir).ok()?;
|
||||
for entry in entries {
|
||||
let entry = entry.ok()?;
|
||||
let path = entry.path();
|
||||
if path.extension().and_then(|e| e.to_str()) == Some("md") {
|
||||
return Some(path);
|
||||
}
|
||||
}
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use std::fs;
|
||||
|
||||
/// Create a new-format article: minimal JSON + .md file.
|
||||
fn create_new_article(base: &Path, sys_title: &str, published: bool, tags: &[&str]) {
|
||||
let dir = base.join(sys_title);
|
||||
fs::create_dir_all(&dir).unwrap();
|
||||
|
||||
let tags_json: Vec<String> = tags.iter().map(|t| format!("\"{}\"", t)).collect();
|
||||
let json = format!(
|
||||
r#"{{
|
||||
"sys_title": "{}",
|
||||
"title": "{}",
|
||||
"published": {},
|
||||
"author": "Test Author",
|
||||
"tags": [{}],
|
||||
"published_on": 2000
|
||||
}}"#,
|
||||
sys_title, sys_title, published, tags_json.join(", ")
|
||||
);
|
||||
fs::write(dir.join(format!("{}.json", sys_title)), json).unwrap();
|
||||
|
||||
let md = format!("# {}\n\nThis is the **content** of the article.\n\n```rust\nfn main() {{}}\n```\n", sys_title);
|
||||
fs::write(dir.join(format!("{}.md", sys_title)), md).unwrap();
|
||||
}
|
||||
|
||||
/// Create a legacy-format article: full JSON, no .md file.
|
||||
fn create_legacy_article(base: &Path, sys_title: &str, published: bool, tags: &[&str]) {
|
||||
let dir = base.join(sys_title);
|
||||
fs::create_dir_all(&dir).unwrap();
|
||||
|
||||
let tags_json: Vec<String> = tags.iter().map(|t| format!("\"{}\"", t)).collect();
|
||||
let json = format!(
|
||||
r#"{{
|
||||
"title": "{}",
|
||||
"sys_title": "{}",
|
||||
"published": {},
|
||||
"content": {{
|
||||
"author": "Test Author",
|
||||
"content": {{
|
||||
"sections": [{{ "heading": "Intro", "paragraphs": ["Hello **world**."] }}],
|
||||
"codes": [],
|
||||
"images": []
|
||||
}},
|
||||
"tags": [{}],
|
||||
"publishedOn": 1000
|
||||
}},
|
||||
"published_on": 2000
|
||||
}}"#,
|
||||
sys_title, sys_title, published, tags_json.join(", ")
|
||||
);
|
||||
fs::write(dir.join(format!("{}.json", sys_title)), json).unwrap();
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn load_new_format() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
create_new_article(tmp.path(), "my-article", true, &["rust"]);
|
||||
|
||||
let article = ContentLoader::load_from_dir(&tmp.path().join("my-article")).unwrap();
|
||||
assert_eq!(article.sys_title, "my-article");
|
||||
assert_eq!(article.author, "Test Author");
|
||||
assert_eq!(article.tags, vec!["rust"]);
|
||||
assert!(article.content_html.contains("<h1>my-article</h1>"));
|
||||
assert!(article.content_html.contains("<strong>content</strong>"));
|
||||
assert!(article.content_html.contains("<pre><code"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn load_legacy_format() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
create_legacy_article(tmp.path(), "legacy-art", true, &["test"]);
|
||||
|
||||
let article = ContentLoader::load_from_dir(&tmp.path().join("legacy-art")).unwrap();
|
||||
assert_eq!(article.sys_title, "legacy-art");
|
||||
assert_eq!(article.author, "Test Author");
|
||||
assert!(article.content_html.contains("<h2>Intro</h2>"));
|
||||
assert!(article.content_html.contains("<strong>world</strong>"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn load_all_mixed() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
create_new_article(tmp.path(), "new-art", true, &["rust"]);
|
||||
create_legacy_article(tmp.path(), "old-art", true, &["go"]);
|
||||
|
||||
let articles = ContentLoader::load_all(tmp.path()).unwrap();
|
||||
assert_eq!(articles.len(), 2);
|
||||
|
||||
let new = articles.iter().find(|a| a.sys_title == "new-art").unwrap();
|
||||
assert!(new.content_html.contains("<h1>"));
|
||||
|
||||
let old = articles.iter().find(|a| a.sys_title == "old-art").unwrap();
|
||||
assert!(old.content_html.contains("<h2>Intro</h2>"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn skip_non_directory_entries() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
create_new_article(tmp.path(), "valid-article", true, &[]);
|
||||
fs::write(tmp.path().join("README.md"), "not an article").unwrap();
|
||||
|
||||
let articles = ContentLoader::load_all(tmp.path()).unwrap();
|
||||
assert_eq!(articles.len(), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn skip_directory_without_json() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
create_new_article(tmp.path(), "valid-article", true, &[]);
|
||||
fs::create_dir_all(tmp.path().join("images")).unwrap();
|
||||
fs::write(tmp.path().join("images/photo.png"), b"fake image").unwrap();
|
||||
|
||||
let articles = ContentLoader::load_all(tmp.path()).unwrap();
|
||||
assert_eq!(articles.len(), 1);
|
||||
}
|
||||
}
|
||||
7
crates/wo-route/Cargo.toml
Normal file
7
crates/wo-route/Cargo.toml
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
[package]
|
||||
name = "wo-route"
|
||||
version = "0.1.0"
|
||||
edition = "2021"
|
||||
|
||||
[dependencies]
|
||||
wo-http = { path = "../wo-http" }
|
||||
5
crates/wo-route/src/lib.rs
Normal file
5
crates/wo-route/src/lib.rs
Normal file
|
|
@ -0,0 +1,5 @@
|
|||
mod pattern;
|
||||
mod router;
|
||||
|
||||
pub use pattern::{Pattern, Segment};
|
||||
pub use router::{RouteParams, Router};
|
||||
146
crates/wo-route/src/pattern.rs
Normal file
146
crates/wo-route/src/pattern.rs
Normal file
|
|
@ -0,0 +1,146 @@
|
|||
/// A compiled URL pattern segment.
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub enum Segment {
|
||||
/// Exact literal match (e.g., "blog").
|
||||
Literal(String),
|
||||
/// Named parameter (e.g., ":sys_title").
|
||||
Param(String),
|
||||
/// Wildcard matching the rest of the path (e.g., "*path").
|
||||
Wildcard(String),
|
||||
}
|
||||
|
||||
/// A compiled URL pattern like "/blog/:sys_title" or "/static/*path".
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct Pattern {
|
||||
pub segments: Vec<Segment>,
|
||||
}
|
||||
|
||||
impl Pattern {
|
||||
/// Compile a pattern string into segments.
|
||||
///
|
||||
/// - `/blog/:sys_title` → `[Literal("blog"), Param("sys_title")]`
|
||||
/// - `/static/*path` → `[Literal("static"), Wildcard("path")]`
|
||||
/// - `/` → `[]`
|
||||
pub fn compile(pattern: &str) -> Self {
|
||||
let segments = pattern
|
||||
.trim_start_matches('/')
|
||||
.split('/')
|
||||
.filter(|s| !s.is_empty())
|
||||
.map(|s| {
|
||||
if let Some(name) = s.strip_prefix(':') {
|
||||
Segment::Param(name.to_string())
|
||||
} else if let Some(name) = s.strip_prefix('*') {
|
||||
Segment::Wildcard(name.to_string())
|
||||
} else {
|
||||
Segment::Literal(s.to_string())
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
|
||||
Self { segments }
|
||||
}
|
||||
|
||||
/// Try to match a URL path against this pattern.
|
||||
///
|
||||
/// Returns `Some(params)` if the path matches, where params is a list
|
||||
/// of `(name, value)` pairs for any `:param` or `*wildcard` segments.
|
||||
pub fn matches(&self, path: &str) -> Option<Vec<(String, String)>> {
|
||||
let path_segments: Vec<&str> = path
|
||||
.trim_start_matches('/')
|
||||
.split('/')
|
||||
.filter(|s| !s.is_empty())
|
||||
.collect();
|
||||
|
||||
let mut params = Vec::new();
|
||||
let mut pi = 0; // path segment index
|
||||
|
||||
for seg in &self.segments {
|
||||
match seg {
|
||||
Segment::Literal(lit) => {
|
||||
if pi >= path_segments.len() || path_segments[pi] != lit.as_str() {
|
||||
return None;
|
||||
}
|
||||
pi += 1;
|
||||
}
|
||||
Segment::Param(name) => {
|
||||
if pi >= path_segments.len() {
|
||||
return None;
|
||||
}
|
||||
params.push((name.clone(), path_segments[pi].to_string()));
|
||||
pi += 1;
|
||||
}
|
||||
Segment::Wildcard(name) => {
|
||||
if pi >= path_segments.len() {
|
||||
return None;
|
||||
}
|
||||
let rest = path_segments[pi..].join("/");
|
||||
params.push((name.clone(), rest));
|
||||
return Some(params);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// All pattern segments consumed; path must also be fully consumed.
|
||||
if pi == path_segments.len() {
|
||||
Some(params)
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn root_path() {
|
||||
let p = Pattern::compile("/");
|
||||
assert!(p.matches("/").is_some());
|
||||
assert!(p.matches("/blog").is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn literal_path() {
|
||||
let p = Pattern::compile("/about");
|
||||
assert!(p.matches("/about").is_some());
|
||||
assert!(p.matches("/contact").is_none());
|
||||
assert!(p.matches("/about/extra").is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn param_extraction() {
|
||||
let p = Pattern::compile("/blog/:sys_title");
|
||||
let params = p.matches("/blog/linux-misc").unwrap();
|
||||
assert_eq!(params, vec![("sys_title".into(), "linux-misc".into())]);
|
||||
|
||||
assert!(p.matches("/blog").is_none());
|
||||
assert!(p.matches("/blog/linux-misc/extra").is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wildcard() {
|
||||
let p = Pattern::compile("/static/*path");
|
||||
let params = p.matches("/static/styles/main.css").unwrap();
|
||||
assert_eq!(params, vec![("path".into(), "styles/main.css".into())]);
|
||||
|
||||
let params = p.matches("/static/logo.png").unwrap();
|
||||
assert_eq!(params, vec![("path".into(), "logo.png".into())]);
|
||||
|
||||
assert!(p.matches("/static").is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn multi_segment() {
|
||||
let p = Pattern::compile("/tag/:tag");
|
||||
let params = p.matches("/tag/rust").unwrap();
|
||||
assert_eq!(params, vec![("tag".into(), "rust".into())]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn no_match() {
|
||||
let p = Pattern::compile("/blog/:sys_title");
|
||||
assert!(p.matches("/about").is_none());
|
||||
assert!(p.matches("/").is_none());
|
||||
}
|
||||
}
|
||||
127
crates/wo-route/src/router.rs
Normal file
127
crates/wo-route/src/router.rs
Normal file
|
|
@ -0,0 +1,127 @@
|
|||
use std::collections::HashMap;
|
||||
|
||||
use wo_http::request::{Method, Request};
|
||||
|
||||
use crate::pattern::Pattern;
|
||||
|
||||
/// Extracted route parameters.
|
||||
#[derive(Debug, Clone, Default)]
|
||||
pub struct RouteParams {
|
||||
params: HashMap<String, String>,
|
||||
}
|
||||
|
||||
impl RouteParams {
|
||||
pub fn get(&self, key: &str) -> Option<&str> {
|
||||
self.params.get(key).map(|s| s.as_str())
|
||||
}
|
||||
|
||||
pub fn from_pairs(pairs: Vec<(String, String)>) -> Self {
|
||||
Self {
|
||||
params: pairs.into_iter().collect(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// A route entry: method + pattern + handler name.
|
||||
struct Route {
|
||||
method: Method,
|
||||
pattern: Pattern,
|
||||
handler: String,
|
||||
}
|
||||
|
||||
/// URL router that matches requests to named handlers.
|
||||
pub struct Router {
|
||||
routes: Vec<Route>,
|
||||
}
|
||||
|
||||
impl Router {
|
||||
pub fn new() -> Self {
|
||||
Self { routes: Vec::new() }
|
||||
}
|
||||
|
||||
/// Add a route. Handler is a string name that the caller maps to a function.
|
||||
pub fn add(&mut self, method: Method, pattern: &str, handler: &str) {
|
||||
self.routes.push(Route {
|
||||
method,
|
||||
pattern: Pattern::compile(pattern),
|
||||
handler: handler.to_string(),
|
||||
});
|
||||
}
|
||||
|
||||
/// Match a request to a route. Returns the handler name and extracted params.
|
||||
pub fn dispatch(&self, request: &Request) -> Option<(String, RouteParams)> {
|
||||
for route in &self.routes {
|
||||
if route.method != request.method {
|
||||
continue;
|
||||
}
|
||||
if let Some(pairs) = route.pattern.matches(&request.path) {
|
||||
return Some((
|
||||
route.handler.clone(),
|
||||
RouteParams::from_pairs(pairs),
|
||||
));
|
||||
}
|
||||
}
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn make_request(method: Method, path: &str) -> Request {
|
||||
Request {
|
||||
method,
|
||||
path: path.to_string(),
|
||||
query: None,
|
||||
headers: HashMap::new(),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dispatch_routes() {
|
||||
let mut router = Router::new();
|
||||
router.add(Method::Get, "/", "home");
|
||||
router.add(Method::Get, "/blog/:sys_title", "article");
|
||||
router.add(Method::Get, "/about", "about");
|
||||
router.add(Method::Get, "/tag/:tag", "tag_listing");
|
||||
router.add(Method::Get, "/static/*path", "static_file");
|
||||
|
||||
// Home.
|
||||
let (handler, _) = router.dispatch(&make_request(Method::Get, "/")).unwrap();
|
||||
assert_eq!(handler, "home");
|
||||
|
||||
// Article.
|
||||
let (handler, params) = router
|
||||
.dispatch(&make_request(Method::Get, "/blog/linux-misc"))
|
||||
.unwrap();
|
||||
assert_eq!(handler, "article");
|
||||
assert_eq!(params.get("sys_title"), Some("linux-misc"));
|
||||
|
||||
// About.
|
||||
let (handler, _) = router.dispatch(&make_request(Method::Get, "/about")).unwrap();
|
||||
assert_eq!(handler, "about");
|
||||
|
||||
// Tag.
|
||||
let (handler, params) = router
|
||||
.dispatch(&make_request(Method::Get, "/tag/rust"))
|
||||
.unwrap();
|
||||
assert_eq!(handler, "tag_listing");
|
||||
assert_eq!(params.get("tag"), Some("rust"));
|
||||
|
||||
// Static.
|
||||
let (handler, params) = router
|
||||
.dispatch(&make_request(Method::Get, "/static/styles/main.css"))
|
||||
.unwrap();
|
||||
assert_eq!(handler, "static_file");
|
||||
assert_eq!(params.get("path"), Some("styles/main.css"));
|
||||
|
||||
// No match.
|
||||
assert!(router
|
||||
.dispatch(&make_request(Method::Get, "/nonexistent"))
|
||||
.is_none());
|
||||
|
||||
// Wrong method.
|
||||
assert!(router.dispatch(&make_request(Method::Post, "/")).is_none());
|
||||
}
|
||||
}
|
||||
28
crates/wo-rt/Cargo.toml
Normal file
28
crates/wo-rt/Cargo.toml
Normal file
|
|
@ -0,0 +1,28 @@
|
|||
[package]
|
||||
name = "wo-rt"
|
||||
version = "0.1.0"
|
||||
edition = "2021"
|
||||
|
||||
[dependencies]
|
||||
wo-model = { path = "../wo-model" }
|
||||
wo-store = { path = "../wo-store" }
|
||||
wo-watch = { path = "../wo-watch" }
|
||||
wo-event = { path = "../wo-event" }
|
||||
wo-sub = { path = "../wo-sub" }
|
||||
wo-http = { path = "../wo-http" }
|
||||
wo-route = { path = "../wo-route" }
|
||||
wo-htmlx = { path = "../wo-htmlx" }
|
||||
wo-md = { path = "../wo-md" }
|
||||
wo-serve = { path = "../wo-serve" }
|
||||
libc = { workspace = true }
|
||||
serde = { workspace = true }
|
||||
serde_json = { workspace = true }
|
||||
bincode = { workspace = true }
|
||||
|
||||
[[bin]]
|
||||
name = "writeonce"
|
||||
path = "src/main.rs"
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile = "3"
|
||||
bincode = { workspace = true }
|
||||
31
crates/wo-rt/src/config.rs
Normal file
31
crates/wo-rt/src/config.rs
Normal file
|
|
@ -0,0 +1,31 @@
|
|||
use std::path::PathBuf;
|
||||
|
||||
/// Runtime configuration.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct Config {
|
||||
/// Path to the content directory (JSON + MD source files).
|
||||
pub content_dir: PathBuf,
|
||||
/// Path to the data directory (.seg + .idx derived files).
|
||||
pub data_dir: PathBuf,
|
||||
/// Path to the templates directory (.htmlx files).
|
||||
pub templates_dir: PathBuf,
|
||||
/// Path to the static assets directory (CSS, images).
|
||||
pub static_dir: PathBuf,
|
||||
/// HTTP bind address (e.g., "0.0.0.0:3000").
|
||||
pub bind_addr: String,
|
||||
/// Force a full rebuild on startup (ignore existing data/).
|
||||
pub rebuild_on_start: bool,
|
||||
}
|
||||
|
||||
impl Config {
|
||||
pub fn new(content_dir: impl Into<PathBuf>, data_dir: impl Into<PathBuf>) -> Self {
|
||||
Self {
|
||||
content_dir: content_dir.into(),
|
||||
data_dir: data_dir.into(),
|
||||
templates_dir: PathBuf::from("templates"),
|
||||
static_dir: PathBuf::from("static"),
|
||||
bind_addr: "0.0.0.0:3000".into(),
|
||||
rebuild_on_start: false,
|
||||
}
|
||||
}
|
||||
}
|
||||
130
crates/wo-rt/src/handlers.rs
Normal file
130
crates/wo-rt/src/handlers.rs
Normal file
|
|
@ -0,0 +1,130 @@
|
|||
use wo_htmlx::{self, Value};
|
||||
use wo_http::response::Response;
|
||||
use wo_route::RouteParams;
|
||||
use wo_store::Store;
|
||||
use wo_htmlx::TemplateRegistry;
|
||||
|
||||
/// Render the homepage: list of published articles.
|
||||
pub fn handle_home(store: &Store, templates: &TemplateRegistry) -> Response {
|
||||
let articles = match store.list_published(0, 20) {
|
||||
Ok(a) => a,
|
||||
Err(e) => return Response::internal_error(&e.to_string()),
|
||||
};
|
||||
|
||||
let articles_val: Vec<Value> = articles
|
||||
.iter()
|
||||
.map(|a| {
|
||||
let json = serde_json::to_value(a).unwrap_or_default();
|
||||
Value::from_json(&json)
|
||||
})
|
||||
.collect();
|
||||
|
||||
let mut ctx = std::collections::BTreeMap::new();
|
||||
ctx.insert("articles".into(), Value::List(articles_val));
|
||||
ctx.insert("page_title".into(), Value::String("writeonce".into()));
|
||||
let context = Value::Object(ctx);
|
||||
|
||||
render_page("home", &context, templates)
|
||||
}
|
||||
|
||||
/// Render a single article page.
|
||||
pub fn handle_article(params: &RouteParams, store: &Store, templates: &TemplateRegistry) -> Response {
|
||||
let sys_title = match params.get("sys_title") {
|
||||
Some(t) => t,
|
||||
None => return Response::not_found(),
|
||||
};
|
||||
|
||||
let article = match store.get_by_title(sys_title) {
|
||||
Ok(Some(a)) => a,
|
||||
Ok(None) => return Response::not_found(),
|
||||
Err(e) => return Response::internal_error(&e.to_string()),
|
||||
};
|
||||
|
||||
let json = serde_json::to_value(&article).unwrap_or_default();
|
||||
let mut ctx = std::collections::BTreeMap::new();
|
||||
ctx.insert("article".into(), Value::from_json(&json));
|
||||
ctx.insert("page_title".into(), Value::String(article.title.clone()));
|
||||
let context = Value::Object(ctx);
|
||||
|
||||
render_page("article", &context, templates)
|
||||
}
|
||||
|
||||
/// Render a tag listing page.
|
||||
pub fn handle_tag(params: &RouteParams, store: &Store, templates: &TemplateRegistry) -> Response {
|
||||
let tag = match params.get("tag") {
|
||||
Some(t) => t,
|
||||
None => return Response::not_found(),
|
||||
};
|
||||
|
||||
let articles = match store.list_by_tag(tag) {
|
||||
Ok(a) => a,
|
||||
Err(e) => return Response::internal_error(&e.to_string()),
|
||||
};
|
||||
|
||||
let articles_val: Vec<Value> = articles
|
||||
.iter()
|
||||
.map(|a| {
|
||||
let json = serde_json::to_value(a).unwrap_or_default();
|
||||
Value::from_json(&json)
|
||||
})
|
||||
.collect();
|
||||
|
||||
let mut ctx = std::collections::BTreeMap::new();
|
||||
ctx.insert("articles".into(), Value::List(articles_val));
|
||||
ctx.insert("page_title".into(), Value::String(format!("tag: {}", tag)));
|
||||
ctx.insert("tag".into(), Value::String(tag.to_string()));
|
||||
let context = Value::Object(ctx);
|
||||
|
||||
render_page("home", &context, templates)
|
||||
}
|
||||
|
||||
/// Render a static page (about, contact).
|
||||
pub fn handle_static_page(name: &str, templates: &TemplateRegistry) -> Response {
|
||||
let mut ctx = std::collections::BTreeMap::new();
|
||||
ctx.insert("page_title".into(), Value::String(name.to_string()));
|
||||
let context = Value::Object(ctx);
|
||||
|
||||
render_page(name, &context, templates)
|
||||
}
|
||||
|
||||
/// Render a page template composed with layout, header, and footer.
|
||||
fn render_page(template_name: &str, context: &Value, templates: &TemplateRegistry) -> Response {
|
||||
let page_nodes = match templates.get(template_name) {
|
||||
Some(nodes) => nodes,
|
||||
None => return Response::not_found(),
|
||||
};
|
||||
|
||||
let partials = templates.partials();
|
||||
|
||||
// Render the page content.
|
||||
let page_html = wo_htmlx::render(page_nodes, context, partials);
|
||||
|
||||
// Compose with layout if it exists.
|
||||
let html = if let Some(layout_nodes) = templates.get("layout") {
|
||||
// Inject page content and header/footer.
|
||||
let header_html = templates
|
||||
.get("header")
|
||||
.map(|n| wo_htmlx::render(n, context, partials))
|
||||
.unwrap_or_default();
|
||||
let footer_html = templates
|
||||
.get("footer")
|
||||
.map(|n| wo_htmlx::render(n, context, partials))
|
||||
.unwrap_or_default();
|
||||
|
||||
let mut layout_ctx = std::collections::BTreeMap::new();
|
||||
layout_ctx.insert("content".into(), Value::String(page_html));
|
||||
layout_ctx.insert("header".into(), Value::String(header_html));
|
||||
layout_ctx.insert("footer".into(), Value::String(footer_html));
|
||||
if let Value::Object(map) = context {
|
||||
for (k, v) in map {
|
||||
layout_ctx.insert(k.clone(), v.clone());
|
||||
}
|
||||
}
|
||||
let layout_context = Value::Object(layout_ctx);
|
||||
wo_htmlx::render(layout_nodes, &layout_context, partials)
|
||||
} else {
|
||||
page_html
|
||||
};
|
||||
|
||||
Response::html(html)
|
||||
}
|
||||
6
crates/wo-rt/src/lib.rs
Normal file
6
crates/wo-rt/src/lib.rs
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
mod config;
|
||||
mod handlers;
|
||||
mod runtime;
|
||||
|
||||
pub use config::Config;
|
||||
pub use runtime::{Runtime, RuntimeHandle};
|
||||
34
crates/wo-rt/src/main.rs
Normal file
34
crates/wo-rt/src/main.rs
Normal file
|
|
@ -0,0 +1,34 @@
|
|||
use std::path::PathBuf;
|
||||
use wo_rt::{Config, Runtime};
|
||||
|
||||
fn main() {
|
||||
let content_dir = PathBuf::from("content");
|
||||
let data_dir = PathBuf::from("data");
|
||||
|
||||
let config = Config {
|
||||
content_dir,
|
||||
data_dir,
|
||||
templates_dir: PathBuf::from("templates"),
|
||||
static_dir: PathBuf::from("static"),
|
||||
bind_addr: "0.0.0.0:3000".into(),
|
||||
rebuild_on_start: true,
|
||||
};
|
||||
|
||||
let mut rt = match Runtime::new(&config) {
|
||||
Ok(rt) => rt,
|
||||
Err(e) => {
|
||||
eprintln!("failed to start: {}", e);
|
||||
std::process::exit(1);
|
||||
}
|
||||
};
|
||||
|
||||
eprintln!("writeonce started — content={} data={}",
|
||||
config.content_dir.display(), config.data_dir.display());
|
||||
|
||||
if let Err(e) = rt.run() {
|
||||
eprintln!("runtime error: {}", e);
|
||||
std::process::exit(1);
|
||||
}
|
||||
|
||||
eprintln!("writeonce shut down");
|
||||
}
|
||||
481
crates/wo-rt/src/runtime.rs
Normal file
481
crates/wo-rt/src/runtime.rs
Normal file
|
|
@ -0,0 +1,481 @@
|
|||
use std::collections::HashMap;
|
||||
use std::io;
|
||||
use std::os::unix::io::RawFd;
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
use std::sync::Arc;
|
||||
use std::time::Duration;
|
||||
|
||||
use wo_event::{EventLoop, Interest, SignalFd, TimerFd, Token};
|
||||
use wo_htmlx::TemplateRegistry;
|
||||
use wo_http::connection::{Connection, State};
|
||||
use wo_http::listener::TcpListener;
|
||||
use wo_http::request::Method;
|
||||
use wo_http::response::Response;
|
||||
use wo_route::{RouteParams, Router};
|
||||
use wo_store::Store;
|
||||
use wo_sub::{ChangeType, Subscription, SubscriptionManager};
|
||||
use wo_watch::{ContentChange, ContentWatcher};
|
||||
|
||||
use crate::handlers;
|
||||
use crate::Config;
|
||||
|
||||
// Well-known tokens.
|
||||
const TOKEN_WATCHER: Token = 1;
|
||||
const TOKEN_SIGNAL: Token = 2;
|
||||
const TOKEN_TIMER: Token = 3;
|
||||
const TOKEN_NOTIFY: Token = 4;
|
||||
const TOKEN_HTTP_LISTENER: Token = 5;
|
||||
// Token ranges.
|
||||
const TOKEN_HTTP_BASE: Token = 10000;
|
||||
const TOKEN_HTTP_MAX: Token = 19999;
|
||||
const TOKEN_SUB_BASE: Token = 20000;
|
||||
|
||||
/// The writeonce runtime: single process, single event loop, all fds on one epoll.
|
||||
pub struct Runtime {
|
||||
event_loop: EventLoop,
|
||||
store: Store,
|
||||
watcher: ContentWatcher,
|
||||
subscriptions: SubscriptionManager,
|
||||
signal_fd: SignalFd,
|
||||
timer_fd: TimerFd,
|
||||
running: Arc<AtomicBool>,
|
||||
// HTTP
|
||||
listener: Option<TcpListener>,
|
||||
connections: HashMap<Token, Connection>,
|
||||
next_http_token: Token,
|
||||
router: Router,
|
||||
templates: TemplateRegistry,
|
||||
static_dir: std::path::PathBuf,
|
||||
// Subscribed browser connections.
|
||||
next_sub_token: Token,
|
||||
sub_token_to_fd: HashMap<Token, RawFd>,
|
||||
}
|
||||
|
||||
/// Handle returned by `Runtime::start()` for external interaction.
|
||||
pub struct RuntimeHandle {
|
||||
running: Arc<AtomicBool>,
|
||||
}
|
||||
|
||||
impl RuntimeHandle {
|
||||
pub fn shutdown(&self) {
|
||||
self.running.store(false, Ordering::SeqCst);
|
||||
}
|
||||
}
|
||||
|
||||
impl Runtime {
|
||||
pub fn new(config: &Config) -> io::Result<Self> {
|
||||
let event_loop = EventLoop::new()?;
|
||||
|
||||
let store = if config.rebuild_on_start {
|
||||
let mut s = Store::open(&config.content_dir, &config.data_dir)?;
|
||||
s.rebuild()?;
|
||||
s
|
||||
} else {
|
||||
Store::open(&config.content_dir, &config.data_dir)?
|
||||
};
|
||||
|
||||
let watcher = ContentWatcher::new(&config.content_dir)?;
|
||||
let subscriptions = SubscriptionManager::new()?;
|
||||
let signal_fd = SignalFd::new()?;
|
||||
let timer_fd = TimerFd::new()?;
|
||||
timer_fd.set(Duration::from_secs(60), Duration::from_secs(60))?;
|
||||
|
||||
// Register system fds.
|
||||
event_loop.register(watcher.fd(), Interest::Readable, TOKEN_WATCHER)?;
|
||||
event_loop.register(signal_fd.fd(), Interest::Readable, TOKEN_SIGNAL)?;
|
||||
event_loop.register(timer_fd.fd(), Interest::Readable, TOKEN_TIMER)?;
|
||||
event_loop.register(subscriptions.notify_fd(), Interest::Readable, TOKEN_NOTIFY)?;
|
||||
|
||||
// HTTP listener (optional — may not bind in test mode).
|
||||
let listener = if !config.bind_addr.is_empty() {
|
||||
match TcpListener::bind(&config.bind_addr) {
|
||||
Ok(l) => {
|
||||
event_loop.register(l.fd(), Interest::Readable, TOKEN_HTTP_LISTENER)?;
|
||||
Some(l)
|
||||
}
|
||||
Err(e) => {
|
||||
eprintln!("wo-rt: failed to bind {}: {}", config.bind_addr, e);
|
||||
None
|
||||
}
|
||||
}
|
||||
} else {
|
||||
None
|
||||
};
|
||||
|
||||
// Load templates.
|
||||
let templates = TemplateRegistry::load(&config.templates_dir)?;
|
||||
|
||||
// Build router.
|
||||
let mut router = Router::new();
|
||||
router.add(Method::Get, "/", "home");
|
||||
router.add(Method::Get, "/blog/:sys_title", "article");
|
||||
router.add(Method::Get, "/about", "about");
|
||||
router.add(Method::Get, "/contact", "contact");
|
||||
router.add(Method::Get, "/tag/:tag", "tag_listing");
|
||||
router.add(Method::Get, "/static/*path", "static_file");
|
||||
|
||||
Ok(Self {
|
||||
event_loop,
|
||||
store,
|
||||
watcher,
|
||||
subscriptions,
|
||||
signal_fd,
|
||||
timer_fd,
|
||||
running: Arc::new(AtomicBool::new(false)),
|
||||
listener,
|
||||
connections: HashMap::new(),
|
||||
next_http_token: TOKEN_HTTP_BASE,
|
||||
router,
|
||||
templates,
|
||||
static_dir: config.static_dir.clone(),
|
||||
next_sub_token: TOKEN_SUB_BASE,
|
||||
sub_token_to_fd: HashMap::new(),
|
||||
})
|
||||
}
|
||||
|
||||
pub fn run(&mut self) -> io::Result<()> {
|
||||
self.running.store(true, Ordering::SeqCst);
|
||||
|
||||
if let Some(ref listener) = self.listener {
|
||||
let (_, port) = listener.local_addr()?;
|
||||
eprintln!("writeonce listening on port {}", port);
|
||||
}
|
||||
|
||||
while self.running.load(Ordering::SeqCst) {
|
||||
let events = self.event_loop.poll(Some(Duration::from_millis(500)))?;
|
||||
|
||||
for event in events {
|
||||
match event.token {
|
||||
TOKEN_WATCHER if event.readable => {
|
||||
self.handle_watcher_event()?;
|
||||
}
|
||||
TOKEN_SIGNAL if event.readable => {
|
||||
let _ = self.signal_fd.read();
|
||||
self.running.store(false, Ordering::SeqCst);
|
||||
}
|
||||
TOKEN_TIMER if event.readable => {
|
||||
let _ = self.timer_fd.read();
|
||||
}
|
||||
TOKEN_NOTIFY if event.readable => {
|
||||
// Notifications already written by SubscriptionManager.
|
||||
}
|
||||
TOKEN_HTTP_LISTENER if event.readable => {
|
||||
self.handle_accept()?;
|
||||
}
|
||||
token if token >= TOKEN_SUB_BASE => {
|
||||
// Subscribed connection.
|
||||
if event.hangup || event.error {
|
||||
if let Some(&fd) = self.sub_token_to_fd.get(&token) {
|
||||
self.subscriptions.unsubscribe(fd);
|
||||
let _ = self.event_loop.deregister(fd);
|
||||
self.sub_token_to_fd.remove(&token);
|
||||
// Don't close fd here — Connection owns it.
|
||||
}
|
||||
}
|
||||
}
|
||||
token if token >= TOKEN_HTTP_BASE => {
|
||||
self.handle_http_event(token, &event)?;
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
|
||||
// Clean up completed connections.
|
||||
let done_tokens: Vec<Token> = self
|
||||
.connections
|
||||
.iter()
|
||||
.filter(|(_, c)| c.state == State::Done)
|
||||
.map(|(&t, _)| t)
|
||||
.collect();
|
||||
for token in done_tokens {
|
||||
if let Some(conn) = self.connections.remove(&token) {
|
||||
let _ = self.event_loop.deregister(conn.fd);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn handle(&self) -> RuntimeHandle {
|
||||
RuntimeHandle {
|
||||
running: self.running.clone(),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn store(&self) -> &Store {
|
||||
&self.store
|
||||
}
|
||||
|
||||
pub fn store_mut(&mut self) -> &mut Store {
|
||||
&mut self.store
|
||||
}
|
||||
|
||||
/// Subscribe a fd (binary format, for internal use).
|
||||
pub fn subscribe(&mut self, fd: RawFd, query: Subscription) -> io::Result<Token> {
|
||||
let token = self.next_sub_token;
|
||||
self.next_sub_token += 1;
|
||||
self.event_loop.register(fd, Interest::Readable, token)?;
|
||||
self.subscriptions.subscribe(fd, query);
|
||||
self.sub_token_to_fd.insert(token, fd);
|
||||
Ok(token)
|
||||
}
|
||||
|
||||
fn handle_accept(&mut self) -> io::Result<()> {
|
||||
let listener = match &self.listener {
|
||||
Some(l) => l,
|
||||
None => return Ok(()),
|
||||
};
|
||||
|
||||
// Accept all pending connections.
|
||||
while let Some(client_fd) = listener.accept()? {
|
||||
if self.next_http_token > TOKEN_HTTP_MAX {
|
||||
self.next_http_token = TOKEN_HTTP_BASE; // recycle tokens
|
||||
}
|
||||
let token = self.next_http_token;
|
||||
self.next_http_token += 1;
|
||||
|
||||
self.event_loop
|
||||
.register(client_fd, Interest::Readable, token)?;
|
||||
self.connections.insert(token, Connection::new(client_fd));
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn handle_http_event(&mut self, token: Token, event: &wo_event::Event) -> io::Result<()> {
|
||||
if event.hangup || event.error {
|
||||
if let Some(conn) = self.connections.remove(&token) {
|
||||
let _ = self.event_loop.deregister(conn.fd);
|
||||
}
|
||||
return Ok(());
|
||||
}
|
||||
|
||||
// Read phase: parse the request without holding mutable borrow during dispatch.
|
||||
let conn = match self.connections.get_mut(&token) {
|
||||
Some(c) => c,
|
||||
None => return Ok(()),
|
||||
};
|
||||
|
||||
match conn.state {
|
||||
State::ReadingRequest if event.readable => {
|
||||
let open = conn.read()?;
|
||||
if !open {
|
||||
conn.set_done();
|
||||
return Ok(());
|
||||
}
|
||||
|
||||
if let Some(request) = conn.try_parse() {
|
||||
// Dispatch the handler (borrows self immutably via store/templates).
|
||||
let dispatch_result = self.router.dispatch(&request);
|
||||
let response = match dispatch_result {
|
||||
Some((ref handler, ref params)) => {
|
||||
dispatch_handler(handler, params, &self.store, &self.templates, &self.static_dir)
|
||||
}
|
||||
None => Response::not_found(),
|
||||
};
|
||||
|
||||
// Now get the connection again to set the response.
|
||||
let conn = self.connections.get_mut(&token).unwrap();
|
||||
conn.set_response(&response);
|
||||
if conn.write()? {
|
||||
conn.set_done();
|
||||
} else {
|
||||
let fd = conn.fd;
|
||||
self.event_loop.modify(fd, Interest::Writable, token)?;
|
||||
}
|
||||
}
|
||||
}
|
||||
State::WritingResponse if event.writable => {
|
||||
if conn.write()? {
|
||||
conn.set_done();
|
||||
}
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn handle_watcher_event(&mut self) -> io::Result<()> {
|
||||
let changes = self.watcher.process_events()?;
|
||||
|
||||
for change in &changes {
|
||||
match change {
|
||||
ContentChange::Created(sys_title) | ContentChange::Modified(sys_title) => {
|
||||
self.store.rebuild()?;
|
||||
let version = self.store.article_version(sys_title).unwrap_or(0);
|
||||
let tags = self
|
||||
.store
|
||||
.get_by_title(sys_title)?
|
||||
.map(|a| a.tags.clone())
|
||||
.unwrap_or_default();
|
||||
let change_type = match change {
|
||||
ContentChange::Created(_) => ChangeType::Created,
|
||||
_ => ChangeType::Updated,
|
||||
};
|
||||
self.subscriptions.notify(sys_title, &tags, change_type, version)?;
|
||||
}
|
||||
ContentChange::Deleted(sys_title) => {
|
||||
let version = self.store.article_version(sys_title).unwrap_or(0);
|
||||
self.subscriptions.notify(sys_title, &[], ChangeType::Deleted, version)?;
|
||||
self.store.rebuild()?;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
fn dispatch_handler(
|
||||
handler: &str,
|
||||
params: &RouteParams,
|
||||
store: &Store,
|
||||
templates: &TemplateRegistry,
|
||||
static_dir: &std::path::Path,
|
||||
) -> Response {
|
||||
match handler {
|
||||
"home" => handlers::handle_home(store, templates),
|
||||
"article" => handlers::handle_article(params, store, templates),
|
||||
"tag_listing" => handlers::handle_tag(params, store, templates),
|
||||
"about" => handlers::handle_static_page("about", templates),
|
||||
"contact" => handlers::handle_static_page("contact", templates),
|
||||
"static_file" => {
|
||||
let path = params.get("path").unwrap_or("");
|
||||
wo_serve::send_file(static_dir, path)
|
||||
}
|
||||
_ => Response::not_found(),
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use std::fs;
|
||||
|
||||
fn create_test_article(base: &std::path::Path, sys_title: &str, published: bool) {
|
||||
let dir = base.join(sys_title);
|
||||
fs::create_dir_all(&dir).unwrap();
|
||||
let json = format!(
|
||||
r#"{{
|
||||
"title": "{}",
|
||||
"sys_title": "{}",
|
||||
"published": {},
|
||||
"content": {{
|
||||
"author": "Author",
|
||||
"content": {{
|
||||
"sections": [{{ "heading": "Intro", "paragraphs": ["Hello."] }}],
|
||||
"codes": [],
|
||||
"images": []
|
||||
}},
|
||||
"tags": ["test"],
|
||||
"publishedOn": 1000
|
||||
}},
|
||||
"published_on": 1000
|
||||
}}"#,
|
||||
sys_title, sys_title, published
|
||||
);
|
||||
fs::write(dir.join(format!("{}.json", sys_title)), json).unwrap();
|
||||
}
|
||||
|
||||
fn test_config(content: &std::path::Path, data: &std::path::Path) -> Config {
|
||||
Config {
|
||||
bind_addr: String::new(), // no HTTP in tests
|
||||
..Config::new(content, data)
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn runtime_creates_and_queries() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let content = tmp.path().join("content");
|
||||
let data = tmp.path().join("data");
|
||||
|
||||
create_test_article(&content, "art-one", true);
|
||||
create_test_article(&content, "art-two", true);
|
||||
|
||||
let config = test_config(&content, &data);
|
||||
let rt = Runtime::new(&config).unwrap();
|
||||
|
||||
let art = rt.store().get_by_title("art-one").unwrap().unwrap();
|
||||
assert_eq!(art.sys_title, "art-one");
|
||||
assert_eq!(rt.store().count_published().unwrap(), 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn runtime_subscribe_and_notify() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let content = tmp.path().join("content");
|
||||
let data = tmp.path().join("data");
|
||||
|
||||
create_test_article(&content, "art-one", true);
|
||||
|
||||
let config = test_config(&content, &data);
|
||||
let mut rt = Runtime::new(&config).unwrap();
|
||||
|
||||
let mut fds = [0i32; 2];
|
||||
assert_eq!(unsafe { libc::pipe2(fds.as_mut_ptr(), libc::O_NONBLOCK) }, 0);
|
||||
let (read_fd, write_fd) = (fds[0], fds[1]);
|
||||
|
||||
rt.subscribe(write_fd, Subscription::ByTitle("art-one".into())).unwrap();
|
||||
|
||||
rt.subscriptions
|
||||
.notify("art-one", &["test".into()], ChangeType::Updated, 1)
|
||||
.unwrap();
|
||||
|
||||
let mut len_buf = [0u8; 4];
|
||||
let ret = unsafe { libc::read(read_fd, len_buf.as_mut_ptr() as *mut libc::c_void, 4) };
|
||||
assert_eq!(ret, 4);
|
||||
let len = u32::from_le_bytes(len_buf) as usize;
|
||||
|
||||
let mut payload = vec![0u8; len];
|
||||
let ret = unsafe { libc::read(read_fd, payload.as_mut_ptr() as *mut libc::c_void, len) };
|
||||
assert_eq!(ret as usize, len);
|
||||
|
||||
let notification: wo_sub::Notification = bincode::deserialize(&payload).unwrap();
|
||||
assert_eq!(notification.sys_title, "art-one");
|
||||
|
||||
unsafe { libc::close(read_fd); libc::close(write_fd); }
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn runtime_shutdown_via_handle() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let content = tmp.path().join("content");
|
||||
let data = tmp.path().join("data");
|
||||
fs::create_dir_all(&content).unwrap();
|
||||
|
||||
let config = test_config(&content, &data);
|
||||
let mut rt = Runtime::new(&config).unwrap();
|
||||
let handle = rt.handle();
|
||||
|
||||
std::thread::spawn(move || {
|
||||
std::thread::sleep(Duration::from_millis(100));
|
||||
handle.shutdown();
|
||||
});
|
||||
|
||||
rt.run().unwrap();
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn runtime_rebuild_on_start() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let content = tmp.path().join("content");
|
||||
let data = tmp.path().join("data");
|
||||
|
||||
create_test_article(&content, "art-one", true);
|
||||
|
||||
let config = test_config(&content, &data);
|
||||
let rt = Runtime::new(&config).unwrap();
|
||||
assert_eq!(rt.store().count_published().unwrap(), 1);
|
||||
drop(rt);
|
||||
|
||||
let config = Config {
|
||||
rebuild_on_start: true,
|
||||
..test_config(&content, &data)
|
||||
};
|
||||
let rt = Runtime::new(&config).unwrap();
|
||||
assert_eq!(rt.store().count_published().unwrap(), 1);
|
||||
}
|
||||
}
|
||||
13
crates/wo-seg/Cargo.toml
Normal file
13
crates/wo-seg/Cargo.toml
Normal file
|
|
@ -0,0 +1,13 @@
|
|||
[package]
|
||||
name = "wo-seg"
|
||||
version = "0.1.0"
|
||||
edition = "2021"
|
||||
|
||||
[dependencies]
|
||||
wo-model = { path = "../wo-model" }
|
||||
serde = { workspace = true }
|
||||
bincode = { workspace = true }
|
||||
libc = { workspace = true }
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile = "3"
|
||||
111
crates/wo-seg/src/header.rs
Normal file
111
crates/wo-seg/src/header.rs
Normal file
|
|
@ -0,0 +1,111 @@
|
|||
use std::io;
|
||||
|
||||
/// Magic bytes: "WOSF" (WriteOnce Segment File).
|
||||
pub const MAGIC: [u8; 4] = *b"WOSF";
|
||||
|
||||
/// Segment file version.
|
||||
pub const VERSION: u16 = 1;
|
||||
|
||||
/// Fixed header size in bytes.
|
||||
pub const HEADER_SIZE: u64 = 32;
|
||||
|
||||
/// The segment file header, stored at byte 0 of every .seg file.
|
||||
///
|
||||
/// Layout (32 bytes):
|
||||
/// ```text
|
||||
/// [0..4] magic: b"WOSF"
|
||||
/// [4..6] version: u16 LE
|
||||
/// [6..8] flags: u16 LE (reserved)
|
||||
/// [8..16] record_count: u64 LE
|
||||
/// [16..24] data_start: u64 LE
|
||||
/// [24..32] reserved: 8 bytes
|
||||
/// ```
|
||||
#[derive(Debug, Clone, Copy, PartialEq)]
|
||||
pub struct SegHeader {
|
||||
pub version: u16,
|
||||
pub flags: u16,
|
||||
pub record_count: u64,
|
||||
pub data_start: u64,
|
||||
}
|
||||
|
||||
impl SegHeader {
|
||||
/// Create a new header with default values.
|
||||
pub fn new() -> Self {
|
||||
Self {
|
||||
version: VERSION,
|
||||
flags: 0,
|
||||
record_count: 0,
|
||||
data_start: HEADER_SIZE,
|
||||
}
|
||||
}
|
||||
|
||||
/// Serialize the header into a 32-byte buffer.
|
||||
pub fn to_bytes(&self) -> [u8; 32] {
|
||||
let mut buf = [0u8; 32];
|
||||
buf[0..4].copy_from_slice(&MAGIC);
|
||||
buf[4..6].copy_from_slice(&self.version.to_le_bytes());
|
||||
buf[6..8].copy_from_slice(&self.flags.to_le_bytes());
|
||||
buf[8..16].copy_from_slice(&self.record_count.to_le_bytes());
|
||||
buf[16..24].copy_from_slice(&self.data_start.to_le_bytes());
|
||||
// [24..32] reserved, stays zero
|
||||
buf
|
||||
}
|
||||
|
||||
/// Parse a header from a 32-byte buffer.
|
||||
pub fn from_bytes(buf: &[u8; 32]) -> io::Result<Self> {
|
||||
if &buf[0..4] != &MAGIC {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::InvalidData,
|
||||
format!(
|
||||
"invalid magic bytes: expected {:?}, got {:?}",
|
||||
MAGIC,
|
||||
&buf[0..4]
|
||||
),
|
||||
));
|
||||
}
|
||||
|
||||
let version = u16::from_le_bytes([buf[4], buf[5]]);
|
||||
if version != VERSION {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::InvalidData,
|
||||
format!("unsupported version: {}", version),
|
||||
));
|
||||
}
|
||||
|
||||
let flags = u16::from_le_bytes([buf[6], buf[7]]);
|
||||
let record_count = u64::from_le_bytes(buf[8..16].try_into().unwrap());
|
||||
let data_start = u64::from_le_bytes(buf[16..24].try_into().unwrap());
|
||||
|
||||
Ok(Self {
|
||||
version,
|
||||
flags,
|
||||
record_count,
|
||||
data_start,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn round_trip() {
|
||||
let header = SegHeader {
|
||||
version: VERSION,
|
||||
flags: 0,
|
||||
record_count: 42,
|
||||
data_start: HEADER_SIZE,
|
||||
};
|
||||
let bytes = header.to_bytes();
|
||||
let parsed = SegHeader::from_bytes(&bytes).unwrap();
|
||||
assert_eq!(header, parsed);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn bad_magic() {
|
||||
let mut bytes = SegHeader::new().to_bytes();
|
||||
bytes[0] = b'X';
|
||||
assert!(SegHeader::from_bytes(&bytes).is_err());
|
||||
}
|
||||
}
|
||||
14
crates/wo-seg/src/lib.rs
Normal file
14
crates/wo-seg/src/lib.rs
Normal file
|
|
@ -0,0 +1,14 @@
|
|||
mod header;
|
||||
mod reader;
|
||||
mod writer;
|
||||
|
||||
pub use header::{SegHeader, HEADER_SIZE, MAGIC};
|
||||
pub use reader::SegReader;
|
||||
pub use writer::SegWriter;
|
||||
|
||||
/// Record flags.
|
||||
pub const FLAG_ACTIVE: u8 = 0x00;
|
||||
pub const FLAG_TOMBSTONED: u8 = 0x01;
|
||||
|
||||
/// Size of the per-record header: 4 bytes length + 1 byte flags.
|
||||
pub const RECORD_HEADER_SIZE: usize = 5;
|
||||
221
crates/wo-seg/src/reader.rs
Normal file
221
crates/wo-seg/src/reader.rs
Normal file
|
|
@ -0,0 +1,221 @@
|
|||
use std::fs::File;
|
||||
use std::io;
|
||||
use std::os::unix::io::AsRawFd;
|
||||
use std::path::Path;
|
||||
|
||||
use wo_model::Article;
|
||||
|
||||
use crate::header::SegHeader;
|
||||
use crate::{FLAG_TOMBSTONED, RECORD_HEADER_SIZE};
|
||||
|
||||
/// Reads articles from a .seg file.
|
||||
pub struct SegReader {
|
||||
file: File,
|
||||
header: SegHeader,
|
||||
}
|
||||
|
||||
impl SegReader {
|
||||
/// Open an existing .seg file and validate its header.
|
||||
pub fn open(path: &Path) -> io::Result<Self> {
|
||||
let file = File::open(path)?;
|
||||
let mut header_buf = [0u8; 32];
|
||||
pread_exact(&file, &mut header_buf, 0)?;
|
||||
let header = SegHeader::from_bytes(&header_buf)?;
|
||||
|
||||
Ok(Self { file, header })
|
||||
}
|
||||
|
||||
/// Read a single article record at the given byte offset.
|
||||
///
|
||||
/// Returns `None` if the record is tombstoned.
|
||||
pub fn read_at(&self, offset: u64) -> io::Result<Option<Article>> {
|
||||
// Read the record header: [u32 length][u8 flags]
|
||||
let mut rec_header = [0u8; RECORD_HEADER_SIZE];
|
||||
pread_exact(&self.file, &mut rec_header, offset)?;
|
||||
|
||||
let payload_len = u32::from_le_bytes(rec_header[0..4].try_into().unwrap()) as usize;
|
||||
let flags = rec_header[4];
|
||||
|
||||
if flags == FLAG_TOMBSTONED {
|
||||
return Ok(None);
|
||||
}
|
||||
|
||||
// Read the payload.
|
||||
let mut payload = vec![0u8; payload_len];
|
||||
pread_exact(&self.file, &mut payload, offset + RECORD_HEADER_SIZE as u64)?;
|
||||
|
||||
let article: Article = bincode::deserialize(&payload).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::InvalidData, format!("bincode: {}", e))
|
||||
})?;
|
||||
|
||||
Ok(Some(article))
|
||||
}
|
||||
|
||||
/// Iterate all active (non-tombstoned) records.
|
||||
///
|
||||
/// Returns `(byte_offset, Article)` pairs.
|
||||
pub fn iter(&self) -> io::Result<Vec<(u64, Article)>> {
|
||||
let mut results = Vec::new();
|
||||
let mut pos = self.header.data_start;
|
||||
|
||||
for _ in 0..self.header.record_count {
|
||||
// Read record header.
|
||||
let mut rec_header = [0u8; RECORD_HEADER_SIZE];
|
||||
pread_exact(&self.file, &mut rec_header, pos)?;
|
||||
|
||||
let payload_len = u32::from_le_bytes(rec_header[0..4].try_into().unwrap()) as usize;
|
||||
let flags = rec_header[4];
|
||||
let record_offset = pos;
|
||||
|
||||
pos += RECORD_HEADER_SIZE as u64 + payload_len as u64;
|
||||
|
||||
if flags == FLAG_TOMBSTONED {
|
||||
continue;
|
||||
}
|
||||
|
||||
let mut payload = vec![0u8; payload_len];
|
||||
pread_exact(
|
||||
&self.file,
|
||||
&mut payload,
|
||||
record_offset + RECORD_HEADER_SIZE as u64,
|
||||
)?;
|
||||
|
||||
let article: Article = bincode::deserialize(&payload).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::InvalidData, format!("bincode: {}", e))
|
||||
})?;
|
||||
|
||||
results.push((record_offset, article));
|
||||
}
|
||||
|
||||
Ok(results)
|
||||
}
|
||||
|
||||
/// Return the parsed file header.
|
||||
pub fn header(&self) -> &SegHeader {
|
||||
&self.header
|
||||
}
|
||||
}
|
||||
|
||||
/// Positional read using pread(2). Does not modify the file offset.
|
||||
fn pread_exact(file: &File, buf: &mut [u8], offset: u64) -> io::Result<()> {
|
||||
let fd = file.as_raw_fd();
|
||||
let mut read = 0usize;
|
||||
while read < buf.len() {
|
||||
let ret = unsafe {
|
||||
libc::pread(
|
||||
fd,
|
||||
buf[read..].as_mut_ptr() as *mut libc::c_void,
|
||||
buf.len() - read,
|
||||
(offset + read as u64) as libc::off_t,
|
||||
)
|
||||
};
|
||||
if ret < 0 {
|
||||
return Err(io::Error::last_os_error());
|
||||
}
|
||||
if ret == 0 {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::UnexpectedEof,
|
||||
"pread returned 0",
|
||||
));
|
||||
}
|
||||
read += ret as usize;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::SegWriter;
|
||||
use wo_model::*;
|
||||
|
||||
fn sample_article(sys_title: &str) -> Article {
|
||||
Article {
|
||||
sys_title: sys_title.to_string(),
|
||||
title: format!("Title: {}", sys_title),
|
||||
published: true,
|
||||
author: "Author".into(),
|
||||
tags: vec!["test".into()],
|
||||
published_on: Some(2000),
|
||||
content_html: "<h1>Hello world.</h1>".into(),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn write_and_read_single() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let path = tmp.path().join("test.seg");
|
||||
|
||||
let article = sample_article("test-one");
|
||||
|
||||
let offset = {
|
||||
let mut writer = SegWriter::create(&path).unwrap();
|
||||
writer.append(&article).unwrap()
|
||||
};
|
||||
|
||||
let reader = SegReader::open(&path).unwrap();
|
||||
assert_eq!(reader.header().record_count, 1);
|
||||
|
||||
let loaded = reader.read_at(offset).unwrap().unwrap();
|
||||
assert_eq!(loaded.sys_title, "test-one");
|
||||
assert_eq!(loaded, article);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn write_and_read_multiple() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let path = tmp.path().join("multi.seg");
|
||||
|
||||
let articles: Vec<Article> = (0..5).map(|i| sample_article(&format!("art-{}", i))).collect();
|
||||
|
||||
let offsets: Vec<u64> = {
|
||||
let mut writer = SegWriter::create(&path).unwrap();
|
||||
articles.iter().map(|a| writer.append(a).unwrap()).collect()
|
||||
};
|
||||
|
||||
let reader = SegReader::open(&path).unwrap();
|
||||
assert_eq!(reader.header().record_count, 5);
|
||||
|
||||
// Read each by offset.
|
||||
for (i, offset) in offsets.iter().enumerate() {
|
||||
let loaded = reader.read_at(*offset).unwrap().unwrap();
|
||||
assert_eq!(loaded.sys_title, format!("art-{}", i));
|
||||
}
|
||||
|
||||
// Iterate all.
|
||||
let all = reader.iter().unwrap();
|
||||
assert_eq!(all.len(), 5);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tombstone_record() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let path = tmp.path().join("tomb.seg");
|
||||
|
||||
let offset = {
|
||||
let mut writer = SegWriter::create(&path).unwrap();
|
||||
let o = writer.append(&sample_article("to-delete")).unwrap();
|
||||
writer.append(&sample_article("to-keep")).unwrap();
|
||||
writer.tombstone(o).unwrap();
|
||||
o
|
||||
};
|
||||
|
||||
let reader = SegReader::open(&path).unwrap();
|
||||
|
||||
// Direct read returns None for tombstoned.
|
||||
assert!(reader.read_at(offset).unwrap().is_none());
|
||||
|
||||
// Iter skips tombstoned.
|
||||
let all = reader.iter().unwrap();
|
||||
assert_eq!(all.len(), 1);
|
||||
assert_eq!(all[0].1.sys_title, "to-keep");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn corrupted_header() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let path = tmp.path().join("bad.seg");
|
||||
std::fs::write(&path, b"not a segment file at all!!!!!xx").unwrap();
|
||||
assert!(SegReader::open(&path).is_err());
|
||||
}
|
||||
}
|
||||
129
crates/wo-seg/src/writer.rs
Normal file
129
crates/wo-seg/src/writer.rs
Normal file
|
|
@ -0,0 +1,129 @@
|
|||
use std::fs::{File, OpenOptions};
|
||||
use std::io;
|
||||
use std::os::unix::io::AsRawFd;
|
||||
use std::path::Path;
|
||||
|
||||
use wo_model::Article;
|
||||
|
||||
use crate::header::{SegHeader, HEADER_SIZE};
|
||||
use crate::{FLAG_ACTIVE, RECORD_HEADER_SIZE};
|
||||
|
||||
/// Writes articles to a .seg file.
|
||||
///
|
||||
/// Records are appended sequentially. Each append returns the byte offset
|
||||
/// of the record, which can be stored in an index for direct access.
|
||||
pub struct SegWriter {
|
||||
file: File,
|
||||
header: SegHeader,
|
||||
/// Current write position (end of file).
|
||||
pos: u64,
|
||||
}
|
||||
|
||||
impl SegWriter {
|
||||
/// Create a new .seg file at the given path.
|
||||
///
|
||||
/// Writes the initial header and optionally pre-allocates disk space
|
||||
/// using `fallocate` to reduce fragmentation.
|
||||
pub fn create(path: &Path) -> io::Result<Self> {
|
||||
let file = OpenOptions::new()
|
||||
.read(true)
|
||||
.write(true)
|
||||
.create(true)
|
||||
.truncate(true)
|
||||
.open(path)?;
|
||||
|
||||
let header = SegHeader::new();
|
||||
// Write the header at position 0.
|
||||
pwrite_all(&file, &header.to_bytes(), 0)?;
|
||||
|
||||
// Pre-allocate 1 MB to reduce fragmentation.
|
||||
let _ = fallocate_safe(&file, HEADER_SIZE as i64, 1024 * 1024);
|
||||
|
||||
Ok(Self {
|
||||
file,
|
||||
header,
|
||||
pos: HEADER_SIZE,
|
||||
})
|
||||
}
|
||||
|
||||
/// Append an article as an active record.
|
||||
///
|
||||
/// Returns the byte offset of the record start (the position of the
|
||||
/// length prefix), which can be used for direct reads via `SegReader::read_at`.
|
||||
pub fn append(&mut self, article: &Article) -> io::Result<u64> {
|
||||
let payload = bincode::serialize(article).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::InvalidData, format!("bincode: {}", e))
|
||||
})?;
|
||||
|
||||
let record_offset = self.pos;
|
||||
let payload_len = payload.len() as u32;
|
||||
|
||||
// Build record: [u32 length][u8 flags][payload]
|
||||
let mut record = Vec::with_capacity(RECORD_HEADER_SIZE + payload.len());
|
||||
record.extend_from_slice(&payload_len.to_le_bytes());
|
||||
record.push(FLAG_ACTIVE);
|
||||
record.extend_from_slice(&payload);
|
||||
|
||||
pwrite_all(&self.file, &record, record_offset)?;
|
||||
|
||||
self.pos += record.len() as u64;
|
||||
self.header.record_count += 1;
|
||||
|
||||
// Update the header with the new record count.
|
||||
pwrite_all(&self.file, &self.header.to_bytes(), 0)?;
|
||||
|
||||
Ok(record_offset)
|
||||
}
|
||||
|
||||
/// Tombstone a record at the given offset.
|
||||
///
|
||||
/// Sets the flags byte to `FLAG_TOMBSTONED` without modifying the payload.
|
||||
pub fn tombstone(&self, record_offset: u64) -> io::Result<()> {
|
||||
let flags_offset = record_offset + 4; // skip the u32 length
|
||||
pwrite_all(&self.file, &[crate::FLAG_TOMBSTONED], flags_offset)
|
||||
}
|
||||
|
||||
/// Return the current header (record count, etc.).
|
||||
pub fn header(&self) -> &SegHeader {
|
||||
&self.header
|
||||
}
|
||||
|
||||
/// Sync file data and metadata to disk.
|
||||
pub fn sync(&self) -> io::Result<()> {
|
||||
self.file.sync_all()
|
||||
}
|
||||
}
|
||||
|
||||
/// Positional write using pwrite(2). Does not modify the file offset.
|
||||
fn pwrite_all(file: &File, buf: &[u8], offset: u64) -> io::Result<()> {
|
||||
let fd = file.as_raw_fd();
|
||||
let mut written = 0usize;
|
||||
while written < buf.len() {
|
||||
let ret = unsafe {
|
||||
libc::pwrite(
|
||||
fd,
|
||||
buf[written..].as_ptr() as *const libc::c_void,
|
||||
buf.len() - written,
|
||||
(offset + written as u64) as libc::off_t,
|
||||
)
|
||||
};
|
||||
if ret < 0 {
|
||||
return Err(io::Error::last_os_error());
|
||||
}
|
||||
if ret == 0 {
|
||||
return Err(io::Error::new(io::ErrorKind::WriteZero, "pwrite returned 0"));
|
||||
}
|
||||
written += ret as usize;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Best-effort fallocate. Non-fatal if unsupported.
|
||||
fn fallocate_safe(file: &File, offset: i64, len: i64) -> io::Result<()> {
|
||||
let ret = unsafe { libc::fallocate(file.as_raw_fd(), 0, offset, len) };
|
||||
if ret < 0 {
|
||||
Err(io::Error::last_os_error())
|
||||
} else {
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
11
crates/wo-serve/Cargo.toml
Normal file
11
crates/wo-serve/Cargo.toml
Normal file
|
|
@ -0,0 +1,11 @@
|
|||
[package]
|
||||
name = "wo-serve"
|
||||
version = "0.1.0"
|
||||
edition = "2021"
|
||||
|
||||
[dependencies]
|
||||
wo-http = { path = "../wo-http" }
|
||||
libc = { workspace = true }
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile = "3"
|
||||
7
crates/wo-serve/src/lib.rs
Normal file
7
crates/wo-serve/src/lib.rs
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
mod mime;
|
||||
mod resolve;
|
||||
mod sendfile;
|
||||
|
||||
pub use mime::content_type_for;
|
||||
pub use resolve::resolve_path;
|
||||
pub use sendfile::send_file;
|
||||
44
crates/wo-serve/src/mime.rs
Normal file
44
crates/wo-serve/src/mime.rs
Normal file
|
|
@ -0,0 +1,44 @@
|
|||
/// Map a file extension to its HTTP Content-Type.
|
||||
pub fn content_type_for(path: &str) -> &'static str {
|
||||
let ext = path.rsplit('.').next().unwrap_or("");
|
||||
match ext {
|
||||
"html" | "htmlx" => "text/html; charset=utf-8",
|
||||
"css" => "text/css; charset=utf-8",
|
||||
"js" => "application/javascript; charset=utf-8",
|
||||
"json" => "application/json; charset=utf-8",
|
||||
"png" => "image/png",
|
||||
"jpg" | "jpeg" => "image/jpeg",
|
||||
"gif" => "image/gif",
|
||||
"svg" => "image/svg+xml",
|
||||
"ico" => "image/x-icon",
|
||||
"woff" => "font/woff",
|
||||
"woff2" => "font/woff2",
|
||||
"ttf" => "font/ttf",
|
||||
"txt" => "text/plain; charset=utf-8",
|
||||
"xml" => "application/xml; charset=utf-8",
|
||||
_ => "application/octet-stream",
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn known_types() {
|
||||
assert_eq!(content_type_for("main.css"), "text/css; charset=utf-8");
|
||||
assert_eq!(content_type_for("logo.png"), "image/png");
|
||||
assert_eq!(content_type_for("app.js"), "application/javascript; charset=utf-8");
|
||||
assert_eq!(content_type_for("favicon.ico"), "image/x-icon");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unknown_type() {
|
||||
assert_eq!(content_type_for("file.xyz"), "application/octet-stream");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn nested_path() {
|
||||
assert_eq!(content_type_for("styles/code-theme.css"), "text/css; charset=utf-8");
|
||||
}
|
||||
}
|
||||
80
crates/wo-serve/src/resolve.rs
Normal file
80
crates/wo-serve/src/resolve.rs
Normal file
|
|
@ -0,0 +1,80 @@
|
|||
use std::io;
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
/// Resolve a request path relative to a static directory.
|
||||
///
|
||||
/// Returns the absolute path to the file, or an error if:
|
||||
/// - The path contains `..` (directory traversal)
|
||||
/// - The resolved path escapes the base directory
|
||||
/// - The file doesn't exist
|
||||
pub fn resolve_path(base: &Path, request_path: &str) -> io::Result<PathBuf> {
|
||||
// Reject any path with `..`
|
||||
if request_path.contains("..") {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::PermissionDenied,
|
||||
"path traversal rejected",
|
||||
));
|
||||
}
|
||||
|
||||
let clean = request_path.trim_start_matches('/');
|
||||
let resolved = base.join(clean);
|
||||
|
||||
// Canonicalize and verify it's still under base.
|
||||
let canonical = resolved.canonicalize().map_err(|_| {
|
||||
io::Error::new(io::ErrorKind::NotFound, "file not found")
|
||||
})?;
|
||||
|
||||
let canonical_base = base.canonicalize().map_err(|_| {
|
||||
io::Error::new(io::ErrorKind::NotFound, "base directory not found")
|
||||
})?;
|
||||
|
||||
if !canonical.starts_with(&canonical_base) {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::PermissionDenied,
|
||||
"path traversal rejected",
|
||||
));
|
||||
}
|
||||
|
||||
if !canonical.is_file() {
|
||||
return Err(io::Error::new(io::ErrorKind::NotFound, "not a file"));
|
||||
}
|
||||
|
||||
Ok(canonical)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use std::fs;
|
||||
|
||||
#[test]
|
||||
fn valid_path() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
fs::write(tmp.path().join("test.css"), "body {}").unwrap();
|
||||
|
||||
let path = resolve_path(tmp.path(), "test.css").unwrap();
|
||||
assert!(path.ends_with("test.css"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn nested_path() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
fs::create_dir_all(tmp.path().join("styles")).unwrap();
|
||||
fs::write(tmp.path().join("styles/main.css"), "body {}").unwrap();
|
||||
|
||||
let path = resolve_path(tmp.path(), "styles/main.css").unwrap();
|
||||
assert!(path.ends_with("styles/main.css"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn reject_traversal() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
assert!(resolve_path(tmp.path(), "../etc/passwd").is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn reject_nonexistent() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
assert!(resolve_path(tmp.path(), "nonexistent.css").is_err());
|
||||
}
|
||||
}
|
||||
109
crates/wo-serve/src/sendfile.rs
Normal file
109
crates/wo-serve/src/sendfile.rs
Normal file
|
|
@ -0,0 +1,109 @@
|
|||
use std::fs;
|
||||
use std::io;
|
||||
use std::os::unix::io::RawFd;
|
||||
use std::path::Path;
|
||||
|
||||
use wo_http::response::Response;
|
||||
|
||||
use crate::mime::content_type_for;
|
||||
use crate::resolve::resolve_path;
|
||||
|
||||
/// Serve a static file. Returns a Response with the file contents.
|
||||
///
|
||||
/// For small files, reads into memory and returns a normal Response.
|
||||
/// For zero-copy serving via sendfile, use `send_file_zero_copy` instead.
|
||||
pub fn send_file(static_dir: &Path, request_path: &str) -> Response {
|
||||
let file_path = match resolve_path(static_dir, request_path) {
|
||||
Ok(p) => p,
|
||||
Err(e) => {
|
||||
return match e.kind() {
|
||||
io::ErrorKind::NotFound => Response::not_found(),
|
||||
io::ErrorKind::PermissionDenied => {
|
||||
Response::new(403, "Forbidden")
|
||||
}
|
||||
_ => Response::internal_error(&e.to_string()),
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
let body = match fs::read(&file_path) {
|
||||
Ok(b) => b,
|
||||
Err(_) => return Response::internal_error("failed to read file"),
|
||||
};
|
||||
|
||||
let ct = content_type_for(file_path.to_str().unwrap_or(""));
|
||||
let mut response = Response::ok(body, ct);
|
||||
response.header("Cache-Control", "public, max-age=3600");
|
||||
response
|
||||
}
|
||||
|
||||
/// Zero-copy file serving using sendfile(2).
|
||||
///
|
||||
/// Sends the file directly from the page cache to the socket fd.
|
||||
/// Returns the number of bytes sent, or an error.
|
||||
pub fn sendfile_to_fd(socket_fd: RawFd, file_path: &Path) -> io::Result<usize> {
|
||||
let file = fs::File::open(file_path)?;
|
||||
let file_fd = {
|
||||
use std::os::unix::io::AsRawFd;
|
||||
file.as_raw_fd()
|
||||
};
|
||||
|
||||
let metadata = file.metadata()?;
|
||||
let file_size = metadata.len() as usize;
|
||||
|
||||
let mut offset: libc::off_t = 0;
|
||||
let mut sent = 0usize;
|
||||
|
||||
while sent < file_size {
|
||||
let n = unsafe {
|
||||
libc::sendfile(
|
||||
socket_fd,
|
||||
file_fd,
|
||||
&mut offset,
|
||||
file_size - sent,
|
||||
)
|
||||
};
|
||||
if n < 0 {
|
||||
let err = io::Error::last_os_error();
|
||||
if err.raw_os_error() == Some(libc::EAGAIN) {
|
||||
break; // partial send, caller should retry on EPOLLOUT
|
||||
}
|
||||
return Err(err);
|
||||
}
|
||||
if n == 0 {
|
||||
break;
|
||||
}
|
||||
sent += n as usize;
|
||||
}
|
||||
|
||||
Ok(sent)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn serve_existing_file() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
fs::write(tmp.path().join("test.css"), "body { color: red; }").unwrap();
|
||||
|
||||
let response = send_file(tmp.path(), "test.css");
|
||||
assert_eq!(response.status, 200);
|
||||
assert!(String::from_utf8_lossy(&response.body).contains("color: red"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn serve_nonexistent_file() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let response = send_file(tmp.path(), "nope.css");
|
||||
assert_eq!(response.status, 404);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn serve_traversal_attempt() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let response = send_file(tmp.path(), "../etc/passwd");
|
||||
assert_eq!(response.status, 403);
|
||||
}
|
||||
}
|
||||
13
crates/wo-store/Cargo.toml
Normal file
13
crates/wo-store/Cargo.toml
Normal file
|
|
@ -0,0 +1,13 @@
|
|||
[package]
|
||||
name = "wo-store"
|
||||
version = "0.1.0"
|
||||
edition = "2021"
|
||||
|
||||
[dependencies]
|
||||
wo-model = { path = "../wo-model" }
|
||||
wo-seg = { path = "../wo-seg" }
|
||||
wo-index = { path = "../wo-index" }
|
||||
serde_json = { workspace = true }
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile = "3"
|
||||
355
crates/wo-store/src/lib.rs
Normal file
355
crates/wo-store/src/lib.rs
Normal file
|
|
@ -0,0 +1,355 @@
|
|||
use std::collections::HashMap;
|
||||
use std::io;
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
use wo_index::{DateIndex, TagIndex, TitleIndex};
|
||||
use wo_model::{Article, ContentLoader};
|
||||
use wo_seg::{SegReader, SegWriter};
|
||||
|
||||
/// Unified storage engine composing .seg files and indexes.
|
||||
///
|
||||
/// Provides the query API that the rest of the system uses.
|
||||
/// Handles cold-start (rebuild from content/) and incremental ingestion.
|
||||
pub struct Store {
|
||||
seg_path: PathBuf,
|
||||
index_dir: PathBuf,
|
||||
content_dir: PathBuf,
|
||||
title_idx: TitleIndex,
|
||||
date_idx: DateIndex,
|
||||
tag_idx: TagIndex,
|
||||
/// Monotonic version counter per sys_title (for subscription diffs).
|
||||
versions: HashMap<String, u64>,
|
||||
version_counter: u64,
|
||||
}
|
||||
|
||||
impl Store {
|
||||
/// Open an existing data directory, or cold-start from content/.
|
||||
///
|
||||
/// If `data_dir` contains valid .seg and index files, opens them.
|
||||
/// Otherwise, rebuilds everything from `content_dir`.
|
||||
pub fn open(content_dir: &Path, data_dir: &Path) -> io::Result<Self> {
|
||||
let seg_path = data_dir.join("articles.seg");
|
||||
let index_dir = data_dir.join("index");
|
||||
|
||||
if seg_path.exists() && index_dir.join("title.idx").exists() {
|
||||
Self::open_existing(content_dir, &seg_path, &index_dir)
|
||||
} else {
|
||||
let mut store = Self::init_empty(content_dir, &seg_path, &index_dir)?;
|
||||
store.rebuild()?;
|
||||
Ok(store)
|
||||
}
|
||||
}
|
||||
|
||||
fn open_existing(
|
||||
content_dir: &Path,
|
||||
seg_path: &Path,
|
||||
index_dir: &Path,
|
||||
) -> io::Result<Self> {
|
||||
let title_idx = TitleIndex::open(&index_dir.join("title.idx"))?;
|
||||
let date_idx = DateIndex::open(&index_dir.join("date.idx"))?;
|
||||
let tag_idx = TagIndex::open(&index_dir.join("tags.idx"))?;
|
||||
|
||||
// Build version map from existing seg records.
|
||||
let reader = SegReader::open(seg_path)?;
|
||||
let mut versions = HashMap::new();
|
||||
let mut version_counter = 0u64;
|
||||
for (_, article) in reader.iter()? {
|
||||
version_counter += 1;
|
||||
versions.insert(article.sys_title.clone(), version_counter);
|
||||
}
|
||||
|
||||
Ok(Self {
|
||||
seg_path: seg_path.to_path_buf(),
|
||||
index_dir: index_dir.to_path_buf(),
|
||||
content_dir: content_dir.to_path_buf(),
|
||||
title_idx,
|
||||
date_idx,
|
||||
tag_idx,
|
||||
versions,
|
||||
version_counter,
|
||||
})
|
||||
}
|
||||
|
||||
fn init_empty(
|
||||
content_dir: &Path,
|
||||
seg_path: &Path,
|
||||
index_dir: &Path,
|
||||
) -> io::Result<Self> {
|
||||
std::fs::create_dir_all(seg_path.parent().unwrap())?;
|
||||
std::fs::create_dir_all(index_dir)?;
|
||||
|
||||
// Create empty seg file.
|
||||
let _writer = SegWriter::create(seg_path)?;
|
||||
|
||||
// Create empty indexes.
|
||||
TitleIndex::build(&index_dir.join("title.idx"), &[])?;
|
||||
DateIndex::build(&index_dir.join("date.idx"), &mut [])?;
|
||||
TagIndex::build(&index_dir.join("tags.idx"), &[])?;
|
||||
|
||||
let title_idx = TitleIndex::open(&index_dir.join("title.idx"))?;
|
||||
let date_idx = DateIndex::open(&index_dir.join("date.idx"))?;
|
||||
let tag_idx = TagIndex::open(&index_dir.join("tags.idx"))?;
|
||||
|
||||
Ok(Self {
|
||||
seg_path: seg_path.to_path_buf(),
|
||||
index_dir: index_dir.to_path_buf(),
|
||||
content_dir: content_dir.to_path_buf(),
|
||||
title_idx,
|
||||
date_idx,
|
||||
tag_idx,
|
||||
versions: HashMap::new(),
|
||||
version_counter: 0,
|
||||
})
|
||||
}
|
||||
|
||||
/// Full rebuild: load all articles from content/, rewrite .seg and indexes.
|
||||
pub fn rebuild(&mut self) -> io::Result<()> {
|
||||
let articles = ContentLoader::load_all(&self.content_dir)?;
|
||||
|
||||
// Write all articles to a new .seg file.
|
||||
let mut writer = SegWriter::create(&self.seg_path)?;
|
||||
let mut index_entries: Vec<(String, i64, Vec<String>, u64)> = Vec::new();
|
||||
|
||||
self.versions.clear();
|
||||
self.version_counter = 0;
|
||||
|
||||
for article in &articles {
|
||||
let offset = writer.append(article)?;
|
||||
let timestamp = article.published_on.unwrap_or(0);
|
||||
|
||||
index_entries.push((
|
||||
article.sys_title.clone(),
|
||||
timestamp,
|
||||
article.tags.clone(),
|
||||
offset,
|
||||
));
|
||||
|
||||
self.version_counter += 1;
|
||||
self.versions
|
||||
.insert(article.sys_title.clone(), self.version_counter);
|
||||
}
|
||||
|
||||
writer.sync()?;
|
||||
|
||||
// Rebuild all indexes.
|
||||
wo_index::rebuild_all(&self.index_dir, &index_entries)?;
|
||||
|
||||
// Reload indexes.
|
||||
self.title_idx = TitleIndex::open(&self.index_dir.join("title.idx"))?;
|
||||
self.date_idx = DateIndex::open(&self.index_dir.join("date.idx"))?;
|
||||
self.tag_idx = TagIndex::open(&self.index_dir.join("tags.idx"))?;
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Look up a single article by sys_title.
|
||||
pub fn get_by_title(&self, sys_title: &str) -> io::Result<Option<Article>> {
|
||||
match self.title_idx.get(sys_title) {
|
||||
Some(offset) => {
|
||||
let reader = SegReader::open(&self.seg_path)?;
|
||||
reader.read_at(offset)
|
||||
}
|
||||
None => Ok(None),
|
||||
}
|
||||
}
|
||||
|
||||
/// List published articles, most recent first, with pagination.
|
||||
pub fn list_published(&self, skip: usize, limit: usize) -> io::Result<Vec<Article>> {
|
||||
let offsets = self.date_idx.latest(skip + limit);
|
||||
let reader = SegReader::open(&self.seg_path)?;
|
||||
|
||||
let mut articles = Vec::new();
|
||||
for &offset in offsets.iter().skip(skip).take(limit) {
|
||||
if let Some(article) = reader.read_at(offset)? {
|
||||
if article.published {
|
||||
articles.push(article);
|
||||
}
|
||||
}
|
||||
}
|
||||
Ok(articles)
|
||||
}
|
||||
|
||||
/// List all articles with a given tag.
|
||||
pub fn list_by_tag(&self, tag: &str) -> io::Result<Vec<Article>> {
|
||||
let offsets = match self.tag_idx.get(tag) {
|
||||
Some(offsets) => offsets,
|
||||
None => return Ok(vec![]),
|
||||
};
|
||||
|
||||
let reader = SegReader::open(&self.seg_path)?;
|
||||
let mut articles = Vec::new();
|
||||
for &offset in offsets {
|
||||
if let Some(article) = reader.read_at(offset)? {
|
||||
articles.push(article);
|
||||
}
|
||||
}
|
||||
Ok(articles)
|
||||
}
|
||||
|
||||
/// List articles published within a timestamp range.
|
||||
pub fn list_by_date_range(&self, start: i64, end: i64) -> io::Result<Vec<Article>> {
|
||||
let offsets = self.date_idx.range(start, end);
|
||||
let reader = SegReader::open(&self.seg_path)?;
|
||||
|
||||
let mut articles = Vec::new();
|
||||
for offset in offsets {
|
||||
if let Some(article) = reader.read_at(offset)? {
|
||||
articles.push(article);
|
||||
}
|
||||
}
|
||||
Ok(articles)
|
||||
}
|
||||
|
||||
/// Count of published articles.
|
||||
pub fn count_published(&self) -> io::Result<usize> {
|
||||
// For now, iterate and count. With small article counts this is fine.
|
||||
let reader = SegReader::open(&self.seg_path)?;
|
||||
let count = reader
|
||||
.iter()?
|
||||
.into_iter()
|
||||
.filter(|(_, a)| a.published)
|
||||
.count();
|
||||
Ok(count)
|
||||
}
|
||||
|
||||
/// Ingest a single article from a JSON file path.
|
||||
///
|
||||
/// Appends to .seg and triggers a full index rebuild.
|
||||
/// Returns the sys_title of the ingested article.
|
||||
pub fn ingest_article(&mut self, json_path: &Path) -> io::Result<String> {
|
||||
let article = ContentLoader::load_from_file(json_path)?;
|
||||
let sys_title = article.sys_title.clone();
|
||||
|
||||
// For simplicity, rebuild the entire store.
|
||||
// A future optimization can do incremental append + index update.
|
||||
self.rebuild()?;
|
||||
|
||||
Ok(sys_title)
|
||||
}
|
||||
|
||||
/// Get the current version number for a sys_title.
|
||||
pub fn article_version(&self, sys_title: &str) -> Option<u64> {
|
||||
self.versions.get(sys_title).copied()
|
||||
}
|
||||
|
||||
/// Get a reference to the content directory path.
|
||||
pub fn content_dir(&self) -> &Path {
|
||||
&self.content_dir
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use std::fs;
|
||||
|
||||
fn create_test_article(base: &Path, sys_title: &str, published: bool, tags: &[&str], ts: i64) {
|
||||
let dir = base.join(sys_title);
|
||||
fs::create_dir_all(&dir).unwrap();
|
||||
|
||||
let tags_json: Vec<String> = tags.iter().map(|t| format!("\"{}\"", t)).collect();
|
||||
let json = format!(
|
||||
r#"{{
|
||||
"title": "{}",
|
||||
"sys_title": "{}",
|
||||
"published": {},
|
||||
"content": {{
|
||||
"author": "Author",
|
||||
"content": {{
|
||||
"sections": [{{ "heading": "Intro", "paragraphs": ["Hello."] }}],
|
||||
"codes": [],
|
||||
"images": []
|
||||
}},
|
||||
"tags": [{}],
|
||||
"publishedOn": {}
|
||||
}},
|
||||
"published_on": {}
|
||||
}}"#,
|
||||
sys_title,
|
||||
sys_title,
|
||||
published,
|
||||
tags_json.join(", "),
|
||||
ts,
|
||||
ts
|
||||
);
|
||||
fs::write(dir.join(format!("{}.json", sys_title)), json).unwrap();
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn cold_start_and_queries() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let content = tmp.path().join("content");
|
||||
let data = tmp.path().join("data");
|
||||
|
||||
create_test_article(&content, "art-one", true, &["rust"], 1000);
|
||||
create_test_article(&content, "art-two", true, &["rust", "linux"], 2000);
|
||||
create_test_article(&content, "art-three", false, &["draft"], 3000);
|
||||
create_test_article(&content, "art-four", true, &["linux"], 4000);
|
||||
|
||||
let store = Store::open(&content, &data).unwrap();
|
||||
|
||||
// Title lookup.
|
||||
let art = store.get_by_title("art-one").unwrap().unwrap();
|
||||
assert_eq!(art.sys_title, "art-one");
|
||||
|
||||
assert!(store.get_by_title("nonexistent").unwrap().is_none());
|
||||
|
||||
// Tag query.
|
||||
let rust_articles = store.list_by_tag("rust").unwrap();
|
||||
assert_eq!(rust_articles.len(), 2);
|
||||
|
||||
let linux_articles = store.list_by_tag("linux").unwrap();
|
||||
assert_eq!(linux_articles.len(), 2);
|
||||
|
||||
// Date range.
|
||||
let range = store.list_by_date_range(1500, 3500).unwrap();
|
||||
assert_eq!(range.len(), 2); // art-two(2000) and art-three(3000)
|
||||
|
||||
// Published count.
|
||||
let count = store.count_published().unwrap();
|
||||
assert_eq!(count, 3);
|
||||
|
||||
// Version tracking.
|
||||
assert!(store.article_version("art-one").is_some());
|
||||
assert!(store.article_version("nonexistent").is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rebuild_after_delete() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let content = tmp.path().join("content");
|
||||
let data = tmp.path().join("data");
|
||||
|
||||
create_test_article(&content, "art-one", true, &["test"], 1000);
|
||||
|
||||
// First open: cold start.
|
||||
let store = Store::open(&content, &data).unwrap();
|
||||
assert_eq!(store.count_published().unwrap(), 1);
|
||||
drop(store);
|
||||
|
||||
// Delete data/ and reopen: should rebuild.
|
||||
fs::remove_dir_all(&data).unwrap();
|
||||
let store = Store::open(&content, &data).unwrap();
|
||||
assert_eq!(store.count_published().unwrap(), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn reopen_existing() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let content = tmp.path().join("content");
|
||||
let data = tmp.path().join("data");
|
||||
|
||||
create_test_article(&content, "art-one", true, &["test"], 1000);
|
||||
|
||||
// First open: cold start builds .seg + indexes.
|
||||
let store = Store::open(&content, &data).unwrap();
|
||||
assert_eq!(store.count_published().unwrap(), 1);
|
||||
drop(store);
|
||||
|
||||
// Second open: loads existing files, no rebuild.
|
||||
let store = Store::open(&content, &data).unwrap();
|
||||
assert_eq!(store.count_published().unwrap(), 1);
|
||||
let art = store.get_by_title("art-one").unwrap().unwrap();
|
||||
assert_eq!(art.sys_title, "art-one");
|
||||
}
|
||||
}
|
||||
16
crates/wo-sub/Cargo.toml
Normal file
16
crates/wo-sub/Cargo.toml
Normal file
|
|
@ -0,0 +1,16 @@
|
|||
[package]
|
||||
name = "wo-sub"
|
||||
version = "0.1.0"
|
||||
edition = "2021"
|
||||
|
||||
[dependencies]
|
||||
wo-model = { path = "../wo-model" }
|
||||
wo-store = { path = "../wo-store" }
|
||||
wo-event = { path = "../wo-event" }
|
||||
libc = { workspace = true }
|
||||
bincode = { workspace = true }
|
||||
serde = { workspace = true }
|
||||
serde_json = { workspace = true }
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile = "3"
|
||||
470
crates/wo-sub/src/lib.rs
Normal file
470
crates/wo-sub/src/lib.rs
Normal file
|
|
@ -0,0 +1,470 @@
|
|||
use std::collections::HashMap;
|
||||
use std::io;
|
||||
use std::os::unix::io::RawFd;
|
||||
|
||||
use serde::{Deserialize, Serialize};
|
||||
use wo_event::EventFd;
|
||||
|
||||
/// A subscription query pattern.
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub enum Subscription {
|
||||
/// Subscribe to changes for a single article by sys_title.
|
||||
ByTitle(String),
|
||||
/// Subscribe to changes for all articles with a given tag.
|
||||
ByTag(String),
|
||||
/// Subscribe to all content changes.
|
||||
All,
|
||||
}
|
||||
|
||||
/// The type of content change.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
|
||||
pub enum ChangeType {
|
||||
Created,
|
||||
Updated,
|
||||
Deleted,
|
||||
}
|
||||
|
||||
/// Notification payload written to subscriber fds.
|
||||
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
|
||||
pub struct Notification {
|
||||
pub change_type: ChangeType,
|
||||
pub sys_title: String,
|
||||
pub version: u64,
|
||||
}
|
||||
|
||||
/// Wire format for notification payloads.
|
||||
#[derive(Debug, Clone, Copy, PartialEq)]
|
||||
pub enum FdFormat {
|
||||
/// Length-prefixed bincode (for internal subscribers).
|
||||
Binary,
|
||||
/// Length-prefixed JSON (for browser socket fds).
|
||||
Json,
|
||||
}
|
||||
|
||||
/// Manages subscriptions and delivers notifications to subscriber fds.
|
||||
///
|
||||
/// Subscribers register interest via `subscribe()` with a query pattern.
|
||||
/// When content changes, `notify()` evaluates which subscriptions match
|
||||
/// and writes a length-prefixed notification to each matching fd.
|
||||
pub struct SubscriptionManager {
|
||||
/// sys_title → list of subscriber fds.
|
||||
by_title: HashMap<String, Vec<RawFd>>,
|
||||
/// tag → list of subscriber fds.
|
||||
by_tag: HashMap<String, Vec<RawFd>>,
|
||||
/// Fds subscribed to all changes.
|
||||
global: Vec<RawFd>,
|
||||
/// fd → list of subscriptions (for cleanup on disconnect).
|
||||
fd_registry: HashMap<RawFd, Vec<Subscription>>,
|
||||
/// fd → wire format.
|
||||
fd_format: HashMap<RawFd, FdFormat>,
|
||||
/// EventFd for signaling the event loop that notifications are pending.
|
||||
notify_efd: EventFd,
|
||||
}
|
||||
|
||||
impl SubscriptionManager {
|
||||
/// Create a new subscription manager.
|
||||
pub fn new() -> io::Result<Self> {
|
||||
Ok(Self {
|
||||
by_title: HashMap::new(),
|
||||
by_tag: HashMap::new(),
|
||||
global: Vec::new(),
|
||||
fd_registry: HashMap::new(),
|
||||
fd_format: HashMap::new(),
|
||||
notify_efd: EventFd::new()?,
|
||||
})
|
||||
}
|
||||
|
||||
/// The eventfd for signaling the event loop.
|
||||
pub fn notify_fd(&self) -> RawFd {
|
||||
self.notify_efd.fd()
|
||||
}
|
||||
|
||||
/// Register a subscription for the given fd (binary format).
|
||||
pub fn subscribe(&mut self, fd: RawFd, query: Subscription) {
|
||||
self.subscribe_with_format(fd, query, FdFormat::Binary);
|
||||
}
|
||||
|
||||
/// Register a subscription for the given fd (JSON format for browsers).
|
||||
pub fn subscribe_json(&mut self, fd: RawFd, query: Subscription) {
|
||||
self.subscribe_with_format(fd, query, FdFormat::Json);
|
||||
}
|
||||
|
||||
fn subscribe_with_format(&mut self, fd: RawFd, query: Subscription, format: FdFormat) {
|
||||
match &query {
|
||||
Subscription::ByTitle(title) => {
|
||||
self.by_title.entry(title.clone()).or_default().push(fd);
|
||||
}
|
||||
Subscription::ByTag(tag) => {
|
||||
self.by_tag.entry(tag.clone()).or_default().push(fd);
|
||||
}
|
||||
Subscription::All => {
|
||||
self.global.push(fd);
|
||||
}
|
||||
}
|
||||
self.fd_registry.entry(fd).or_default().push(query);
|
||||
self.fd_format.insert(fd, format);
|
||||
}
|
||||
|
||||
/// Remove all subscriptions for a fd (e.g., on EPOLLHUP).
|
||||
pub fn unsubscribe(&mut self, fd: RawFd) {
|
||||
self.fd_format.remove(&fd);
|
||||
if let Some(subs) = self.fd_registry.remove(&fd) {
|
||||
for sub in subs {
|
||||
match sub {
|
||||
Subscription::ByTitle(title) => {
|
||||
if let Some(fds) = self.by_title.get_mut(&title) {
|
||||
fds.retain(|&f| f != fd);
|
||||
}
|
||||
}
|
||||
Subscription::ByTag(tag) => {
|
||||
if let Some(fds) = self.by_tag.get_mut(&tag) {
|
||||
fds.retain(|&f| f != fd);
|
||||
}
|
||||
}
|
||||
Subscription::All => {
|
||||
self.global.retain(|&f| f != fd);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Notify subscribers about a content change.
|
||||
///
|
||||
/// Evaluates which subscriptions match the changed sys_title and tags,
|
||||
/// then writes a length-prefixed notification to each matching fd.
|
||||
pub fn notify(
|
||||
&self,
|
||||
sys_title: &str,
|
||||
tags: &[String],
|
||||
change_type: ChangeType,
|
||||
version: u64,
|
||||
) -> io::Result<usize> {
|
||||
let notification = Notification {
|
||||
change_type,
|
||||
sys_title: sys_title.to_string(),
|
||||
version,
|
||||
};
|
||||
|
||||
// Pre-serialize both formats (only actually used if there are subscribers).
|
||||
let bincode_payload = bincode::serialize(¬ification).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::InvalidData, format!("bincode: {}", e))
|
||||
})?;
|
||||
let json_payload = serde_json::to_vec(¬ification).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::InvalidData, format!("json: {}", e))
|
||||
})?;
|
||||
|
||||
// Build wire formats: [u32 length][payload]
|
||||
let bincode_wire = build_wire(&bincode_payload);
|
||||
let json_wire = build_wire(&json_payload);
|
||||
|
||||
// Collect all fds that should receive this notification.
|
||||
let mut target_fds: Vec<RawFd> = Vec::new();
|
||||
|
||||
// Title subscribers.
|
||||
if let Some(fds) = self.by_title.get(sys_title) {
|
||||
target_fds.extend(fds);
|
||||
}
|
||||
|
||||
// Tag subscribers.
|
||||
for tag in tags {
|
||||
if let Some(fds) = self.by_tag.get(tag) {
|
||||
target_fds.extend(fds);
|
||||
}
|
||||
}
|
||||
|
||||
// Global subscribers.
|
||||
target_fds.extend(&self.global);
|
||||
|
||||
// Deduplicate.
|
||||
target_fds.sort_unstable();
|
||||
target_fds.dedup();
|
||||
|
||||
let mut sent = 0;
|
||||
for &fd in &target_fds {
|
||||
let format = self.fd_format.get(&fd).copied().unwrap_or(FdFormat::Binary);
|
||||
let wire = match format {
|
||||
FdFormat::Binary => &bincode_wire,
|
||||
FdFormat::Json => &json_wire,
|
||||
};
|
||||
if write_all_fd(fd, wire).is_ok() {
|
||||
sent += 1;
|
||||
}
|
||||
}
|
||||
|
||||
// Signal the event loop.
|
||||
if sent > 0 {
|
||||
let _ = self.notify_efd.write(1);
|
||||
}
|
||||
|
||||
Ok(sent)
|
||||
}
|
||||
|
||||
/// Number of active subscriber fds.
|
||||
pub fn subscriber_count(&self) -> usize {
|
||||
self.fd_registry.len()
|
||||
}
|
||||
}
|
||||
|
||||
fn build_wire(payload: &[u8]) -> Vec<u8> {
|
||||
let len = payload.len() as u32;
|
||||
let mut wire = Vec::with_capacity(4 + payload.len());
|
||||
wire.extend_from_slice(&len.to_le_bytes());
|
||||
wire.extend_from_slice(payload);
|
||||
wire
|
||||
}
|
||||
|
||||
/// Write all bytes to a raw fd.
|
||||
fn write_all_fd(fd: RawFd, buf: &[u8]) -> io::Result<()> {
|
||||
let mut written = 0;
|
||||
while written < buf.len() {
|
||||
let ret = unsafe {
|
||||
libc::write(
|
||||
fd,
|
||||
buf[written..].as_ptr() as *const libc::c_void,
|
||||
buf.len() - written,
|
||||
)
|
||||
};
|
||||
if ret < 0 {
|
||||
return Err(io::Error::last_os_error());
|
||||
}
|
||||
if ret == 0 {
|
||||
return Err(io::Error::new(io::ErrorKind::WriteZero, "write returned 0"));
|
||||
}
|
||||
written += ret as usize;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Convenience macro for compile-time subscription wiring.
|
||||
///
|
||||
/// ```ignore
|
||||
/// register!(sub_manager, fd, ByTitle("linux-misc"));
|
||||
/// register!(sub_manager, fd, ByTag("rust"));
|
||||
/// register!(sub_manager, fd, All);
|
||||
/// ```
|
||||
#[macro_export]
|
||||
macro_rules! register {
|
||||
($mgr:expr, $fd:expr, ByTitle($title:expr)) => {
|
||||
$mgr.subscribe($fd, $crate::Subscription::ByTitle($title.into()))
|
||||
};
|
||||
($mgr:expr, $fd:expr, ByTag($tag:expr)) => {
|
||||
$mgr.subscribe($fd, $crate::Subscription::ByTag($tag.into()))
|
||||
};
|
||||
($mgr:expr, $fd:expr, All) => {
|
||||
$mgr.subscribe($fd, $crate::Subscription::All)
|
||||
};
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// Create a pipe and return (read_fd, write_fd).
|
||||
fn make_pipe() -> (RawFd, RawFd) {
|
||||
let mut fds = [0i32; 2];
|
||||
assert_eq!(unsafe { libc::pipe2(fds.as_mut_ptr(), libc::O_NONBLOCK) }, 0);
|
||||
(fds[0], fds[1])
|
||||
}
|
||||
|
||||
fn read_notification(fd: RawFd) -> Notification {
|
||||
let mut len_buf = [0u8; 4];
|
||||
let ret = unsafe { libc::read(fd, len_buf.as_mut_ptr() as *mut libc::c_void, 4) };
|
||||
assert!(ret == 4, "failed to read length prefix");
|
||||
let len = u32::from_le_bytes(len_buf) as usize;
|
||||
|
||||
let mut payload = vec![0u8; len];
|
||||
let ret = unsafe { libc::read(fd, payload.as_mut_ptr() as *mut libc::c_void, len) };
|
||||
assert_eq!(ret as usize, len);
|
||||
|
||||
bincode::deserialize(&payload).unwrap()
|
||||
}
|
||||
|
||||
fn close_fd(fd: RawFd) {
|
||||
unsafe { libc::close(fd) };
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn title_subscription() {
|
||||
let mut mgr = SubscriptionManager::new().unwrap();
|
||||
|
||||
let (r1, w1) = make_pipe();
|
||||
let (r2, w2) = make_pipe();
|
||||
|
||||
mgr.subscribe(w1, Subscription::ByTitle("art-one".into()));
|
||||
mgr.subscribe(w2, Subscription::ByTitle("art-one".into()));
|
||||
|
||||
let sent = mgr
|
||||
.notify("art-one", &[], ChangeType::Updated, 1)
|
||||
.unwrap();
|
||||
assert_eq!(sent, 2);
|
||||
|
||||
let n1 = read_notification(r1);
|
||||
assert_eq!(n1.sys_title, "art-one");
|
||||
assert_eq!(n1.change_type, ChangeType::Updated);
|
||||
assert_eq!(n1.version, 1);
|
||||
|
||||
let n2 = read_notification(r2);
|
||||
assert_eq!(n2.sys_title, "art-one");
|
||||
|
||||
// Unrelated title: no notifications.
|
||||
let sent = mgr
|
||||
.notify("art-two", &[], ChangeType::Created, 2)
|
||||
.unwrap();
|
||||
assert_eq!(sent, 0);
|
||||
|
||||
close_fd(r1); close_fd(w1);
|
||||
close_fd(r2); close_fd(w2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tag_subscription() {
|
||||
let mut mgr = SubscriptionManager::new().unwrap();
|
||||
|
||||
let (r, w) = make_pipe();
|
||||
mgr.subscribe(w, Subscription::ByTag("rust".into()));
|
||||
|
||||
let sent = mgr
|
||||
.notify("art-one", &["rust".into(), "linux".into()], ChangeType::Updated, 1)
|
||||
.unwrap();
|
||||
assert_eq!(sent, 1);
|
||||
|
||||
let n = read_notification(r);
|
||||
assert_eq!(n.sys_title, "art-one");
|
||||
|
||||
// No "rust" tag: no notification.
|
||||
let sent = mgr
|
||||
.notify("art-two", &["python".into()], ChangeType::Created, 2)
|
||||
.unwrap();
|
||||
assert_eq!(sent, 0);
|
||||
|
||||
close_fd(r); close_fd(w);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn global_subscription() {
|
||||
let mut mgr = SubscriptionManager::new().unwrap();
|
||||
|
||||
let (r, w) = make_pipe();
|
||||
mgr.subscribe(w, Subscription::All);
|
||||
|
||||
let sent = mgr.notify("anything", &[], ChangeType::Deleted, 5).unwrap();
|
||||
assert_eq!(sent, 1);
|
||||
|
||||
let n = read_notification(r);
|
||||
assert_eq!(n.change_type, ChangeType::Deleted);
|
||||
|
||||
close_fd(r); close_fd(w);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unsubscribe_removes_fd() {
|
||||
let mut mgr = SubscriptionManager::new().unwrap();
|
||||
|
||||
let (r, w) = make_pipe();
|
||||
mgr.subscribe(w, Subscription::ByTitle("art-one".into()));
|
||||
mgr.subscribe(w, Subscription::All);
|
||||
assert_eq!(mgr.subscriber_count(), 1);
|
||||
|
||||
mgr.unsubscribe(w);
|
||||
assert_eq!(mgr.subscriber_count(), 0);
|
||||
|
||||
let sent = mgr.notify("art-one", &[], ChangeType::Updated, 1).unwrap();
|
||||
assert_eq!(sent, 0);
|
||||
|
||||
close_fd(r); close_fd(w);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dedup_notifications() {
|
||||
let mut mgr = SubscriptionManager::new().unwrap();
|
||||
|
||||
let (r, w) = make_pipe();
|
||||
// Same fd subscribed to both title AND tag — should only get one notification.
|
||||
mgr.subscribe(w, Subscription::ByTitle("art-one".into()));
|
||||
mgr.subscribe(w, Subscription::ByTag("rust".into()));
|
||||
|
||||
let sent = mgr
|
||||
.notify("art-one", &["rust".into()], ChangeType::Updated, 1)
|
||||
.unwrap();
|
||||
assert_eq!(sent, 1); // deduped
|
||||
|
||||
let n = read_notification(r);
|
||||
assert_eq!(n.sys_title, "art-one");
|
||||
|
||||
close_fd(r); close_fd(w);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn json_format_subscription() {
|
||||
let mut mgr = SubscriptionManager::new().unwrap();
|
||||
|
||||
let (r, w) = make_pipe();
|
||||
mgr.subscribe_json(w, Subscription::ByTitle("art-one".into()));
|
||||
|
||||
mgr.notify("art-one", &[], ChangeType::Updated, 1).unwrap();
|
||||
|
||||
// Read length-prefixed JSON.
|
||||
let mut len_buf = [0u8; 4];
|
||||
let ret = unsafe { libc::read(r, len_buf.as_mut_ptr() as *mut libc::c_void, 4) };
|
||||
assert_eq!(ret, 4);
|
||||
let len = u32::from_le_bytes(len_buf) as usize;
|
||||
|
||||
let mut payload = vec![0u8; len];
|
||||
let ret = unsafe { libc::read(r, payload.as_mut_ptr() as *mut libc::c_void, len) };
|
||||
assert_eq!(ret as usize, len);
|
||||
|
||||
// Should be valid JSON.
|
||||
let json: serde_json::Value = serde_json::from_slice(&payload).unwrap();
|
||||
assert_eq!(json["sys_title"], "art-one");
|
||||
assert_eq!(json["change_type"], "Updated");
|
||||
assert_eq!(json["version"], 1);
|
||||
|
||||
close_fd(r); close_fd(w);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn mixed_format_subscribers() {
|
||||
let mut mgr = SubscriptionManager::new().unwrap();
|
||||
|
||||
let (r_bin, w_bin) = make_pipe();
|
||||
let (r_json, w_json) = make_pipe();
|
||||
|
||||
mgr.subscribe(w_bin, Subscription::ByTitle("art-one".into()));
|
||||
mgr.subscribe_json(w_json, Subscription::ByTitle("art-one".into()));
|
||||
|
||||
let sent = mgr.notify("art-one", &[], ChangeType::Created, 5).unwrap();
|
||||
assert_eq!(sent, 2);
|
||||
|
||||
// Binary subscriber gets bincode.
|
||||
let n = read_notification(r_bin);
|
||||
assert_eq!(n.sys_title, "art-one");
|
||||
assert_eq!(n.version, 5);
|
||||
|
||||
// JSON subscriber gets JSON.
|
||||
let mut len_buf = [0u8; 4];
|
||||
unsafe { libc::read(r_json, len_buf.as_mut_ptr() as *mut libc::c_void, 4) };
|
||||
let len = u32::from_le_bytes(len_buf) as usize;
|
||||
let mut payload = vec![0u8; len];
|
||||
unsafe { libc::read(r_json, payload.as_mut_ptr() as *mut libc::c_void, len) };
|
||||
let json: serde_json::Value = serde_json::from_slice(&payload).unwrap();
|
||||
assert_eq!(json["sys_title"], "art-one");
|
||||
|
||||
close_fd(r_bin); close_fd(w_bin);
|
||||
close_fd(r_json); close_fd(w_json);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn register_macro() {
|
||||
let mut mgr = SubscriptionManager::new().unwrap();
|
||||
let (r, w) = make_pipe();
|
||||
|
||||
register!(mgr, w, ByTitle("test-article"));
|
||||
register!(mgr, w, ByTag("rust"));
|
||||
register!(mgr, w, All);
|
||||
|
||||
assert_eq!(mgr.subscriber_count(), 1);
|
||||
|
||||
close_fd(r); close_fd(w);
|
||||
}
|
||||
}
|
||||
11
crates/wo-watch/Cargo.toml
Normal file
11
crates/wo-watch/Cargo.toml
Normal file
|
|
@ -0,0 +1,11 @@
|
|||
[package]
|
||||
name = "wo-watch"
|
||||
version = "0.1.0"
|
||||
edition = "2021"
|
||||
|
||||
[dependencies]
|
||||
wo-store = { path = "../wo-store" }
|
||||
libc = { workspace = true }
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile = "3"
|
||||
280
crates/wo-watch/src/lib.rs
Normal file
280
crates/wo-watch/src/lib.rs
Normal file
|
|
@ -0,0 +1,280 @@
|
|||
use std::collections::HashMap;
|
||||
use std::io;
|
||||
use std::os::unix::io::RawFd;
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
/// A content change detected by the watcher.
|
||||
#[derive(Debug, Clone, PartialEq)]
|
||||
pub enum ContentChange {
|
||||
Created(String), // sys_title
|
||||
Modified(String), // sys_title
|
||||
Deleted(String), // sys_title
|
||||
}
|
||||
|
||||
/// Watches a content directory for article file changes using inotify.
|
||||
///
|
||||
/// Expects the directory structure:
|
||||
/// ```text
|
||||
/// content_dir/
|
||||
/// {sys_title}/
|
||||
/// {sys_title}.json
|
||||
/// ```
|
||||
pub struct ContentWatcher {
|
||||
inotify_fd: RawFd,
|
||||
/// Maps inotify watch descriptor → directory path.
|
||||
wd_to_path: HashMap<i32, PathBuf>,
|
||||
/// Maps directory path → watch descriptor.
|
||||
path_to_wd: HashMap<PathBuf, i32>,
|
||||
content_dir: PathBuf,
|
||||
}
|
||||
|
||||
const EVENT_MASK: u32 =
|
||||
(libc::IN_CREATE | libc::IN_MODIFY | libc::IN_DELETE | libc::IN_MOVED_TO | libc::IN_MOVED_FROM)
|
||||
as u32;
|
||||
|
||||
impl ContentWatcher {
|
||||
/// Create a new watcher on the given content directory.
|
||||
///
|
||||
/// Adds inotify watches on the content dir itself and all subdirectories.
|
||||
pub fn new(content_dir: &Path) -> io::Result<Self> {
|
||||
let fd = unsafe { libc::inotify_init1(libc::IN_NONBLOCK | libc::IN_CLOEXEC) };
|
||||
if fd < 0 {
|
||||
return Err(io::Error::last_os_error());
|
||||
}
|
||||
|
||||
let mut watcher = Self {
|
||||
inotify_fd: fd,
|
||||
wd_to_path: HashMap::new(),
|
||||
path_to_wd: HashMap::new(),
|
||||
content_dir: content_dir.to_path_buf(),
|
||||
};
|
||||
|
||||
// Watch the root content directory (for new article dirs).
|
||||
watcher.add_watch(content_dir)?;
|
||||
|
||||
// Watch each existing subdirectory.
|
||||
if content_dir.exists() {
|
||||
for entry in std::fs::read_dir(content_dir)? {
|
||||
let entry = entry?;
|
||||
if entry.path().is_dir() {
|
||||
watcher.add_watch(&entry.path())?;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Ok(watcher)
|
||||
}
|
||||
|
||||
/// The inotify file descriptor, for registration on an event loop.
|
||||
pub fn fd(&self) -> RawFd {
|
||||
self.inotify_fd
|
||||
}
|
||||
|
||||
/// Read and process pending inotify events.
|
||||
///
|
||||
/// Returns a list of content changes. Call this when the event loop
|
||||
/// signals that the inotify fd is readable.
|
||||
pub fn process_events(&mut self) -> io::Result<Vec<ContentChange>> {
|
||||
let mut buf = [0u8; 4096];
|
||||
let n = unsafe {
|
||||
libc::read(
|
||||
self.inotify_fd,
|
||||
buf.as_mut_ptr() as *mut libc::c_void,
|
||||
buf.len(),
|
||||
)
|
||||
};
|
||||
|
||||
if n < 0 {
|
||||
let err = io::Error::last_os_error();
|
||||
if err.raw_os_error() == Some(libc::EAGAIN) {
|
||||
return Ok(vec![]);
|
||||
}
|
||||
return Err(err);
|
||||
}
|
||||
|
||||
let mut changes = Vec::new();
|
||||
let mut offset = 0usize;
|
||||
|
||||
while offset < n as usize {
|
||||
let event = unsafe { &*(buf.as_ptr().add(offset) as *const libc::inotify_event) };
|
||||
let name_len = event.len as usize;
|
||||
let name = if name_len > 0 {
|
||||
let name_ptr = unsafe { buf.as_ptr().add(offset + std::mem::size_of::<libc::inotify_event>()) };
|
||||
let name_bytes = unsafe { std::slice::from_raw_parts(name_ptr, name_len) };
|
||||
let end = name_bytes.iter().position(|&b| b == 0).unwrap_or(name_len);
|
||||
Some(String::from_utf8_lossy(&name_bytes[..end]).to_string())
|
||||
} else {
|
||||
None
|
||||
};
|
||||
|
||||
offset += std::mem::size_of::<libc::inotify_event>() + name_len;
|
||||
|
||||
let Some(name) = name else { continue };
|
||||
let mask = event.mask;
|
||||
|
||||
// Event on the root content directory: subdirectory created/deleted.
|
||||
if let Some(dir_path) = self.wd_to_path.get(&event.wd) {
|
||||
let dir_path = dir_path.clone();
|
||||
|
||||
if dir_path == self.content_dir {
|
||||
let sub_path = self.content_dir.join(&name);
|
||||
if (mask & libc::IN_CREATE as u32) != 0 || (mask & libc::IN_MOVED_TO as u32) != 0 {
|
||||
if sub_path.is_dir() {
|
||||
let _ = self.add_watch(&sub_path);
|
||||
changes.push(ContentChange::Created(name.clone()));
|
||||
}
|
||||
}
|
||||
if (mask & libc::IN_DELETE as u32) != 0 || (mask & libc::IN_MOVED_FROM as u32) != 0 {
|
||||
self.remove_watch(&sub_path);
|
||||
changes.push(ContentChange::Deleted(name));
|
||||
}
|
||||
} else {
|
||||
// Event inside an article subdirectory.
|
||||
if name.ends_with(".json") || name.ends_with(".md") {
|
||||
let sys_title = dir_path
|
||||
.file_name()
|
||||
.unwrap()
|
||||
.to_string_lossy()
|
||||
.to_string();
|
||||
|
||||
if (mask & libc::IN_MODIFY as u32) != 0 {
|
||||
changes.push(ContentChange::Modified(sys_title));
|
||||
} else if (mask & libc::IN_CREATE as u32) != 0
|
||||
|| (mask & libc::IN_MOVED_TO as u32) != 0
|
||||
{
|
||||
changes.push(ContentChange::Created(sys_title));
|
||||
} else if (mask & libc::IN_DELETE as u32) != 0
|
||||
|| (mask & libc::IN_MOVED_FROM as u32) != 0
|
||||
{
|
||||
changes.push(ContentChange::Deleted(sys_title));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Deduplicate: keep only the last change per sys_title.
|
||||
let mut seen = HashMap::new();
|
||||
for change in changes {
|
||||
let key = match &change {
|
||||
ContentChange::Created(t) | ContentChange::Modified(t) | ContentChange::Deleted(t) => {
|
||||
t.clone()
|
||||
}
|
||||
};
|
||||
seen.insert(key, change);
|
||||
}
|
||||
|
||||
Ok(seen.into_values().collect())
|
||||
}
|
||||
|
||||
fn add_watch(&mut self, path: &Path) -> io::Result<()> {
|
||||
let c_path = std::ffi::CString::new(path.to_str().unwrap_or("")).map_err(|_| {
|
||||
io::Error::new(io::ErrorKind::InvalidInput, "invalid path for inotify")
|
||||
})?;
|
||||
|
||||
let wd = unsafe { libc::inotify_add_watch(self.inotify_fd, c_path.as_ptr(), EVENT_MASK) };
|
||||
if wd < 0 {
|
||||
return Err(io::Error::last_os_error());
|
||||
}
|
||||
|
||||
self.wd_to_path.insert(wd, path.to_path_buf());
|
||||
self.path_to_wd.insert(path.to_path_buf(), wd);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn remove_watch(&mut self, path: &Path) {
|
||||
if let Some(wd) = self.path_to_wd.remove(path) {
|
||||
unsafe { libc::inotify_rm_watch(self.inotify_fd, wd) };
|
||||
self.wd_to_path.remove(&wd);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for ContentWatcher {
|
||||
fn drop(&mut self) {
|
||||
unsafe { libc::close(self.inotify_fd) };
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use std::fs;
|
||||
|
||||
#[test]
|
||||
fn detect_json_modification() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let content = tmp.path().join("content");
|
||||
let art_dir = content.join("my-article");
|
||||
fs::create_dir_all(&art_dir).unwrap();
|
||||
fs::write(art_dir.join("my-article.json"), r#"{"test": true}"#).unwrap();
|
||||
|
||||
let mut watcher = ContentWatcher::new(&content).unwrap();
|
||||
|
||||
// Modify the file.
|
||||
fs::write(art_dir.join("my-article.json"), r#"{"test": false}"#).unwrap();
|
||||
|
||||
// Give inotify a moment.
|
||||
std::thread::sleep(std::time::Duration::from_millis(50));
|
||||
|
||||
let changes = watcher.process_events().unwrap();
|
||||
assert!(!changes.is_empty());
|
||||
assert!(changes.iter().any(|c| matches!(c, ContentChange::Modified(t) if t == "my-article")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn detect_new_directory() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let content = tmp.path().join("content");
|
||||
fs::create_dir_all(&content).unwrap();
|
||||
|
||||
let mut watcher = ContentWatcher::new(&content).unwrap();
|
||||
|
||||
// Create a new article directory.
|
||||
let art_dir = content.join("new-article");
|
||||
fs::create_dir_all(&art_dir).unwrap();
|
||||
|
||||
std::thread::sleep(std::time::Duration::from_millis(50));
|
||||
|
||||
let changes = watcher.process_events().unwrap();
|
||||
assert!(changes.iter().any(|c| matches!(c, ContentChange::Created(t) if t == "new-article")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn detect_json_delete() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let content = tmp.path().join("content");
|
||||
let art_dir = content.join("delete-me");
|
||||
fs::create_dir_all(&art_dir).unwrap();
|
||||
fs::write(art_dir.join("delete-me.json"), "{}").unwrap();
|
||||
|
||||
let mut watcher = ContentWatcher::new(&content).unwrap();
|
||||
|
||||
// Delete the JSON file.
|
||||
fs::remove_file(art_dir.join("delete-me.json")).unwrap();
|
||||
|
||||
std::thread::sleep(std::time::Duration::from_millis(50));
|
||||
|
||||
let changes = watcher.process_events().unwrap();
|
||||
assert!(changes.iter().any(|c| matches!(c, ContentChange::Deleted(t) if t == "delete-me")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ignore_non_json_files() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let content = tmp.path().join("content");
|
||||
let art_dir = content.join("my-article");
|
||||
fs::create_dir_all(&art_dir).unwrap();
|
||||
|
||||
let mut watcher = ContentWatcher::new(&content).unwrap();
|
||||
|
||||
// Create a non-JSON file.
|
||||
fs::write(art_dir.join("notes.txt"), "not json").unwrap();
|
||||
|
||||
std::thread::sleep(std::time::Duration::from_millis(50));
|
||||
|
||||
let changes = watcher.process_events().unwrap();
|
||||
// Non-JSON files should not trigger Modified events.
|
||||
assert!(changes.iter().all(|c| !matches!(c, ContentChange::Modified(_))));
|
||||
}
|
||||
}
|
||||
122
docs/00-linux.md
Normal file
122
docs/00-linux.md
Normal file
|
|
@ -0,0 +1,122 @@
|
|||
## Linux Kernel Features
|
||||
|
||||
Kernel primitives that the writeonce binary can leverage, mapped to the architectural needs identified in [01-problem.md](./01-problem.md) and [02-recovery.md](./02-recovery.md).
|
||||
|
||||
### File Watching — Content Directory
|
||||
|
||||
| Syscall | Purpose |
|
||||
| ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| `inotify` | Watch the content directory for file creates, modifications, and deletes. Triggers re-indexing and subscriber notification when articles change. Replaces the S3 + Lambda event pipeline entirely. |
|
||||
| `fanotify` | Alternative to inotify with broader scope (filesystem-level events). Useful if watching needs to span mount points or require permission-based filtering. |
|
||||
|
||||
### Async I/O — Server and Subscription Manager
|
||||
|
||||
| Syscall | Purpose |
|
||||
| ---------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| `epoll` | Event-driven I/O multiplexing for the HTTP server and SSE connections. Handles many concurrent subscriber connections without a thread per client. Foundation for the async runtime. |
|
||||
| `io_uring` | Modern async I/O interface (Linux 5.1+). Supports batched, zero-copy submission of read/write/accept operations. Candidate for the embedded storage engine's disk reads and the HTTP server's socket handling in a single event loop. |
|
||||
|
||||
### Efficient File Serving — Content Delivery
|
||||
|
||||
| Syscall | Purpose |
|
||||
| ---------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| `sendfile` | Zero-copy transfer from file descriptor to socket. Serves markdown files and static frontend assets directly from disk to the client without copying through userspace. |
|
||||
| `splice` / `tee` | Zero-copy data transfer between file descriptors via kernel pipe buffers. Useful for streaming .seg file reads directly to HTTP responses. |
|
||||
|
||||
### Embedded Storage — .seg Files and Indexing
|
||||
|
||||
| Syscall | Purpose |
|
||||
| -------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| `mmap` | Memory-map .seg index files for O(1)/O(log n) lookups by `blog-title` without loading entire files into heap memory. Keeps the storage engine's memory footprint proportional to working set, not total content size. |
|
||||
| `madvise` | Hint to the kernel about mmap access patterns (`MADV_SEQUENTIAL` for full scans, `MADV_RANDOM` for index lookups). Improves page cache behavior for both date-ordered listing and title-based retrieval. |
|
||||
| `fallocate` | Pre-allocate disk space for .seg files and append-only logs. Prevents fragmentation and ensures writes don't fail mid-operation due to disk pressure. |
|
||||
| `pread` / `pwritev2` | Positional read/write without seeking. Allows concurrent reads from different offsets in the same .seg file without locking a shared file offset. Pairs well with io_uring for batched operations. |
|
||||
|
||||
### Event Notification — Subscription Manager
|
||||
|
||||
| Syscall | Purpose |
|
||||
| --------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| `eventfd` | Lightweight signaling between the file watcher thread and the subscription manager. When inotify detects a content change, eventfd wakes the async event loop to push diffs to SSE subscribers. |
|
||||
| `timerfd` | Timer as a file descriptor. Can drive periodic tasks (index compaction, subscriber keepalive pings) within the same epoll/io_uring event loop without a separate timer thread. |
|
||||
|
||||
### Process and Resource Management
|
||||
|
||||
| Syscall | Purpose |
|
||||
| -------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| `pidfd` | File descriptor for the process itself. Enables clean self-monitoring and graceful shutdown signaling within the single-binary model. |
|
||||
| `memfd_create` | Anonymous memory-backed file descriptors. Useful for transient storage (e.g., building a .seg index in memory before atomically swapping it to disk). |
|
||||
|
||||
## Runtime
|
||||
|
||||
- Develop async runtime environment built on the kernel primitives above.
|
||||
- Core event loop: `epoll` or `io_uring` driving all I/O (file watch, HTTP, SSE, storage reads).
|
||||
- Single-threaded event loop with thread pool for blocking operations (disk I/O, index rebuilds).
|
||||
- All file descriptors (inotify, eventfd, timerfd, sockets) registered on one multiplexer — one process, one loop, no external dependencies.
|
||||
|
||||
## Database Subscription
|
||||
|
||||
No SSE. No WebSocket. Subscriptions are handled at the kernel level using file descriptors and the event loop that already drives the rest of the system.
|
||||
|
||||
### Why Not SSE or WebSocket
|
||||
|
||||
Both are HTTP-layer protocols that add framing, connection management, and protocol negotiation on top of what the kernel already provides. If the binary already owns the event loop (`epoll` / `io_uring`), the content directory (`inotify`), and the subscriber sockets (raw fds), there is no reason to layer another protocol on top. The kernel *is* the subscription engine.
|
||||
|
||||
### Register Macro
|
||||
|
||||
A subscription is a registration of interest in a content query, bound to a file descriptor. The `register` macro wires a query pattern to the event loop at compile time:
|
||||
|
||||
```rust
|
||||
register! {
|
||||
#{blog-title} => notify(fd)
|
||||
}
|
||||
```
|
||||
|
||||
This expands to:
|
||||
|
||||
1. **Parse the query pattern** — `#{blog-title}` resolves to a specific article's `sys_title` key in the .seg index.
|
||||
2. **Bind to inotify watch** — The content file backing that `blog-title` gets an inotify watch descriptor. When the file is modified, inotify fires.
|
||||
3. **Map watch → subscriber fd** — The event loop maps the inotify event to every file descriptor that registered interest in that `blog-title`.
|
||||
4. **Write to fd** — The subscriber's socket fd receives the diff (or a notification payload) via a direct `write()` — no HTTP framing, no SSE `data:` lines, no WebSocket frames.
|
||||
|
||||
### Flow
|
||||
|
||||
```
|
||||
inotify (content file changed)
|
||||
|
|
||||
v
|
||||
epoll wakes → lookup blog-title in subscription table
|
||||
|
|
||||
v
|
||||
for each registered fd:
|
||||
write(fd, diff_payload)
|
||||
```
|
||||
|
||||
### Subscription Table
|
||||
|
||||
The subscription table is an in-process map:
|
||||
|
||||
```
|
||||
blog-title → Vec<RawFd>
|
||||
```
|
||||
|
||||
- **Register**: client connects → fd is added to the vec for the requested `blog-title`.
|
||||
- **Deregister**: client disconnects → fd is removed (detected via `EPOLLHUP` / `EPOLLRDHUP`).
|
||||
- **Notify**: inotify fires for a content file → resolve `blog-title` from path → write to all registered fds.
|
||||
|
||||
No broker, no message queue, no protocol layer. The kernel's fd lifecycle (`epoll` for readiness, `inotify` for content changes, `close` for cleanup) handles the full subscription lifecycle.
|
||||
|
||||
### Query Patterns
|
||||
|
||||
The `register` macro supports patterns beyond single titles:
|
||||
|
||||
```rust
|
||||
register! {
|
||||
#{blog-title} // single article by sys_title
|
||||
#[tag="rust"] // all articles matching a tag
|
||||
#[published > "2026-01-01"] // date-range filter
|
||||
#* // all content changes
|
||||
}
|
||||
```
|
||||
|
||||
Each pattern resolves to a set of inotify watch descriptors. When the watched set changes (new article added that matches the filter), the subscription table updates automatically during re-indexing.
|
||||
|
||||
66
docs/01-problem.md
Normal file
66
docs/01-problem.md
Normal file
|
|
@ -0,0 +1,66 @@
|
|||
# Problem Statement
|
||||
|
||||
The current writeonce architecture works, but it carries weight that the project doesn't need. This document identifies the structural problems that motivate the redesign described in [02-recovery.md](./02-recovery.md).
|
||||
|
||||
## Too Many Moving Parts
|
||||
|
||||
A single article edit touches five systems before it reaches a reader:
|
||||
|
||||
```
|
||||
local file --> S3 bucket --> Lambda --> Rust API --> PostgreSQL --> Angular app
|
||||
```
|
||||
|
||||
Each hop is a failure point. Each system has its own deployment, its own logs, its own configuration. The Lambda needs AWS credentials and an API token. The API needs a DATABASE_URL and an AWS_INFRA_BASE_URL. The infrastructure API needs its own AWS SDK config. For a platform that serves markdown files, this is disproportionate complexity.
|
||||
|
||||
## External Database for Derived Data
|
||||
|
||||
PostgreSQL stores articles as JSONB — but the database is not the source of truth. The articles repository is. The database is a derived cache that requires:
|
||||
|
||||
- A running PostgreSQL 17 instance
|
||||
- Diesel ORM with migrations
|
||||
- Connection pooling (r2d2)
|
||||
- A separate Rust API server to mediate access
|
||||
|
||||
If the database dies, recovery means re-syncing every article from S3 through the Lambda pipeline. The database adds operational burden without adding authority.
|
||||
|
||||
## Five Repositories for One Product
|
||||
|
||||
The project spans five git submodules across four languages:
|
||||
|
||||
| Repo | Language | Purpose |
|
||||
|------|----------|---------|
|
||||
| writeonce-articles-s3 | JSON + MD | Content |
|
||||
| lambda-function | Go | Sync trigger |
|
||||
| writeonce-api | Rust | API server |
|
||||
| aws-infra | Rust | AWS bridge |
|
||||
| writeonce-app | Angular/TS | Frontend |
|
||||
|
||||
Each has its own CI/CD pipeline, Dockerfile, and deployment target. Coordinating changes across repos (e.g., adding a new article field) requires touching multiple codebases, multiple pipelines, and multiple deploys.
|
||||
|
||||
## No Real-Time Content Updates
|
||||
|
||||
The current flow is request-response only. When an article is updated:
|
||||
|
||||
1. Author syncs files to S3
|
||||
2. Lambda fires and upserts via API
|
||||
3. The frontend knows nothing until the next page load
|
||||
|
||||
There is no mechanism for the client to learn that content has changed. No subscriptions, no push, no invalidation. The architecture explicitly avoids WebSocket, but offers no alternative for real-time awareness.
|
||||
|
||||
## AWS Dependency for a File-Based System
|
||||
|
||||
The content is markdown and JSON files — inherently local, portable, and simple. But the current pipeline requires:
|
||||
|
||||
- An S3 bucket to host them
|
||||
- A Lambda function to watch for changes
|
||||
- AWS SDK configuration in two Rust services
|
||||
- Pulumi infrastructure-as-code to manage the Lambda + S3 setup
|
||||
- IAM credentials across multiple components
|
||||
|
||||
The cloud infrastructure exists to shuttle files from one place to another. The files themselves don't need the cloud — they need to be read, indexed, and served.
|
||||
|
||||
## Summary
|
||||
|
||||
The core problems are **accidental complexity** and **infrastructure overhead**. The content model (markdown + JSON metadata) is sound. The content-as-code principle is right. But the delivery mechanism — five repos, three languages on the backend, an external database, a cloud event pipeline — is heavier than the problem requires.
|
||||
|
||||
The question `02-recovery.md` answers: what if the database, the server, and the client were one thing?
|
||||
158
docs/02-recovery.md
Normal file
158
docs/02-recovery.md
Normal file
|
|
@ -0,0 +1,158 @@
|
|||
# Recovery — The Target Architecture
|
||||
|
||||
This document describes where writeonce is going: a single, self-contained binary that owns its own storage, serves its own content, and pushes updates to connected clients in real-time — with no external database, no cloud pipeline, and no separate API server.
|
||||
|
||||
## Guiding Principle
|
||||
|
||||
**Everything in one process.** The database, the server logic, and the client-facing interface all live in a single codebase and ship as a single executable. If you can run the binary, you have the full platform.
|
||||
|
||||
## Own Database
|
||||
|
||||
The current PostgreSQL instance is a derived cache — it stores JSONB copies of files that already exist as the source of truth. The recovery architecture eliminates this indirection entirely.
|
||||
|
||||
### What Changes
|
||||
|
||||
- **No external database.** No PostgreSQL, no Diesel ORM, no connection pooling, no migrations.
|
||||
- **Local file storage.** Markdown files and JSON metadata files are stored in a local directory, just as they are today in `writeonce-articles-s3/`. The file system *is* the database.
|
||||
- **Custom storage segments (.seg files).** Research area: segment files that provide efficient read access, indexing, and potentially append-only writes for content. Think of these as a lightweight, purpose-built storage layer — not a general-purpose database engine, but enough to support indexed lookups by `blog-title` and ordered listing by date.
|
||||
- **Indexed by blog-title.** The `sys_title` / blog-title field remains the primary key for content retrieval. The embedded storage must support O(1) or O(log n) lookups by this field.
|
||||
|
||||
### What Stays the Same
|
||||
|
||||
- Articles are still structured as JSON metadata + Markdown content pairs.
|
||||
- The `sys_title`, `published`, `tags`, `author`, and section structure remain the content model.
|
||||
- Content is still the source of truth — but now it's read directly from local storage instead of being derived through a sync pipeline.
|
||||
|
||||
## No AWS Infrastructure
|
||||
|
||||
The current architecture uses S3 as a file host and Lambda as a sync trigger. In the target architecture, there is nothing to sync *to* — the files are already where they need to be.
|
||||
|
||||
### What Gets Removed
|
||||
|
||||
| Current Component | Why It Existed | Why It's No Longer Needed |
|
||||
|---|---|---|
|
||||
| S3 bucket | Remote file storage | Files live locally alongside the binary |
|
||||
| Lambda function (Go) | Watch S3 for changes, call API | No remote store to watch — file changes are local |
|
||||
| aws-infra service (Rust) | Bridge to AWS S3/EC2 APIs | No AWS dependency |
|
||||
| Pulumi IaC | Manage Lambda + S3 resources | No cloud resources to manage |
|
||||
|
||||
### What Replaces It
|
||||
|
||||
The binary watches its own content directory. When a file changes (new article, updated metadata), the embedded database re-indexes and notifies subscribers. The deployment model becomes:
|
||||
|
||||
```
|
||||
1. Place the binary on a server
|
||||
2. Point it at a content directory
|
||||
3. It serves
|
||||
```
|
||||
|
||||
No credentials, no IAM roles, no SDK configuration.
|
||||
|
||||
## No Separate API
|
||||
|
||||
Today, `writeonce-api` is a standalone Actix-web server that mediates between the frontend and the database. In the target architecture, the server logic is embedded in the same process as the database and the content renderer.
|
||||
|
||||
### What This Means
|
||||
|
||||
- **No HTTP hop between database and server.** Queries go directly from the request handler to the storage engine in-process. No network serialization, no connection pool, no ORM layer.
|
||||
- **Single codebase.** No multi-repo coordination. A new article field is added once — in the content model — and it flows through storage, indexing, and rendering in the same compilation unit.
|
||||
- **Single deployment.** One binary, one container, one process. No docker-compose orchestrating API + database + infra services.
|
||||
|
||||
The binary still exposes HTTP endpoints — it's still a web server. But it's a web server with an embedded database, not a web server that talks to an external one.
|
||||
|
||||
## Real-Time Subscriptions Without WebSocket
|
||||
|
||||
The current architecture has no mechanism for pushing content updates to connected clients. The target architecture adds real-time subscriptions, but explicitly without WebSocket.
|
||||
|
||||
### Why Not WebSocket
|
||||
|
||||
WebSocket adds connection state management, heartbeat logic, reconnection handling, and protocol upgrade complexity. For a content platform where updates are infrequent (articles are published, not streamed), the overhead isn't justified.
|
||||
|
||||
### Subscription Model
|
||||
|
||||
The target is a subscription mechanism where:
|
||||
|
||||
- A client subscribes to a content query (e.g., "all published articles" or "article with sys_title X")
|
||||
- When the underlying data changes, the server pushes the relevant diff to the subscriber
|
||||
- No polling from the client side
|
||||
|
||||
Candidate approaches to research:
|
||||
|
||||
- **Server-Sent Events (SSE)** — unidirectional push over HTTP. Simple, well-supported, no protocol upgrade. Natural fit for infrequent content updates.
|
||||
- **SpacetimeDB-style subscriptions** — clients register queries, the engine tracks which rows match, and only sends diffs when the result set changes. This is the aspirational model.
|
||||
- **Long polling** — fallback option. Simple but less efficient than SSE for multiple subscribers.
|
||||
|
||||
The key constraint: the subscription mechanism must work without requiring clients to maintain persistent bidirectional connections.
|
||||
|
||||
## Target Architecture
|
||||
|
||||
```
|
||||
content directory
|
||||
(JSON + MD files, .seg index)
|
||||
|
|
||||
| file watch + re-index
|
||||
v
|
||||
+---------------------------+
|
||||
| writeonce binary |
|
||||
| |
|
||||
| +-------------------+ |
|
||||
| | embedded storage | | .seg files, blog-title index
|
||||
| | (read/write/index)| |
|
||||
| +-------------------+ |
|
||||
| | |
|
||||
| +-------------------+ |
|
||||
| | server logic | | route handlers, content queries
|
||||
| | (HTTP endpoints) | |
|
||||
| +-------------------+ |
|
||||
| | |
|
||||
| +-------------------+ |
|
||||
| | subscription mgr | | SSE / query-based push
|
||||
| | (real-time push) | |
|
||||
| +-------------------+ |
|
||||
| |
|
||||
+---------------------------+
|
||||
|
|
||||
HTTP / SSE
|
||||
|
|
||||
v
|
||||
+-------------------+
|
||||
| frontend app | Angular or successor
|
||||
| (browser client) |
|
||||
+-------------------+
|
||||
```
|
||||
|
||||
## Single Repository
|
||||
|
||||
The five current repos collapse into one:
|
||||
|
||||
```
|
||||
writeonce/
|
||||
content/ # articles (JSON + MD), images, assets
|
||||
storage/ # embedded database engine (.seg files, indexing)
|
||||
server/ # HTTP handlers, subscription manager
|
||||
frontend/ # client application
|
||||
writeonce.toml # configuration (port, content dir, index settings)
|
||||
```
|
||||
|
||||
One repo. One build. One deploy artifact.
|
||||
|
||||
## What Needs Research
|
||||
|
||||
| Area | Question | Notes |
|
||||
|------|----------|-------|
|
||||
| **.seg file format** | What storage format gives efficient indexed reads over JSON+MD content? | Look at LSM trees, append-only logs, SQLite's page format for inspiration |
|
||||
| **File watching** | How to efficiently detect content changes on Linux/macOS? | `inotify` on Linux, `kqueue` on macOS, or cross-platform via `notify` crate |
|
||||
| **SSE vs alternatives** | Is SSE sufficient for the subscription model, or is something custom needed? | SSE handles the "push diffs to subscribers" case well for low-frequency updates |
|
||||
| **Index structure** | What index structure supports `blog-title` lookup + date-ordered listing? | B-tree or hash index for title, sorted set for date ordering |
|
||||
| **Language choice** | Continue with Rust for the unified binary? | Rust fits: single binary output, no runtime, strong typing, existing team knowledge |
|
||||
| **Frontend coupling** | Should the frontend be embedded in the binary (serve static assets) or remain separate? | Embedding simplifies deployment; separate allows independent frontend iteration |
|
||||
|
||||
## Migration Path
|
||||
|
||||
The transition from current to target doesn't have to be all-or-nothing:
|
||||
|
||||
1. **Phase 1** — Build the embedded storage engine. Read JSON+MD files from a local directory, index by `blog-title`, serve via HTTP. No AWS, no PostgreSQL. This alone replaces `writeonce-api` + `aws-infra` + `lambda-function` + PostgreSQL.
|
||||
2. **Phase 2** — Add real-time subscriptions (SSE). Clients subscribe to content queries and receive push updates when files change.
|
||||
3. **Phase 3** — Collapse repositories. Move frontend into the unified codebase. Ship as a single binary that serves both API and static assets.
|
||||
|
||||
Each phase produces a working system. The current architecture can run in parallel until the new one is ready.
|
||||
184
docs/03-data.md
Normal file
184
docs/03-data.md
Normal file
|
|
@ -0,0 +1,184 @@
|
|||
# Data Layer — Local Storage with Subscriptions
|
||||
|
||||
This document describes the embedded data layer that replaces PostgreSQL: local `.seg` files with indexing, and a subscription model where clients register queries and receive diffs on route visit — no polling required.
|
||||
|
||||
## .seg File Storage
|
||||
|
||||
The `.seg` (segment) format is the on-disk representation of article data. Each segment file holds serialized article content with positional indexing for fast lookups.
|
||||
|
||||
### Design Goals
|
||||
|
||||
- **No external database process.** The binary reads and writes `.seg` files directly. No socket connections, no protocol negotiation, no separate daemon.
|
||||
- **Indexed by blog-title.** The primary access pattern is `GET /blog/:sys_title`. The storage layer must resolve a `sys_title` to its article content without scanning all files.
|
||||
- **Append-friendly.** New articles and updates append to the segment. Deletes are tombstoned and compacted later.
|
||||
- **Human-readable source.** The JSON + Markdown files remain the authoring format. `.seg` files are a derived index — if they're deleted, they can be rebuilt from the content directory.
|
||||
|
||||
### Proposed Structure
|
||||
|
||||
```
|
||||
content/
|
||||
linux-misc/
|
||||
linux-misc.json # authored metadata (source of truth)
|
||||
linux-misc.md # authored content (source of truth)
|
||||
aws-lambda-pulumi/
|
||||
aws-lambda-pulumi.json
|
||||
aws-lambda-pulumi.md
|
||||
|
||||
data/
|
||||
articles.seg # serialized article records
|
||||
index/
|
||||
title.idx # blog-title -> offset mapping
|
||||
date.idx # publish date -> offset (sorted)
|
||||
tags.idx # tag -> [offsets] (inverted index)
|
||||
```
|
||||
|
||||
The `content/` directory is what the author edits. The `data/` directory is what the engine builds and queries. Losing `data/` is a cold start, not data loss.
|
||||
|
||||
### Segment File Internals
|
||||
|
||||
```
|
||||
+------------------+
|
||||
| segment header | magic bytes, version, record count
|
||||
+------------------+
|
||||
| record 0 | length-prefixed serialized article
|
||||
+------------------+
|
||||
| record 1 |
|
||||
+------------------+
|
||||
| ... |
|
||||
+------------------+
|
||||
| record N |
|
||||
+------------------+
|
||||
```
|
||||
|
||||
Each record is a length-prefixed byte sequence containing the full article (metadata + content merged). Records are addressed by byte offset from the start of the file.
|
||||
|
||||
### Index Files
|
||||
|
||||
**title.idx** — Hash map serialized to disk. Maps `sys_title` (string) to byte offset in `articles.seg`. Loaded into memory at startup for O(1) lookups.
|
||||
|
||||
**date.idx** — Sorted array of `(timestamp, offset)` pairs. Supports range queries for "articles published between X and Y" and ordered listing for the homepage.
|
||||
|
||||
**tags.idx** — Inverted index. Maps each tag string to a list of offsets. Supports "all articles tagged with X" queries.
|
||||
|
||||
On startup, index files are memory-mapped or loaded into heap. On content change, affected indexes are rebuilt incrementally.
|
||||
|
||||
## Subscription Model
|
||||
|
||||
The subscription model is inspired by SpacetimeDB: clients register queries, and the engine tracks which results match. When underlying data changes, only the relevant diffs are pushed to subscribers.
|
||||
|
||||
### How It Works
|
||||
|
||||
```
|
||||
Client A Server Content Dir
|
||||
| | |
|
||||
|--- GET /blog/linux-misc -| |
|
||||
| |-- read from .seg index ---->|
|
||||
|<-- article + SSE stream -| |
|
||||
| | |
|
||||
| (subscribed to | |
|
||||
| sys_title=linux-misc) | |
|
||||
| | |
|
||||
| |<-- file change detected ----|
|
||||
| | |
|
||||
| |-- re-index article -------->|
|
||||
| |-- diff against last push -->|
|
||||
| | |
|
||||
|<-- SSE: updated content -| |
|
||||
| | |
|
||||
```
|
||||
|
||||
### Route-Based Subscription
|
||||
|
||||
When a user visits a route, the response includes both the current content and an SSE stream. The client is automatically subscribed to changes for that query — no explicit subscription handshake needed.
|
||||
|
||||
```
|
||||
GET /blog/linux-misc
|
||||
```
|
||||
|
||||
Response:
|
||||
```
|
||||
HTTP/1.1 200 OK
|
||||
Content-Type: text/html
|
||||
|
||||
<!-- full article content rendered -->
|
||||
|
||||
<!-- SSE connection opened for this query -->
|
||||
<script>
|
||||
const source = new EventSource('/subscribe/blog/linux-misc');
|
||||
source.onmessage = (event) => {
|
||||
// apply diff to current content
|
||||
};
|
||||
</script>
|
||||
```
|
||||
|
||||
The subscription lives as long as the browser tab is open. When the user navigates away, the EventSource closes and the server drops the subscription. No heartbeat management, no reconnection logic beyond what SSE provides natively (automatic reconnect is built into the EventSource API).
|
||||
|
||||
### Query Registration
|
||||
|
||||
Subscriptions are not limited to single-article lookups. The engine supports registering arbitrary content queries:
|
||||
|
||||
| Query Type | Example | Subscription Behavior |
|
||||
|---|---|---|
|
||||
| Single article | `sys_title = "linux-misc"` | Push when this specific article changes |
|
||||
| All published | `published = true` | Push when any article is published or unpublished |
|
||||
| By tag | `tags contains "rust"` | Push when a rust-tagged article is added, removed, or updated |
|
||||
| Homepage list | `published = true ORDER BY date DESC LIMIT 10` | Push when the top-10 list changes |
|
||||
|
||||
The server maintains a registry of active subscriptions. On each content change, it evaluates which subscriptions are affected and pushes diffs only to those clients.
|
||||
|
||||
### Diff Format
|
||||
|
||||
When content changes, the server doesn't resend the full article. It sends a minimal diff:
|
||||
|
||||
```json
|
||||
{
|
||||
"type": "update",
|
||||
"sys_title": "linux-misc",
|
||||
"changes": {
|
||||
"content.sections[2].paragraphs[0]": "Updated paragraph text...",
|
||||
"content.tags": ["linux", "kernel", "new-tag"]
|
||||
},
|
||||
"version": 42
|
||||
}
|
||||
```
|
||||
|
||||
The `version` field enables clients to detect missed updates and request a full resync if needed.
|
||||
|
||||
## Sample Dataset
|
||||
|
||||
To validate the storage engine and subscription model, a sample dataset should exercise the core access patterns:
|
||||
|
||||
### Articles
|
||||
|
||||
| sys_title | tags | published | purpose |
|
||||
|---|---|---|---|
|
||||
| `sample-getting-started` | `[tutorial, beginner]` | true | Basic article, tests single-article subscription |
|
||||
| `sample-rust-patterns` | `[rust, patterns]` | true | Tests tag-based queries |
|
||||
| `sample-draft-wip` | `[draft]` | false | Tests published filter — should not appear in public queries |
|
||||
| `sample-long-form` | `[deep-dive, rust]` | true | Multiple sections, images, code snippets — tests complex content rendering |
|
||||
| `sample-frequently-updated` | `[changelog]` | true | Updated often — tests subscription diff delivery |
|
||||
|
||||
### Test Scenarios
|
||||
|
||||
1. **Cold start** — Delete `data/`, start the binary. It should rebuild `.seg` and index files from `content/` and serve all articles.
|
||||
2. **Single article query** — `GET /blog/sample-getting-started` returns the article and opens an SSE subscription.
|
||||
3. **Live update** — Edit `sample-frequently-updated.json` while a client is subscribed. The client should receive an SSE event with the diff.
|
||||
4. **Tag query** — Subscribe to `tags contains "rust"`. Both `sample-rust-patterns` and `sample-long-form` should be in the result set. Adding a new article tagged `rust` should trigger a push.
|
||||
5. **Publish toggle** — Change `sample-draft-wip` from `published: false` to `true`. Clients subscribed to the homepage list should receive a push with the new article added.
|
||||
|
||||
## SpacetimeDB Reference
|
||||
|
||||
SpacetimeDB is the primary architectural inspiration for the subscription model. Key concepts to study:
|
||||
|
||||
- **Modules** — server logic that runs inside the database, not beside it
|
||||
- **Subscription queries** — clients register SQL-like queries; the engine evaluates them incrementally on each transaction
|
||||
- **Incremental view maintenance** — only recompute the parts of a query result that changed
|
||||
- **Client SDK generation** — type-safe client code generated from the server schema
|
||||
|
||||
Add SpacetimeDB as a reference submodule for quick access to their implementation patterns:
|
||||
|
||||
```bash
|
||||
git submodule add https://github.com/clockworklabs/SpacetimeDB.git references/spacetimedb
|
||||
```
|
||||
|
||||
The goal is not to replicate SpacetimeDB — it's to take its subscription semantics and apply them to a much narrower domain (blog content), where the simplicity of the problem allows a simpler implementation.
|
||||
216
docs/04-ui.md
Normal file
216
docs/04-ui.md
Normal file
|
|
@ -0,0 +1,216 @@
|
|||
# User Interface — Server-Rendered HTMLX
|
||||
|
||||
No Angular. No React. No frontend framework. The UI is a set of `.htmlx` template files that the server parses, populates with content from the embedded database, and serves as plain HTML. Real-time updates arrive via SSE and are applied with minimal client-side scripting.
|
||||
|
||||
## Why Not Angular
|
||||
|
||||
The current `writeonce-app` is an Angular 18 SPA with Tailwind, PrismJS, ngx-markdown, and FontAwesome. It works, but it's a heavy delivery mechanism for what is fundamentally a read-heavy content site:
|
||||
|
||||
- **~200MB of `node_modules`** for a site that renders markdown articles
|
||||
- **Client-side routing** for content that doesn't need it — every article is a distinct URL, not an interactive application
|
||||
- **JavaScript-dependent rendering** — content doesn't exist until Angular boots, hydrates, and fetches from the API
|
||||
- **Separate build pipeline** — `npm run build` produces static assets that must be deployed to nginx independently of the API
|
||||
|
||||
The content is static between updates. The interactivity is limited to navigation and code highlighting. A server-rendered approach matches the actual requirements.
|
||||
|
||||
## HTMLX Templates
|
||||
|
||||
The author defines the site layout using `.htmlx` files — HTML with embedded data bindings that the server resolves at render time.
|
||||
|
||||
### Template Structure
|
||||
|
||||
```
|
||||
templates/
|
||||
layout.htmlx # outer shell: <html>, <head>, <body>
|
||||
header.htmlx # site header, navigation
|
||||
footer.htmlx # site footer
|
||||
home.htmlx # homepage: article list
|
||||
article.htmlx # single article view
|
||||
about.htmlx # static page
|
||||
contact.htmlx # static page
|
||||
components/
|
||||
article-card.htmlx # summary card for article listings
|
||||
code-snippet.htmlx # code block with language + title
|
||||
img-caption.htmlx # image with caption
|
||||
section.htmlx # article section (heading + paragraphs)
|
||||
```
|
||||
|
||||
### Template Syntax
|
||||
|
||||
Templates use a binding syntax that references content from the database. The server parses these bindings, resolves them against the current content, and outputs plain HTML.
|
||||
|
||||
```html
|
||||
<!-- header.htmlx -->
|
||||
<header>
|
||||
<nav>
|
||||
<a href="/">writeonce</a>
|
||||
<a href="/about">about</a>
|
||||
<a href="/contact">contact</a>
|
||||
</nav>
|
||||
</header>
|
||||
```
|
||||
|
||||
```html
|
||||
<!-- article.htmlx -->
|
||||
<article>
|
||||
<h1>{{article.title}}</h1>
|
||||
<p class="meta">by {{article.author}} · {{article.tags}}</p>
|
||||
|
||||
{{#each article.sections}}
|
||||
<section>
|
||||
<h2>{{heading}}</h2>
|
||||
{{#each paragraphs}}
|
||||
<p>{{this}}</p>
|
||||
{{/each}}
|
||||
</section>
|
||||
{{/each}}
|
||||
|
||||
{{#each article.codes}}
|
||||
{{> code-snippet snippet=this}}
|
||||
{{/each}}
|
||||
|
||||
{{#each article.images}}
|
||||
{{> img-caption image=this}}
|
||||
{{/each}}
|
||||
</article>
|
||||
```
|
||||
|
||||
```html
|
||||
<!-- home.htmlx -->
|
||||
<main>
|
||||
<h1>articles</h1>
|
||||
{{#each articles}}
|
||||
{{> article-card article=this}}
|
||||
{{/each}}
|
||||
</main>
|
||||
```
|
||||
|
||||
The `{{> partial}}` syntax includes another `.htmlx` file as a component. The server resolves these at render time — no client-side component tree.
|
||||
|
||||
### Content Subscription in Templates
|
||||
|
||||
Templates declare what data they need. The server resolves these declarations against the embedded database and subscribes the client to changes:
|
||||
|
||||
```html
|
||||
<!-- article.htmlx -->
|
||||
<!-- subscribe: article WHERE sys_title = :route_param -->
|
||||
|
||||
<article>
|
||||
<h1>{{article.title}}</h1>
|
||||
...
|
||||
</article>
|
||||
```
|
||||
|
||||
```html
|
||||
<!-- home.htmlx -->
|
||||
<!-- subscribe: articles WHERE published = true ORDER BY date DESC LIMIT 10 -->
|
||||
|
||||
<main>
|
||||
{{#each articles}}
|
||||
{{> article-card article=this}}
|
||||
{{/each}}
|
||||
</main>
|
||||
```
|
||||
|
||||
The `<!-- subscribe: ... -->` comment is a directive to the server. It declares the query that populates the template's data context. The same query is used to register an SSE subscription for live updates (as described in [03-data.md](./03-data.md)).
|
||||
|
||||
## Rendering Pipeline
|
||||
|
||||
```
|
||||
Browser request
|
||||
|
|
||||
v
|
||||
Route match (/blog/linux-misc)
|
||||
|
|
||||
v
|
||||
Load template (article.htmlx)
|
||||
|
|
||||
v
|
||||
Parse subscribe directive
|
||||
(article WHERE sys_title = "linux-misc")
|
||||
|
|
||||
v
|
||||
Query embedded database (.seg index)
|
||||
|
|
||||
v
|
||||
Resolve template bindings ({{article.title}}, etc.)
|
||||
|
|
||||
v
|
||||
Compose with layout.htmlx + header.htmlx + footer.htmlx
|
||||
|
|
||||
v
|
||||
Inject SSE subscription script
|
||||
|
|
||||
v
|
||||
Send complete HTML response
|
||||
```
|
||||
|
||||
The browser receives a fully rendered page on first load. No JavaScript framework boots. No API call fires. The content is already in the HTML.
|
||||
|
||||
## Live Updates via SSE
|
||||
|
||||
After the initial HTML is delivered, a small inline script opens an SSE connection for the page's subscription query:
|
||||
|
||||
```html
|
||||
<script>
|
||||
const source = new EventSource('/subscribe/blog/linux-misc');
|
||||
source.onmessage = (event) => {
|
||||
const diff = JSON.parse(event.data);
|
||||
applyDiff(diff);
|
||||
};
|
||||
</script>
|
||||
```
|
||||
|
||||
The `applyDiff` function is a lightweight client-side updater — it targets DOM elements by data attribute and patches their content. No virtual DOM, no reconciliation, no framework. For a content site where updates are infrequent and localized (a paragraph changed, a tag was added), direct DOM manipulation is sufficient.
|
||||
|
||||
```html
|
||||
<h1 data-bind="article.title">Linux Misc</h1>
|
||||
<p data-bind="article.sections[0].paragraphs[0]">First paragraph...</p>
|
||||
```
|
||||
|
||||
When a diff arrives for `article.title`, the script finds the element with `data-bind="article.title"` and replaces its text content. This is the minimal client-side code the architecture requires.
|
||||
|
||||
## Code Highlighting
|
||||
|
||||
The current frontend uses PrismJS for syntax highlighting. In the server-rendered model, highlighting can happen at either layer:
|
||||
|
||||
**Server-side (preferred):** The server parses code blocks during template rendering and emits pre-highlighted HTML with CSS classes. The browser only needs the PrismJS CSS theme, not the JavaScript library. This eliminates client-side parsing entirely.
|
||||
|
||||
**Client-side (fallback):** Include PrismJS as a small script that runs on page load and on SSE update. Simpler to implement initially but adds a JavaScript dependency.
|
||||
|
||||
## Markdown Rendering
|
||||
|
||||
The current frontend uses `ngx-markdown` and `marked` to parse markdown in the browser. In the target architecture, markdown is rendered to HTML on the server during template composition. The browser never sees raw markdown.
|
||||
|
||||
This aligns with the content model: the JSON metadata already defines the article structure (sections, paragraphs, code snippets, images). The markdown file provides prose content. The server combines both into final HTML — the template just places the pre-rendered blocks.
|
||||
|
||||
## What Gets Removed
|
||||
|
||||
| Current (Angular) | Target (HTMLX) |
|
||||
|---|---|
|
||||
| `writeonce-app/` (full Angular project) | `templates/` (handful of .htmlx files) |
|
||||
| `node_modules/` (~200MB) | None |
|
||||
| `angular.json`, `tsconfig.json`, `karma.conf.js` | None |
|
||||
| npm build pipeline | Template parsed at request time |
|
||||
| Nginx static file serving | Binary serves its own HTML |
|
||||
| Client-side routing | Server-side route matching |
|
||||
| Client-side markdown parsing | Server-side rendering |
|
||||
| Client-side code highlighting | Server-side or minimal JS |
|
||||
|
||||
## Styling
|
||||
|
||||
Templates use plain CSS. Tailwind can optionally be used as a build-time utility (generating a static CSS file), but there is no runtime CSS framework. The author writes styles in a `styles.css` file that the server serves as a static asset.
|
||||
|
||||
```
|
||||
templates/
|
||||
styles/
|
||||
main.css # site-wide styles
|
||||
article.css # article-specific styles
|
||||
code-theme.css # syntax highlighting theme (PrismJS compatible)
|
||||
```
|
||||
|
||||
## Template Authoring Experience
|
||||
|
||||
The `.htmlx` files are editable by the same author who writes articles. The template syntax is intentionally close to HTML — there's no JSX, no TypeScript, no build step. An author who knows HTML can modify the site layout.
|
||||
|
||||
This closes the loop on the writeonce philosophy: the author writes content (markdown + JSON) and layout (`.htmlx` + CSS) as files, and the binary turns them into a live site.
|
||||
139
docs/05-datalayer.md
Normal file
139
docs/05-datalayer.md
Normal file
|
|
@ -0,0 +1,139 @@
|
|||
# Data Layer — Implementation Status
|
||||
|
||||
The embedded data layer described in [02-recovery.md](./02-recovery.md) and [03-data.md](./03-data.md) has been implemented as a Cargo workspace with 8 crates. All 44 tests pass. No external database, no AWS, no tokio — direct Linux syscalls on a custom event loop.
|
||||
|
||||
## Workspace Structure
|
||||
|
||||
```
|
||||
writeonce-all/
|
||||
Cargo.toml # workspace root
|
||||
docs/ # architecture documentation
|
||||
sample-content/ # 5 test articles for validation
|
||||
crates/
|
||||
wo-model/ # content model
|
||||
wo-seg/ # .seg file format
|
||||
wo-index/ # index files
|
||||
wo-store/ # unified storage engine
|
||||
wo-watch/ # inotify file watcher
|
||||
wo-event/ # epoll event loop
|
||||
wo-sub/ # subscription system
|
||||
wo-rt/ # custom runtime
|
||||
```
|
||||
|
||||
## Crate Summary
|
||||
|
||||
| Crate | Purpose | Tests | Key Types |
|
||||
|-------|---------|-------|-----------|
|
||||
| **wo-model** | Article structs matching existing JSON schema, `ContentLoader` for directory walking | 8 | `Article`, `ArticleContent`, `ArticleBody`, `Section`, `CodeSnippet`, `ContentLoader` |
|
||||
| **wo-seg** | Binary `.seg` file format — length-prefixed records, tombstoning, positional I/O | 6 | `SegWriter`, `SegReader`, `SegHeader` |
|
||||
| **wo-index** | Three index types for O(1) and O(log n) access patterns | 8 | `TitleIndex`, `DateIndex`, `TagIndex` |
|
||||
| **wo-store** | Unified storage engine composing seg + indexes, cold-start rebuild | 3 | `Store` |
|
||||
| **wo-watch** | Content directory watcher using inotify | 4 | `ContentWatcher`, `ContentChange` |
|
||||
| **wo-event** | Custom event loop on epoll with eventfd, timerfd, signalfd | 5 | `EventLoop`, `EventFd`, `TimerFd`, `SignalFd` |
|
||||
| **wo-sub** | Subscription manager with fd-based notifications, `register!` macro | 6 | `SubscriptionManager`, `Subscription`, `Notification` |
|
||||
| **wo-rt** | Runtime tying all crates together — single process, single event loop | 4 | `Runtime`, `RuntimeHandle`, `Config` |
|
||||
|
||||
## Linux Kernel Syscalls Used
|
||||
|
||||
| Syscall | Crate | Purpose |
|
||||
|---------|-------|---------|
|
||||
| `pread` / `pwrite` | wo-seg | Positional read/write for .seg records without seeking |
|
||||
| `fallocate` | wo-seg | Pre-allocate .seg file space to reduce fragmentation |
|
||||
| `epoll_create1` / `epoll_ctl` / `epoll_wait` | wo-event | Event-driven I/O multiplexing for the main loop |
|
||||
| `eventfd` | wo-event, wo-sub | Lightweight signaling between watcher and subscription manager |
|
||||
| `timerfd_create` / `timerfd_settime` | wo-event | Periodic tasks (compaction, keepalive) as file descriptors |
|
||||
| `signalfd` | wo-event | SIGINT/SIGTERM delivered as fd events for graceful shutdown |
|
||||
| `inotify_init1` / `inotify_add_watch` | wo-watch | File system change detection on the content directory |
|
||||
| `pipe2` | wo-sub (tests) | Mock subscriber fds for testing notification delivery |
|
||||
|
||||
## .seg File Format
|
||||
|
||||
```
|
||||
Offset Size Field
|
||||
0 4 Magic: b"WOSF"
|
||||
4 2 Version: u16 LE (1)
|
||||
6 2 Flags: u16 LE (reserved)
|
||||
8 8 Record count: u64 LE
|
||||
16 8 Data start offset: u64 LE
|
||||
24 8 Reserved
|
||||
32+ variable Records: [u32 length][u8 flags][bincode payload]...
|
||||
```
|
||||
|
||||
- Records are addressed by byte offset from file start
|
||||
- Flags: `0x00` = active, `0x01` = tombstoned
|
||||
- Payload: bincode-serialized `Article` struct
|
||||
|
||||
## Index Files
|
||||
|
||||
| File | Format | Access Pattern |
|
||||
|------|--------|----------------|
|
||||
| `title.idx` | On-disk hash table (Robin Hood, load factor 0.5), 138 bytes/slot | O(1) lookup by `sys_title` |
|
||||
| `date.idx` | Sorted `(i64 timestamp, u64 offset)` array, 16 bytes/entry | Binary search for date ranges, latest N |
|
||||
| `tags.idx` | Bincode-serialized `HashMap<String, Vec<u64>>` | Tag-to-offsets inverted index |
|
||||
|
||||
All indexes are derived from `.seg` and rebuildable from `content/` on cold start.
|
||||
|
||||
## Subscription Model
|
||||
|
||||
No SSE. No WebSocket. Notifications are written directly to subscriber file descriptors.
|
||||
|
||||
- **Subscribe**: `SubscriptionManager::subscribe(fd, Subscription::ByTitle("linux-misc"))`
|
||||
- **Notify**: on content change, length-prefixed `Notification` written to matching fds
|
||||
- **Cleanup**: `EPOLLHUP` on epoll triggers automatic `unsubscribe(fd)`
|
||||
- **Dedup**: if a fd matches multiple patterns (title + tag), it receives only one notification
|
||||
|
||||
Subscription patterns:
|
||||
- `Subscription::ByTitle(sys_title)` — single article
|
||||
- `Subscription::ByTag(tag)` — all articles with tag
|
||||
- `Subscription::All` — all content changes
|
||||
|
||||
## Store Query API
|
||||
|
||||
```rust
|
||||
store.get_by_title("linux-misc") -> Option<Article>
|
||||
store.list_published(skip, limit) -> Vec<Article>
|
||||
store.list_by_tag("rust") -> Vec<Article>
|
||||
store.list_by_date_range(start, end) -> Vec<Article>
|
||||
store.count_published() -> usize
|
||||
store.article_version("linux-misc") -> Option<u64>
|
||||
store.rebuild() // full rebuild from content/
|
||||
```
|
||||
|
||||
## Runtime Event Loop
|
||||
|
||||
Single `epoll` instance multiplexing all file descriptors:
|
||||
|
||||
| Token | Fd | Handler |
|
||||
|-------|----|---------|
|
||||
| `WATCHER` | inotify fd | Process file changes → update store → notify subscribers |
|
||||
| `SIGNAL` | signalfd | SIGINT/SIGTERM → graceful shutdown |
|
||||
| `TIMER` | timerfd | Periodic tasks (compaction, stats) |
|
||||
| `NOTIFY` | eventfd | Subscription notification signal |
|
||||
| `1000+` | subscriber fds | Hangup detection → unsubscribe + cleanup |
|
||||
|
||||
## External Dependencies
|
||||
|
||||
| Crate | Version | Purpose |
|
||||
|-------|---------|---------|
|
||||
| `serde` | 1.x | Serialization derives |
|
||||
| `serde_json` | 1.x | JSON parsing for article files |
|
||||
| `bincode` | 1.x | Compact binary serialization for .seg records and notifications |
|
||||
| `libc` | 0.2.x | Raw Linux syscall bindings |
|
||||
|
||||
No tokio. No async-std. No database driver. No HTTP framework (yet).
|
||||
|
||||
## What Comes Next
|
||||
|
||||
The data layer delivers everything the HTTP server and UI layers need:
|
||||
|
||||
1. **`Store` with zero-copy query access** — all article queries resolve in-process
|
||||
2. **Subscription system accepting raw fds** — HTTP layer hands socket fds to `subscribe()`
|
||||
3. **Shared event loop** — HTTP listener socket registers on the same epoll
|
||||
4. **Automatic cold-start** — if `data/` is missing, rebuilds from `content/` on startup
|
||||
5. **Graceful shutdown** — SIGTERM triggers clean fd cleanup
|
||||
|
||||
Next phases per [02-recovery.md](./02-recovery.md):
|
||||
- **HTTP server** — route handlers using the `Store` query API, embedded in the same binary
|
||||
- **HTMLX templates** — server-rendered HTML with `{{bindings}}` per [04-ui.md](./04-ui.md)
|
||||
- **Frontend collapse** — serve static assets from the binary, eliminate the Angular app
|
||||
3
|
||||
246
docs/06-markdown-render.md
Normal file
246
docs/06-markdown-render.md
Normal file
|
|
@ -0,0 +1,246 @@
|
|||
# Markdown File Rendering
|
||||
|
||||
## Current State (writeonce-articles-s3)
|
||||
|
||||
Each article is a directory containing a JSON metadata file and one or more `.md` files:
|
||||
|
||||
```
|
||||
auto-scale-gitlab-runner-using-aws-spot-instance/
|
||||
docker-machine-test-with-t2.md
|
||||
gitlab-runner-config.md
|
||||
stop-test-gitlab-docker-machine.md
|
||||
|
||||
gitlab-runner-with-kubernetes-executor/
|
||||
gitlab-runner-with-kubernetes-executor.json
|
||||
deploy.md
|
||||
permission.md
|
||||
role-binding.md
|
||||
role-defination.md
|
||||
gitlab-runnergitlab-runner-deploy.md
|
||||
```
|
||||
|
||||
The JSON metadata currently defines the full article structure — sections, headings, paragraphs, and code snippet references. Markdown files are limited to code blocks referenced via the `codes[].snippet` field.
|
||||
|
||||
## Problem
|
||||
|
||||
The JSON metadata carries too much content. Headings, paragraphs, prose — all of this is duplicated as JSON strings inside `content.content.sections`. The markdown files only hold code snippets, referenced by `sectionIndex` and `paragraphIndex`.
|
||||
|
||||
This is backwards. The markdown file should be the content. The JSON should be minimal metadata.
|
||||
|
||||
## Target: Markdown-First Content Model
|
||||
|
||||
**The markdown file is the article.** All prose, headings, code blocks, and inline formatting live in the `.md` file. The JSON metadata file holds only what markdown cannot express: system fields, tags, publication state, and author.
|
||||
|
||||
### Minimal JSON Metadata
|
||||
|
||||
```json
|
||||
{
|
||||
"sys_title": "gitlab-runner-with-kubernetes-executor",
|
||||
"title": "Gitlab Runner with Kubernetes Executor",
|
||||
"published": true,
|
||||
"author": "Shoney Arickathil",
|
||||
"tags": ["kubernetes", "gitlab", "ci-cd"],
|
||||
"published_on": 1740950884
|
||||
}
|
||||
```
|
||||
|
||||
No `content.content.sections`. No `content.content.codes`. No `paragraphs[]` arrays. No `sectionIndex`/`paragraphIndex` mapping.
|
||||
|
||||
### Markdown File = Full Article Content
|
||||
|
||||
````markdown
|
||||
# Introduction
|
||||
|
||||
Deploying a Gitlab runner using kubernetes is a great option to overcome
|
||||
the limitations of other gitlab runner executor such as docker and docker machine.
|
||||
|
||||
## Running Gitlab Runner in gitlab namespace
|
||||
|
||||
Create the namespace and apply the deployment:
|
||||
|
||||
```yaml
|
||||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
metadata:
|
||||
name: gitlab-runner
|
||||
namespace: gitlab
|
||||
```
|
||||
````
|
||||
|
||||
## Permissions
|
||||
|
||||
The runner needs RBAC permissions to create pods:
|
||||
|
||||
```yaml
|
||||
apiVersion: rbac.authorization.k8s.io/v1
|
||||
kind: Role
|
||||
metadata:
|
||||
name: gitlab-runner
|
||||
```
|
||||
|
||||
Everything is in the markdown — headings, paragraphs, code blocks with language hints, links, images. The rendering pipeline parses the markdown directly.
|
||||
|
||||
### Directory Structure
|
||||
|
||||
```
|
||||
content/
|
||||
gitlab-runner-with-kubernetes-executor/
|
||||
gitlab-runner-with-kubernetes-executor.json # minimal metadata
|
||||
gitlab-runner-with-kubernetes-executor.md # full article content
|
||||
linux-misc/
|
||||
linux-misc.json
|
||||
linux-misc.md
|
||||
```
|
||||
|
||||
One JSON for metadata. One markdown for content. No scattered `.md` files per code snippet.
|
||||
|
||||
## What Changes
|
||||
|
||||
| Before | After |
|
||||
| -------------------------------------------------------------- | ----------------------------------------------------------------------- |
|
||||
| JSON holds sections, headings, paragraphs as structured arrays | JSON holds only sys_title, title, published, author, tags, published_on |
|
||||
| Markdown files hold only code snippets | Markdown file holds the entire article |
|
||||
| `codes[].snippet` maps filename to sectionIndex/paragraphIndex | No mapping needed — headings and code blocks are inline in markdown |
|
||||
| Renderer reads JSON structure, injects code from .md files | Renderer parses markdown directly into HTML |
|
||||
| Multiple .md files per article (one per code snippet) | One .md file per article |
|
||||
|
||||
## Impact on the Data Layer
|
||||
|
||||
### wo-model
|
||||
|
||||
The `Article` struct simplifies:
|
||||
|
||||
```rust
|
||||
pub struct Article {
|
||||
pub sys_title: String,
|
||||
pub title: String,
|
||||
pub published: bool,
|
||||
pub author: String,
|
||||
pub tags: Vec<String>,
|
||||
pub published_on: Option<i64>,
|
||||
}
|
||||
```
|
||||
|
||||
The nested `ArticleContent` / `ArticleBody` / `Section` / `CodeSnippet` hierarchy is no longer needed. Article content comes from parsing the `.md` file at render time, not from the JSON.
|
||||
|
||||
### wo-md
|
||||
|
||||
Currently handles only inline markdown (`**bold**`, `` `code` ``, links). Needs to become a full markdown-to-HTML renderer:
|
||||
|
||||
- Block elements: headings (`#`, `##`), paragraphs, code fences (` `lang ```), lists, blockquotes
|
||||
- Inline elements: bold, italic, code, links, images
|
||||
- Code fence language extraction for `wo-md::highlight()`
|
||||
- The renderer reads `{sys_title}/{sys_title}.md`, parses it, and returns HTML
|
||||
|
||||
### wo-htmlx
|
||||
|
||||
The `article.htmlx` template simplifies. Instead of iterating `{{#each article.content.content.sections}}`, it renders the pre-parsed markdown HTML:
|
||||
|
||||
```html
|
||||
<article>
|
||||
<h1>{{article.title}}</h1>
|
||||
<p class="meta">by {{article.author}} · {{article.tags}}</p>
|
||||
{{article.content_html}}
|
||||
</article>
|
||||
```
|
||||
|
||||
Where `content_html` is the full HTML output from the markdown renderer.
|
||||
|
||||
### wo-store
|
||||
|
||||
`ContentLoader` reads the `.json` for metadata and the `.md` for content. The `.seg` file stores both. At query time, the markdown is either:
|
||||
|
||||
- Pre-rendered to HTML during ingestion (stored in .seg alongside metadata)
|
||||
- Rendered on-demand at request time (read .md from disk)
|
||||
|
||||
Pre-rendering is preferred — it avoids parsing markdown on every HTTP request.
|
||||
|
||||
## Migration Path
|
||||
|
||||
1. Update `wo-model` with the simplified `Article` struct
|
||||
2. Extend `wo-md` to handle full markdown (block-level parsing, code fences)
|
||||
3. Update `ContentLoader` to read `.json` + `.md` pairs
|
||||
4. Update `wo-store` to store pre-rendered HTML in the .seg file
|
||||
5. Simplify `article.htmlx` template
|
||||
6. Migrate existing articles: extract prose from JSON into `.md` files
|
||||
|
||||
Existing articles with the old JSON format can coexist during migration — `ContentLoader` checks for a `.md` file and falls back to the JSON structure if none exists.
|
||||
|
||||
## Blog Subscription — Live Content Reload
|
||||
|
||||
When a user visits `http://localhost:3000/blog/sample-rust-patterns`, the content should stay live. Any edit to `sample-content/sample-rust-patterns/sample-rust-patterns.md` must auto-reflect in the browser without a page refresh.
|
||||
|
||||
### How It Works
|
||||
|
||||
```
|
||||
Browser visits /blog/sample-rust-patterns
|
||||
│
|
||||
▼
|
||||
1. Server renders article HTML from .seg (pre-rendered from .md)
|
||||
2. Server writes HTML response to socket fd
|
||||
3. Server registers socket fd in subscription table:
|
||||
register!(sub_manager, socket_fd, ByTitle("sample-rust-patterns"))
|
||||
4. Connection transitions to Subscribed state (stays open)
|
||||
│
|
||||
│ (user edits sample-rust-patterns.md)
|
||||
│
|
||||
▼
|
||||
5. inotify fires IN_MODIFY on sample-rust-patterns.md
|
||||
6. ContentWatcher maps file → sys_title "sample-rust-patterns"
|
||||
7. Store rebuilds: re-reads .json + .md, re-renders markdown to HTML, updates .seg + indexes
|
||||
8. SubscriptionManager::notify("sample-rust-patterns", ...) fires
|
||||
9. For each subscribed fd: write(fd, diff_payload)
|
||||
│
|
||||
▼
|
||||
10. Browser receives payload on the open connection
|
||||
11. Client-side script applies the update to the DOM
|
||||
```
|
||||
|
||||
### What Needs to Work
|
||||
|
||||
| Component | Requirement |
|
||||
|-----------|-------------|
|
||||
| **inotify** (wo-watch) | Already watches `content/` directory. `.md` file changes must trigger `ContentChange::Modified(sys_title)` |
|
||||
| **Store rebuild** (wo-store) | On `.md` change: re-read file, re-render markdown to HTML, update `.seg` and indexes |
|
||||
| **Subscription table** (wo-sub) | Route handler registers the browser's socket fd via `register!` after sending initial HTML |
|
||||
| **Notification** (wo-sub) | On content change, write updated `content_html` to all subscribed fds as JSON payload |
|
||||
| **Event loop** (wo-rt) | After writing initial response, transition connection to `Subscribed` state. Keep fd on epoll for hangup detection. |
|
||||
| **Client script** | Injected in the HTML. Reads payloads from the open connection. Replaces article content in the DOM. |
|
||||
|
||||
### Client-Side Script
|
||||
|
||||
Injected by the template renderer into every article page:
|
||||
|
||||
```html
|
||||
<script>
|
||||
// Connection stays open after initial HTML.
|
||||
// Server writes length-prefixed JSON payloads when content changes.
|
||||
const decoder = new TextDecoder();
|
||||
const articleEl = document.querySelector('article');
|
||||
|
||||
fetch(window.location.href, { headers: { 'X-Subscribe': '1' } })
|
||||
.then(r => r.body.getReader())
|
||||
.then(reader => {
|
||||
(function read() {
|
||||
reader.read().then(({ done, value }) => {
|
||||
if (done) return;
|
||||
try {
|
||||
const payload = JSON.parse(decoder.decode(value));
|
||||
if (payload.content_html) {
|
||||
articleEl.innerHTML = payload.content_html;
|
||||
}
|
||||
} catch (e) {}
|
||||
read();
|
||||
});
|
||||
})();
|
||||
});
|
||||
</script>
|
||||
```
|
||||
|
||||
### inotify and .md Files
|
||||
|
||||
The current `ContentWatcher` watches for `.json` file changes. It must also trigger on `.md` file changes:
|
||||
|
||||
- `IN_MODIFY` on `*.md` → `ContentChange::Modified(sys_title)`
|
||||
- The sys_title is derived from the parent directory name (same as for JSON)
|
||||
- Both `.json` and `.md` changes trigger a store rebuild and subscriber notification
|
||||
252
docs/07-ssl.md
Normal file
252
docs/07-ssl.md
Normal file
|
|
@ -0,0 +1,252 @@
|
|||
# SSL and Deployment
|
||||
|
||||
## Problem
|
||||
|
||||
In [01-problem.md](./01-problem.md), the infrastructure overhead was identified — multiple repos, AWS dependencies, separate deployment pipelines. But one problem went unaddressed: the server-side infrastructure that sits in front of the application — nginx reverse proxy, SSL certificates, systemd service management, and deployment to the production host.
|
||||
|
||||
Currently this requires manual SSH, manual nginx config, manual certbot runs. For a single-binary platform, the deployment should be as simple as the architecture.
|
||||
|
||||
## Target
|
||||
|
||||
Given:
|
||||
- SSH access to `writeonce.de` is configured
|
||||
- nginx exists at the default path `/etc/nginx/`
|
||||
- The writeonce binary listens on a local port (e.g., `127.0.0.1:3000`)
|
||||
|
||||
The deployment pipeline should:
|
||||
1. Build the binary
|
||||
2. Copy it to the server
|
||||
3. Create/update the systemd service
|
||||
4. Restart the service
|
||||
5. Configure nginx as a reverse proxy
|
||||
6. Obtain and auto-renew SSL certificates via Let's Encrypt
|
||||
|
||||
## Systemd Service
|
||||
|
||||
The writeonce binary runs as a systemd service for automatic restart, logging, and boot-start.
|
||||
|
||||
### Service File
|
||||
|
||||
```ini
|
||||
# /etc/systemd/system/writeonce.service
|
||||
[Unit]
|
||||
Description=writeonce content platform
|
||||
After=network.target
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
User=writeonce
|
||||
Group=writeonce
|
||||
WorkingDirectory=/opt/writeonce
|
||||
ExecStart=/opt/writeonce/writeonce
|
||||
Restart=on-failure
|
||||
RestartSec=5
|
||||
StandardOutput=journal
|
||||
StandardError=journal
|
||||
|
||||
# Security hardening
|
||||
NoNewPrivileges=true
|
||||
ProtectSystem=strict
|
||||
ProtectHome=true
|
||||
ReadWritePaths=/opt/writeonce/data
|
||||
PrivateTmp=true
|
||||
|
||||
[Install]
|
||||
WantedBy=multi-user.target
|
||||
```
|
||||
|
||||
### Directory Layout on Server
|
||||
|
||||
```
|
||||
/opt/writeonce/
|
||||
writeonce # the binary
|
||||
content/ # article .json + .md files
|
||||
data/ # derived .seg + .idx (rebuilt on start)
|
||||
templates/ # .htmlx templates
|
||||
static/ # CSS, images
|
||||
```
|
||||
|
||||
### Service Management
|
||||
|
||||
```bash
|
||||
# Install / update
|
||||
sudo systemctl daemon-reload
|
||||
sudo systemctl enable writeonce
|
||||
sudo systemctl restart writeonce
|
||||
|
||||
# Check status
|
||||
sudo systemctl status writeonce
|
||||
journalctl -u writeonce -f
|
||||
```
|
||||
|
||||
## Nginx Reverse Proxy
|
||||
|
||||
Nginx sits in front of the writeonce binary, handling SSL termination and proxying requests to `127.0.0.1:3000`.
|
||||
|
||||
### Nginx Config
|
||||
|
||||
```nginx
|
||||
# /etc/nginx/sites-available/writeonce.de
|
||||
server {
|
||||
listen 80;
|
||||
server_name writeonce.de www.writeonce.de;
|
||||
return 301 https://$server_name$request_uri;
|
||||
}
|
||||
|
||||
server {
|
||||
listen 443 ssl http2;
|
||||
server_name writeonce.de www.writeonce.de;
|
||||
|
||||
ssl_certificate /etc/letsencrypt/live/writeonce.de/fullchain.pem;
|
||||
ssl_certificate_key /etc/letsencrypt/live/writeonce.de/privkey.pem;
|
||||
ssl_protocols TLSv1.2 TLSv1.3;
|
||||
ssl_ciphers HIGH:!aNULL:!MD5;
|
||||
ssl_prefer_server_ciphers on;
|
||||
|
||||
# HSTS
|
||||
add_header Strict-Transport-Security "max-age=31536000; includeSubDomains" always;
|
||||
|
||||
location / {
|
||||
proxy_pass http://127.0.0.1:3000;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $scheme;
|
||||
|
||||
# Keep connections open for database subscriptions
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Connection "";
|
||||
proxy_read_timeout 86400s;
|
||||
proxy_send_timeout 86400s;
|
||||
}
|
||||
|
||||
# Static assets — let nginx serve directly for better caching
|
||||
location /static/ {
|
||||
alias /opt/writeonce/static/;
|
||||
expires 1y;
|
||||
add_header Cache-Control "public, immutable";
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### Enable Site
|
||||
|
||||
```bash
|
||||
sudo ln -sf /etc/nginx/sites-available/writeonce.de /etc/nginx/sites-enabled/
|
||||
sudo nginx -t
|
||||
sudo systemctl reload nginx
|
||||
```
|
||||
|
||||
## SSL with Let's Encrypt
|
||||
|
||||
### Initial Certificate
|
||||
|
||||
```bash
|
||||
sudo apt install certbot python3-certbot-nginx
|
||||
sudo certbot --nginx -d writeonce.de -d www.writeonce.de
|
||||
```
|
||||
|
||||
Certbot modifies the nginx config to add SSL directives and obtains the certificate.
|
||||
|
||||
### Auto-Renewal
|
||||
|
||||
Certbot installs a systemd timer that runs twice daily:
|
||||
|
||||
```bash
|
||||
# Check timer
|
||||
systemctl list-timers | grep certbot
|
||||
|
||||
# Manual test
|
||||
sudo certbot renew --dry-run
|
||||
```
|
||||
|
||||
Certificates auto-renew before expiry. Nginx reloads automatically via certbot's deploy hook.
|
||||
|
||||
### Deploy Hook for Nginx Reload
|
||||
|
||||
```bash
|
||||
# /etc/letsencrypt/renewal-hooks/deploy/reload-nginx.sh
|
||||
#!/bin/bash
|
||||
systemctl reload nginx
|
||||
```
|
||||
|
||||
## Deployment Script
|
||||
|
||||
A single script that builds, copies, and restarts:
|
||||
|
||||
```bash
|
||||
#!/bin/bash
|
||||
# deploy.sh — run from the development machine
|
||||
set -e
|
||||
|
||||
SERVER="writeonce.de"
|
||||
REMOTE_DIR="/opt/writeonce"
|
||||
|
||||
echo "Building release binary..."
|
||||
cargo build --release -p wo-rt --bin writeonce
|
||||
|
||||
echo "Copying binary to server..."
|
||||
scp target/release/writeonce $SERVER:$REMOTE_DIR/writeonce.new
|
||||
|
||||
echo "Syncing content and templates..."
|
||||
rsync -az --delete content/ $SERVER:$REMOTE_DIR/content/
|
||||
rsync -az --delete templates/ $SERVER:$REMOTE_DIR/templates/
|
||||
rsync -az --delete static/ $SERVER:$REMOTE_DIR/static/
|
||||
|
||||
echo "Swapping binary and restarting..."
|
||||
ssh $SERVER "
|
||||
sudo mv $REMOTE_DIR/writeonce.new $REMOTE_DIR/writeonce
|
||||
sudo systemctl restart writeonce
|
||||
"
|
||||
|
||||
echo "Deployed. Checking status..."
|
||||
ssh $SERVER "sudo systemctl status writeonce --no-pager"
|
||||
```
|
||||
|
||||
### First-Time Setup
|
||||
|
||||
Run once on the server to create the user, directory, and service:
|
||||
|
||||
```bash
|
||||
#!/bin/bash
|
||||
# setup.sh — run on the server
|
||||
set -e
|
||||
|
||||
# Create user
|
||||
sudo useradd -r -s /bin/false writeonce
|
||||
|
||||
# Create directory
|
||||
sudo mkdir -p /opt/writeonce/{content,data,templates,static}
|
||||
sudo chown -R writeonce:writeonce /opt/writeonce
|
||||
|
||||
# Install service
|
||||
sudo cp writeonce.service /etc/systemd/system/
|
||||
sudo systemctl daemon-reload
|
||||
sudo systemctl enable writeonce
|
||||
|
||||
# Configure nginx
|
||||
sudo cp writeonce.de.nginx /etc/nginx/sites-available/writeonce.de
|
||||
sudo ln -sf /etc/nginx/sites-available/writeonce.de /etc/nginx/sites-enabled/
|
||||
sudo nginx -t
|
||||
sudo systemctl reload nginx
|
||||
|
||||
# SSL
|
||||
sudo certbot --nginx -d writeonce.de -d www.writeonce.de
|
||||
```
|
||||
|
||||
## What This Replaces
|
||||
|
||||
| Before | After |
|
||||
|--------|-------|
|
||||
| Pulumi IaC managing Lambda + S3 | `deploy.sh` with scp + rsync |
|
||||
| AWS Lambda deployment pipeline | `systemctl restart writeonce` |
|
||||
| S3 bucket for content hosting | `rsync content/` to server |
|
||||
| Docker Compose for API + DB | Single binary, one systemd service |
|
||||
| Multiple nginx configs for API + frontend | One nginx config, one proxy_pass |
|
||||
| Manual SSL setup | `certbot --nginx` with auto-renewal |
|
||||
|
||||
## Connection Keepalive for Subscriptions
|
||||
|
||||
The nginx config sets `proxy_read_timeout 86400s` (24 hours) to keep persistent connections open for the database subscription model. When a browser visits an article page and the connection transitions to `Subscribed` state, nginx must not timeout and close the upstream connection.
|
||||
|
||||
If nginx is removed in the future (the binary handles TLS directly via `rustls`), this concern disappears — the binary owns the socket end-to-end.
|
||||
175
docs/future-scope/ai-agents-content-management.md
Normal file
175
docs/future-scope/ai-agents-content-management.md
Normal file
|
|
@ -0,0 +1,175 @@
|
|||
# AI Agents and Content Management
|
||||
|
||||
## Context
|
||||
|
||||
AI agents (Claude Code, Copilot, Cursor, custom agents) work within a specific project or working directory. Their sessions, context, and understanding are scoped to that directory. This works well when projects are completely different domains.
|
||||
|
||||
But writeonce content is not isolated — articles reference each other, share tags, build on concepts from other articles. An agent editing `gitlab-runner-with-kubernetes-executor.md` would benefit from knowing that `auto-scale-gitlab-runner-using-aws-spot-instance.md` exists and covers related infrastructure. Without explicit mappings, the agent treats each article as an island.
|
||||
|
||||
## Problem
|
||||
|
||||
1. **Agents lack cross-article awareness.** When asked to write or update an article about Kubernetes, the agent doesn't know that related articles about Docker, CI/CD, or AWS already exist in the content directory — unless it manually searches.
|
||||
|
||||
2. **No semantic grouping.** Tags provide flat categorization (`kubernetes`, `ci-cd`), but they don't express relationships: "this article is a prerequisite for that one", "these three articles form a series", "this article supersedes that one."
|
||||
|
||||
3. **Context window waste.** Without mappings, the agent must scan all articles to find related content. With explicit mappings, it can load exactly the relevant files.
|
||||
|
||||
## Solution: Metadata-Driven Content Mappings
|
||||
|
||||
Users define relationships between articles in the JSON metadata. These mappings serve two purposes:
|
||||
|
||||
1. **Human navigation** — rendered as "related articles" links on the site
|
||||
2. **Agent context** — when an agent works on an article, it loads the mapped articles into its context for cross-referencing
|
||||
|
||||
### Mapping Fields in JSON Metadata
|
||||
|
||||
Per [06-markdown-render.md](./06-markdown-render.md), the JSON metadata is minimal. Add a `mappings` field:
|
||||
|
||||
```json
|
||||
{
|
||||
"sys_title": "gitlab-runner-with-kubernetes-executor",
|
||||
"title": "Gitlab Runner with Kubernetes Executor",
|
||||
"published": true,
|
||||
"author": "Shoney Arickathil",
|
||||
"tags": ["kubernetes", "gitlab", "ci-cd"],
|
||||
"published_on": 1740950884,
|
||||
"mappings": {
|
||||
"related": ["auto-scale-gitlab-runner-using-aws-spot-instance"],
|
||||
"prerequisite": ["linux-misc"],
|
||||
"series": {
|
||||
"name": "gitlab-runner",
|
||||
"order": 2
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### Mapping Types
|
||||
|
||||
| Type | Meaning | Agent Use |
|
||||
|------|---------|-----------|
|
||||
| `related` | Topically related articles | Agent loads these for cross-reference when editing |
|
||||
| `prerequisite` | Articles the reader should read first | Agent ensures no concept duplication, references prerequisites instead of re-explaining |
|
||||
| `series` | Articles that form an ordered sequence | Agent maintains narrative continuity across the series |
|
||||
| `supersedes` | This article replaces an older one | Agent can mark the old article as outdated or unpublished |
|
||||
| `references` | External articles or URLs the content builds on | Agent checks links are still valid, cites them properly |
|
||||
|
||||
### Directory Structure with Mappings
|
||||
|
||||
```
|
||||
content/
|
||||
gitlab-runner-with-kubernetes-executor/
|
||||
gitlab-runner-with-kubernetes-executor.json # metadata + mappings
|
||||
gitlab-runner-with-kubernetes-executor.md # full article
|
||||
auto-scale-gitlab-runner-using-aws-spot-instance/
|
||||
auto-scale-gitlab-runner-using-aws-spot-instance.json
|
||||
auto-scale-gitlab-runner-using-aws-spot-instance.md
|
||||
linux-misc/
|
||||
linux-misc.json
|
||||
linux-misc.md
|
||||
```
|
||||
|
||||
## Agent Workflows
|
||||
|
||||
### 1. Writing a New Article
|
||||
|
||||
The author asks an agent: "Write an article about deploying GitLab Runner on ECS."
|
||||
|
||||
The agent:
|
||||
1. Scans the content directory for existing articles with tags `gitlab`, `ci-cd`, `aws`
|
||||
2. Finds `gitlab-runner-with-kubernetes-executor` and `auto-scale-gitlab-runner-using-aws-spot-instance`
|
||||
3. Reads their `.md` files to understand what's already covered
|
||||
4. Writes the new article, referencing existing articles rather than re-explaining shared concepts
|
||||
5. Suggests `mappings.related` entries for the new article's JSON
|
||||
|
||||
### 2. Updating an Existing Article
|
||||
|
||||
The author asks: "Update the Kubernetes executor article with the new runner token format."
|
||||
|
||||
The agent:
|
||||
1. Reads the article's JSON metadata and `.md` content
|
||||
2. Reads the `mappings.related` articles to check for consistency
|
||||
3. Makes the update in the `.md` file
|
||||
4. Checks if the change affects any prerequisite or series articles
|
||||
5. inotify detects the `.md` change → store rebuilds → subscribers notified
|
||||
|
||||
### 3. Content Audit
|
||||
|
||||
The author asks: "Which articles reference outdated AWS configurations?"
|
||||
|
||||
The agent:
|
||||
1. Loads all article metadata (the Store already indexes everything)
|
||||
2. Follows `mappings` to build a dependency graph
|
||||
3. Reads the `.md` files of articles tagged with `aws`
|
||||
4. Identifies outdated patterns (old SDK versions, deprecated services)
|
||||
5. Reports findings with links to specific articles and line numbers
|
||||
|
||||
### 4. Series Management
|
||||
|
||||
The author asks: "Add a new part to the gitlab-runner series."
|
||||
|
||||
The agent:
|
||||
1. Finds all articles with `mappings.series.name == "gitlab-runner"`
|
||||
2. Reads them in order to understand the narrative arc
|
||||
3. Writes the new article continuing from where the series left off
|
||||
4. Sets `mappings.series.order` to the next number
|
||||
5. Updates the previous article's mappings to reference the new one
|
||||
|
||||
## Integration with writeonce Architecture
|
||||
|
||||
### Store Index
|
||||
|
||||
Add a mappings index alongside the existing title, date, and tag indexes:
|
||||
|
||||
```
|
||||
data/
|
||||
articles.seg
|
||||
index/
|
||||
title.idx
|
||||
date.idx
|
||||
tags.idx
|
||||
mappings.idx # sys_title → related sys_titles
|
||||
```
|
||||
|
||||
The mappings index allows efficient traversal: "give me all articles related to X" without scanning every article's JSON.
|
||||
|
||||
### Template Rendering
|
||||
|
||||
The `article.htmlx` template can render related articles:
|
||||
|
||||
```html
|
||||
<article>
|
||||
<h1>{{article.title}}</h1>
|
||||
{{article.content_html}}
|
||||
|
||||
{{#each article.related}}
|
||||
<aside class="related">
|
||||
<h3>Related</h3>
|
||||
<ul>
|
||||
<li><a href="/blog/{{sys_title}}">{{title}}</a></li>
|
||||
</ul>
|
||||
</aside>
|
||||
{{/each}}
|
||||
</article>
|
||||
```
|
||||
|
||||
### Subscription
|
||||
|
||||
When a mapped article changes, subscribers to related articles can optionally be notified. If article A lists article B in `mappings.related`, and article B is updated, subscribers to article A can receive a notification that related content changed.
|
||||
|
||||
## Agent Configuration
|
||||
|
||||
For agents to use the mappings effectively, the project can include an agent instruction file (e.g., `CLAUDE.md` or `.agent/instructions.md`):
|
||||
|
||||
```markdown
|
||||
## Content Management
|
||||
|
||||
- Articles are in `content/{sys_title}/{sys_title}.md`
|
||||
- Metadata is in `content/{sys_title}/{sys_title}.json`
|
||||
- Before writing or editing an article, read its `mappings` field and load related articles for context
|
||||
- When creating a new article, suggest appropriate `mappings` based on tags and content overlap
|
||||
- Maintain narrative continuity within `series` mappings
|
||||
- Do not duplicate explanations that exist in `prerequisite` articles — reference them instead
|
||||
```
|
||||
|
||||
This turns the content directory into an agent-navigable knowledge graph where the metadata provides the edges and the markdown files provide the nodes.
|
||||
338
docs/runtime/async.md
Normal file
338
docs/runtime/async.md
Normal file
|
|
@ -0,0 +1,338 @@
|
|||
# Creating an Async Runtime Environment
|
||||
|
||||
## Goal
|
||||
|
||||
Build a minimal async runtime from scratch using Linux kernel primitives. The runtime is not specific to writeonce — it is a general-purpose event loop that any project can use to multiplex I/O without threads, without tokio, and without any external async framework.
|
||||
|
||||
The writeonce project uses this runtime (the `wo-event` and `wo-rt` crates), but the concepts apply to any server, daemon, or event-driven application on Linux.
|
||||
|
||||
## What Is a Runtime?
|
||||
|
||||
A runtime is the loop that decides **what code runs next**. In a synchronous program, the OS scheduler picks the next thread. In an async runtime, a single thread asks the kernel: "which of my file descriptors are ready?" — and runs the corresponding handler.
|
||||
|
||||
```
|
||||
loop {
|
||||
ready_fds = ask_kernel_which_fds_are_ready()
|
||||
for fd in ready_fds {
|
||||
run_handler(fd)
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
That's the entire concept. Everything else — epoll, tokens, interest flags — is implementation detail around this loop.
|
||||
|
||||
## Explaining It with C
|
||||
|
||||
Before Rust, before abstractions, here is a minimal async runtime in C that watches two file descriptors on one thread.
|
||||
|
||||
### Step 1: Create an epoll instance
|
||||
|
||||
```c
|
||||
#include <sys/epoll.h>
|
||||
#include <unistd.h>
|
||||
#include <stdio.h>
|
||||
|
||||
int main() {
|
||||
// Create the event loop
|
||||
int epoll_fd = epoll_create1(0);
|
||||
|
||||
// This single fd is the "runtime" — all other fds register on it
|
||||
printf("epoll fd: %d\n", epoll_fd);
|
||||
}
|
||||
```
|
||||
|
||||
`epoll_create1` returns a file descriptor. This fd *is* the runtime. Every other fd in the system registers itself on this one fd, and the kernel tracks readiness for all of them.
|
||||
|
||||
### Step 2: Register file descriptors
|
||||
|
||||
```c
|
||||
#include <sys/epoll.h>
|
||||
#include <sys/eventfd.h>
|
||||
#include <sys/timerfd.h>
|
||||
#include <time.h>
|
||||
|
||||
int main() {
|
||||
int epoll_fd = epoll_create1(0);
|
||||
|
||||
// Create an eventfd (like a semaphore as a file descriptor)
|
||||
int event_fd = eventfd(0, EFD_NONBLOCK);
|
||||
|
||||
// Create a timerfd (fires every 2 seconds)
|
||||
int timer_fd = timerfd_create(CLOCK_MONOTONIC, TFD_NONBLOCK);
|
||||
struct itimerspec spec = {
|
||||
.it_interval = { .tv_sec = 2, .tv_nsec = 0 },
|
||||
.it_value = { .tv_sec = 2, .tv_nsec = 0 }
|
||||
};
|
||||
timerfd_settime(timer_fd, 0, &spec, NULL);
|
||||
|
||||
// Register both on epoll
|
||||
struct epoll_event ev1 = { .events = EPOLLIN, .data.fd = event_fd };
|
||||
epoll_ctl(epoll_fd, EPOLL_CTL_ADD, event_fd, &ev1);
|
||||
|
||||
struct epoll_event ev2 = { .events = EPOLLIN, .data.fd = timer_fd };
|
||||
epoll_ctl(epoll_fd, EPOLL_CTL_ADD, timer_fd, &ev2);
|
||||
}
|
||||
```
|
||||
|
||||
Two different kinds of fd — an event signal and a timer — both registered on the same epoll instance. The kernel will wake us when either is ready.
|
||||
|
||||
### Step 3: The event loop
|
||||
|
||||
```c
|
||||
#include <sys/epoll.h>
|
||||
#include <sys/eventfd.h>
|
||||
#include <sys/timerfd.h>
|
||||
#include <unistd.h>
|
||||
#include <stdio.h>
|
||||
#include <stdint.h>
|
||||
#include <time.h>
|
||||
|
||||
int main() {
|
||||
int epoll_fd = epoll_create1(0);
|
||||
|
||||
int event_fd = eventfd(0, EFD_NONBLOCK);
|
||||
int timer_fd = timerfd_create(CLOCK_MONOTONIC, TFD_NONBLOCK);
|
||||
|
||||
struct itimerspec spec = {
|
||||
.it_interval = { .tv_sec = 2, .tv_nsec = 0 },
|
||||
.it_value = { .tv_sec = 2, .tv_nsec = 0 }
|
||||
};
|
||||
timerfd_settime(timer_fd, 0, &spec, NULL);
|
||||
|
||||
struct epoll_event ev1 = { .events = EPOLLIN, .data.fd = event_fd };
|
||||
epoll_ctl(epoll_fd, EPOLL_CTL_ADD, event_fd, &ev1);
|
||||
|
||||
struct epoll_event ev2 = { .events = EPOLLIN, .data.fd = timer_fd };
|
||||
epoll_ctl(epoll_fd, EPOLL_CTL_ADD, timer_fd, &ev2);
|
||||
|
||||
printf("Runtime started. Timer fires every 2s.\n");
|
||||
printf("Write to eventfd to trigger it: echo 1 > /proc/%d/fd/%d\n",
|
||||
getpid(), event_fd);
|
||||
|
||||
// The event loop — this IS the runtime
|
||||
struct epoll_event events[10];
|
||||
while (1) {
|
||||
int n = epoll_wait(epoll_fd, events, 10, -1); // block until ready
|
||||
|
||||
for (int i = 0; i < n; i++) {
|
||||
int fd = events[i].data.fd;
|
||||
|
||||
if (fd == timer_fd) {
|
||||
uint64_t expirations;
|
||||
read(timer_fd, &expirations, sizeof(expirations));
|
||||
printf("[timer] fired (%lu expirations)\n", expirations);
|
||||
}
|
||||
else if (fd == event_fd) {
|
||||
uint64_t val;
|
||||
read(event_fd, &val, sizeof(val));
|
||||
printf("[event] signaled (value: %lu)\n", val);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
close(epoll_fd);
|
||||
close(event_fd);
|
||||
close(timer_fd);
|
||||
return 0;
|
||||
}
|
||||
```
|
||||
|
||||
Compile and run:
|
||||
|
||||
```bash
|
||||
gcc -o runtime runtime.c
|
||||
./runtime
|
||||
```
|
||||
|
||||
Output:
|
||||
|
||||
```
|
||||
Runtime started. Timer fires every 2s.
|
||||
[timer] fired (1 expirations)
|
||||
[timer] fired (1 expirations)
|
||||
[timer] fired (1 expirations)
|
||||
...
|
||||
```
|
||||
|
||||
One thread. Two fd types. One loop. The kernel does the scheduling.
|
||||
|
||||
### Step 4: Add a TCP server to the same loop
|
||||
|
||||
```c
|
||||
#include <sys/epoll.h>
|
||||
#include <sys/socket.h>
|
||||
#include <netinet/in.h>
|
||||
#include <unistd.h>
|
||||
#include <stdio.h>
|
||||
#include <fcntl.h>
|
||||
#include <string.h>
|
||||
|
||||
int main() {
|
||||
int epoll_fd = epoll_create1(0);
|
||||
|
||||
// Create a non-blocking TCP listener
|
||||
int listen_fd = socket(AF_INET, SOCK_STREAM | SOCK_NONBLOCK, 0);
|
||||
int opt = 1;
|
||||
setsockopt(listen_fd, SOL_SOCKET, SO_REUSEADDR, &opt, sizeof(opt));
|
||||
|
||||
struct sockaddr_in addr = {
|
||||
.sin_family = AF_INET,
|
||||
.sin_port = htons(8080),
|
||||
.sin_addr.s_addr = INADDR_ANY
|
||||
};
|
||||
bind(listen_fd, (struct sockaddr*)&addr, sizeof(addr));
|
||||
listen(listen_fd, 128);
|
||||
|
||||
// Register listener on epoll
|
||||
struct epoll_event ev = { .events = EPOLLIN, .data.fd = listen_fd };
|
||||
epoll_ctl(epoll_fd, EPOLL_CTL_ADD, listen_fd, &ev);
|
||||
|
||||
printf("Listening on port 8080\n");
|
||||
|
||||
struct epoll_event events[64];
|
||||
while (1) {
|
||||
int n = epoll_wait(epoll_fd, events, 64, -1);
|
||||
|
||||
for (int i = 0; i < n; i++) {
|
||||
int fd = events[i].data.fd;
|
||||
|
||||
if (fd == listen_fd) {
|
||||
// Accept new connection
|
||||
int client_fd = accept4(listen_fd, NULL, NULL, SOCK_NONBLOCK);
|
||||
if (client_fd >= 0) {
|
||||
struct epoll_event cev = { .events = EPOLLIN, .data.fd = client_fd };
|
||||
epoll_ctl(epoll_fd, EPOLL_CTL_ADD, client_fd, &cev);
|
||||
printf("[accept] client fd=%d\n", client_fd);
|
||||
}
|
||||
} else {
|
||||
// Read from client
|
||||
char buf[4096];
|
||||
int nbytes = read(fd, buf, sizeof(buf));
|
||||
if (nbytes <= 0) {
|
||||
// Client disconnected
|
||||
epoll_ctl(epoll_fd, EPOLL_CTL_DEL, fd, NULL);
|
||||
close(fd);
|
||||
printf("[close] fd=%d\n", fd);
|
||||
} else {
|
||||
// Echo response
|
||||
const char *response =
|
||||
"HTTP/1.1 200 OK\r\n"
|
||||
"Content-Length: 13\r\n"
|
||||
"\r\n"
|
||||
"Hello, world!";
|
||||
write(fd, response, strlen(response));
|
||||
epoll_ctl(epoll_fd, EPOLL_CTL_DEL, fd, NULL);
|
||||
close(fd);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
This is a complete HTTP server — no threads, no framework, no library. One `epoll_wait` drives accept, read, write, and close for every connection.
|
||||
|
||||
## From C to Rust: The writeonce Runtime
|
||||
|
||||
The writeonce `wo-event` crate wraps these same syscalls in safe Rust:
|
||||
|
||||
| C syscall | Rust wrapper | Crate |
|
||||
|-----------|-------------|-------|
|
||||
| `epoll_create1` | `EventLoop::new()` | wo-event |
|
||||
| `epoll_ctl(ADD)` | `EventLoop::register(fd, interest, token)` | wo-event |
|
||||
| `epoll_ctl(MOD)` | `EventLoop::modify(fd, interest, token)` | wo-event |
|
||||
| `epoll_ctl(DEL)` | `EventLoop::deregister(fd)` | wo-event |
|
||||
| `epoll_wait` | `EventLoop::poll(timeout)` | wo-event |
|
||||
| `eventfd` | `EventFd::new()` | wo-event |
|
||||
| `timerfd_create` | `TimerFd::new()` | wo-event |
|
||||
| `signalfd` | `SignalFd::new()` | wo-event |
|
||||
|
||||
The key difference from the C examples: instead of matching on raw fd numbers, the Rust runtime assigns a **token** (u64) to each fd. The event loop returns tokens, and the runtime dispatches on them:
|
||||
|
||||
```rust
|
||||
let events = event_loop.poll(Some(Duration::from_millis(500)))?;
|
||||
|
||||
for event in events {
|
||||
match event.token {
|
||||
TOKEN_WATCHER => handle_file_change(),
|
||||
TOKEN_SIGNAL => handle_shutdown(),
|
||||
TOKEN_TIMER => handle_periodic_task(),
|
||||
TOKEN_LISTENER => handle_new_connection(),
|
||||
token if token >= 10000 => handle_http(token),
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
## Why Not tokio?
|
||||
|
||||
tokio is a production-grade async runtime. It handles everything — epoll, thread pools, work stealing, timers, I/O drivers. So why build a custom one?
|
||||
|
||||
| Concern | tokio | Custom runtime |
|
||||
|---------|-------|---------------|
|
||||
| Binary size | Adds ~2-3 MB | Zero — just libc syscalls |
|
||||
| Dependencies | 50+ transitive crates | 1 crate (libc) |
|
||||
| Complexity | Work-stealing scheduler, multi-threaded executor | Single-threaded loop, ~200 lines |
|
||||
| Control | Opaque — runtime internals hidden behind `.await` | Every fd, every syscall, every state transition is explicit |
|
||||
| Learning | Abstracts the kernel away | Forces understanding of what the kernel actually does |
|
||||
|
||||
For a content platform serving markdown files, the workload is: accept connection, read request, query in-memory index, render template, write response. This is microseconds of work per request. A single-threaded event loop handles thousands of concurrent connections without the complexity of a multi-threaded executor.
|
||||
|
||||
## Reusing the Runtime in Other Projects
|
||||
|
||||
The `wo-event` crate has no dependency on writeonce. It provides:
|
||||
|
||||
- `EventLoop` — epoll wrapper with register/deregister/poll
|
||||
- `EventFd` — lightweight signaling
|
||||
- `TimerFd` — periodic timers as fds
|
||||
- `SignalFd` — SIGINT/SIGTERM as fd events
|
||||
- `Event` — readable/writable/hangup status
|
||||
- `Token` — u64 identifier for dispatch
|
||||
|
||||
Any project that needs non-blocking I/O on Linux can use it:
|
||||
|
||||
```rust
|
||||
use wo_event::{EventLoop, EventFd, Interest};
|
||||
use std::time::Duration;
|
||||
|
||||
fn main() {
|
||||
let eloop = EventLoop::new().unwrap();
|
||||
let efd = EventFd::new().unwrap();
|
||||
|
||||
eloop.register(efd.fd(), Interest::Readable, 1).unwrap();
|
||||
|
||||
// Signal from another thread
|
||||
std::thread::spawn(move || {
|
||||
std::thread::sleep(Duration::from_secs(1));
|
||||
efd.write(42).unwrap();
|
||||
});
|
||||
|
||||
// Wait for the signal
|
||||
let events = eloop.poll(Some(Duration::from_secs(5))).unwrap();
|
||||
assert_eq!(events[0].token, 1);
|
||||
println!("Event received!");
|
||||
}
|
||||
```
|
||||
|
||||
## The Mental Model
|
||||
|
||||
```
|
||||
┌─────────────────────────────────┐
|
||||
│ your code │
|
||||
│ (handlers, business logic) │
|
||||
└──────────────┬──────────────────┘
|
||||
│ dispatches on token
|
||||
┌──────────────┴──────────────────┐
|
||||
│ event loop │
|
||||
│ epoll_wait → Vec<Event> │
|
||||
└──────────────┬──────────────────┘
|
||||
│ registered fds
|
||||
┌──────────────┴──────────────────┐
|
||||
│ Linux kernel │
|
||||
│ tracks readiness for all fds │
|
||||
│ inotify, sockets, timers, │
|
||||
│ signals, eventfds — all fds │
|
||||
└─────────────────────────────────┘
|
||||
```
|
||||
|
||||
The kernel is the scheduler. The event loop is the dispatcher. Your code is the handler. Everything in the system — files, sockets, timers, signals — is a file descriptor. One loop to rule them all.
|
||||
253
docs/runtime/fibers.md
Normal file
253
docs/runtime/fibers.md
Normal file
|
|
@ -0,0 +1,253 @@
|
|||
# Runtime Fibers
|
||||
|
||||
Runtime fibers are lightweight, user-space threads managed by an application's runtime system rather than the OS kernel. They enable massive concurrency — millions per machine — because they don't carry the overhead of kernel thread stacks and scheduling. Unlike pre-emptive kernel threads, fibers use cooperative multitasking: they yield control voluntarily at known suspension points.
|
||||
|
||||
## Context Switching
|
||||
|
||||
A context switch is saving the state of one execution unit and restoring another so it can continue running. The cost of this switch is what separates kernel threads from fibers.
|
||||
|
||||
### Kernel Thread Context Switch
|
||||
|
||||
When the OS switches between threads:
|
||||
|
||||
1. Save all CPU registers (general purpose, floating point, SIMD) to kernel memory
|
||||
2. Save the thread's stack pointer
|
||||
3. Flush the TLB (translation lookaside buffer) if switching processes
|
||||
4. Update scheduler data structures
|
||||
5. Restore the next thread's registers and stack pointer
|
||||
6. Return to userspace
|
||||
|
||||
Cost: **1-10 microseconds**, involves a kernel trap (syscall boundary crossing), cache pollution from TLB flush.
|
||||
|
||||
### Fiber Context Switch
|
||||
|
||||
When a runtime switches between fibers:
|
||||
|
||||
1. Save a few registers (stack pointer, instruction pointer, callee-saved registers)
|
||||
2. Swap the stack pointer to the next fiber's stack
|
||||
3. Jump to the next fiber's saved instruction pointer
|
||||
|
||||
Cost: **~10-100 nanoseconds**, entirely in userspace, no kernel involvement, no TLB flush, cache stays warm.
|
||||
|
||||
```
|
||||
Kernel thread switch: ~1,000-10,000 ns (kernel trap + TLB flush)
|
||||
Fiber switch: ~10-100 ns (register swap in userspace)
|
||||
100x cheaper
|
||||
```
|
||||
|
||||
## Types of Multitasking
|
||||
|
||||
### Pre-emptive (Kernel Threads)
|
||||
|
||||
The OS scheduler interrupts threads at arbitrary points using timer interrupts. The thread does not choose when to yield — the kernel forces it.
|
||||
|
||||
```
|
||||
Thread A: ████████──┐ (interrupted by OS)
|
||||
│
|
||||
Thread B: └──████████──┐ (interrupted by OS)
|
||||
│
|
||||
Thread A: └──████████
|
||||
```
|
||||
|
||||
- Threads can be interrupted mid-instruction
|
||||
- Requires locks/mutexes to protect shared state
|
||||
- Fairness guaranteed by the scheduler
|
||||
- Used by: pthreads, std::thread, OS processes
|
||||
|
||||
### Cooperative (Fibers / Green Threads)
|
||||
|
||||
Fibers explicitly yield at known points (I/O boundaries, channel sends, `.await` in Rust). The runtime only switches when the fiber says "I'm done for now."
|
||||
|
||||
```
|
||||
Fiber A: ████████ yield ──┐
|
||||
│
|
||||
Fiber B: └── ████████ yield ──┐
|
||||
│
|
||||
Fiber A: └── ████████
|
||||
```
|
||||
|
||||
- Fibers are never interrupted mid-computation
|
||||
- No locks needed for single-threaded runtimes — only one fiber runs at a time
|
||||
- Starvation possible if a fiber never yields (compute-heavy work blocks the loop)
|
||||
- Used by: Go goroutines, Erlang processes, Lua coroutines, Rust async/await
|
||||
|
||||
### Comparison
|
||||
|
||||
| Property | Pre-emptive (Threads) | Cooperative (Fibers) |
|
||||
|----------|----------------------|---------------------|
|
||||
| Scheduling | OS kernel decides | Runtime decides at yield points |
|
||||
| Context switch cost | ~1-10 us | ~10-100 ns |
|
||||
| Stack size | 1-8 MB per thread (fixed) | Bytes to KB per fiber (growable) |
|
||||
| Max concurrency | ~10,000 threads | ~1,000,000+ fibers |
|
||||
| Synchronization | Locks, mutexes, atomics | Not needed in single-threaded runtime |
|
||||
| Interruption | Any point (timer interrupt) | Only at yield points |
|
||||
| Fairness | Guaranteed by scheduler | Must be designed (fiber must yield) |
|
||||
|
||||
## How Fibers Work Internally
|
||||
|
||||
A fiber needs three things:
|
||||
|
||||
1. **A stack** — a block of memory for local variables and call frames
|
||||
2. **A saved context** — the register state at the point it yielded
|
||||
3. **A function** — the code to run when resumed
|
||||
|
||||
### Minimal Fiber in C
|
||||
|
||||
```c
|
||||
#include <ucontext.h>
|
||||
#include <stdio.h>
|
||||
|
||||
static ucontext_t main_ctx, fiber_ctx;
|
||||
static char fiber_stack[8192];
|
||||
|
||||
void fiber_fn() {
|
||||
printf("Fiber: running\n");
|
||||
// Yield back to main
|
||||
swapcontext(&fiber_ctx, &main_ctx);
|
||||
printf("Fiber: resumed\n");
|
||||
// Yield again
|
||||
swapcontext(&fiber_ctx, &main_ctx);
|
||||
}
|
||||
|
||||
int main() {
|
||||
// Set up fiber context
|
||||
getcontext(&fiber_ctx);
|
||||
fiber_ctx.uc_stack.ss_sp = fiber_stack;
|
||||
fiber_ctx.uc_stack.ss_size = sizeof(fiber_stack);
|
||||
fiber_ctx.uc_link = &main_ctx;
|
||||
makecontext(&fiber_ctx, fiber_fn, 0);
|
||||
|
||||
printf("Main: starting fiber\n");
|
||||
swapcontext(&main_ctx, &fiber_ctx); // switch to fiber
|
||||
|
||||
printf("Main: fiber yielded\n");
|
||||
swapcontext(&main_ctx, &fiber_ctx); // resume fiber
|
||||
|
||||
printf("Main: fiber yielded again\n");
|
||||
swapcontext(&main_ctx, &fiber_ctx); // resume — fiber finishes
|
||||
|
||||
printf("Main: done\n");
|
||||
return 0;
|
||||
}
|
||||
```
|
||||
|
||||
Output:
|
||||
```
|
||||
Main: starting fiber
|
||||
Fiber: running
|
||||
Main: fiber yielded
|
||||
Fiber: resumed
|
||||
Main: fiber yielded again
|
||||
Main: done
|
||||
```
|
||||
|
||||
`swapcontext` saves the current register state to one context struct and loads another — that's the entire fiber switch. No kernel involved.
|
||||
|
||||
### Minimal Fiber in Rust (unsafe)
|
||||
|
||||
Rust's async/await compiles to state machines, not stack-swapping fibers. But you can build raw fibers with inline assembly:
|
||||
|
||||
```rust
|
||||
use std::arch::asm;
|
||||
|
||||
struct Fiber {
|
||||
stack: Vec<u8>,
|
||||
sp: *mut u8, // saved stack pointer
|
||||
}
|
||||
|
||||
impl Fiber {
|
||||
fn new(func: fn()) -> Self {
|
||||
let mut stack = vec![0u8; 8192];
|
||||
let sp = unsafe {
|
||||
let top = stack.as_mut_ptr().add(stack.len());
|
||||
let aligned = (top as usize & !0xF) as *mut u8;
|
||||
// Push the function pointer as the return address
|
||||
let sp = aligned.sub(8);
|
||||
*(sp as *mut fn()) = func;
|
||||
sp
|
||||
};
|
||||
Fiber { stack, sp }
|
||||
}
|
||||
|
||||
unsafe fn switch_to(&mut self, from: &mut *mut u8) {
|
||||
// Save callee-saved registers and swap stack pointers
|
||||
asm!(
|
||||
"push rbx",
|
||||
"push rbp",
|
||||
"push r12",
|
||||
"push r13",
|
||||
"push r14",
|
||||
"push r15",
|
||||
"mov [{from}], rsp", // save current sp
|
||||
"mov rsp, [{to}]", // load fiber sp
|
||||
"pop r15",
|
||||
"pop r14",
|
||||
"pop r13",
|
||||
"pop r12",
|
||||
"pop rbp",
|
||||
"pop rbx",
|
||||
from = in(reg) from,
|
||||
to = in(reg) &self.sp,
|
||||
);
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
This is what runtimes like Go and Erlang do internally — allocate a small stack, save/restore a handful of registers, and jump. The cost is a few nanoseconds.
|
||||
|
||||
## Fibers vs Rust async/await
|
||||
|
||||
Rust chose a different approach than fibers for its async model:
|
||||
|
||||
| Property | Fibers (Go, Erlang) | Rust async/await |
|
||||
|----------|-------------------|-----------------|
|
||||
| Implementation | Stack swapping at runtime | Compiler generates state machines |
|
||||
| Stack | Each fiber has its own stack | No extra stack — state stored in Future struct |
|
||||
| Memory per task | ~2-8 KB minimum (stack) | Bytes — only the live variables at yield points |
|
||||
| Yield mechanism | `swapcontext` / assembly | `.await` compiles to `Poll::Pending` |
|
||||
| Overhead | Stack allocation + register swap | Zero-cost — state machine is a regular struct |
|
||||
| Debuggability | Separate stacks in debugger | State machine is harder to trace |
|
||||
| Preemption | Runtime can preempt (Go does this) | Never preempted — cooperative only |
|
||||
|
||||
Rust's approach is called **stackless coroutines** — no extra stack per task. The compiler transforms each `async fn` into a state machine enum where each variant holds the local variables alive across an `.await` point.
|
||||
|
||||
## Relation to writeonce
|
||||
|
||||
The writeonce runtime (`wo-event`, `wo-rt`) uses neither fibers nor Rust async/await. It uses a **single-threaded event loop with callbacks** — the simplest model:
|
||||
|
||||
```
|
||||
loop {
|
||||
events = epoll_wait()
|
||||
for event in events {
|
||||
match event.token {
|
||||
WATCHER => handle_file_change(),
|
||||
LISTENER => handle_accept(),
|
||||
HTTP_CONN => handle_request(),
|
||||
...
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
This is the same model as nginx, Redis, and Node.js (before libuv's thread pool). It works because:
|
||||
|
||||
- Each handler runs to completion quickly (microseconds)
|
||||
- No handler blocks — all I/O is non-blocking
|
||||
- No concurrent access to shared state — one thing runs at a time
|
||||
- Blocking work (index rebuild) offloads to a thread pool and signals back via eventfd
|
||||
|
||||
If writeonce ever needed millions of concurrent long-lived tasks (not just connections), fibers would be the next step. But for a content platform serving articles, the event loop is sufficient — and far simpler to reason about.
|
||||
|
||||
## Summary
|
||||
|
||||
```
|
||||
Kernel threads: OS-managed, pre-emptive, expensive switch, ~10K max
|
||||
Fibers: Runtime-managed, cooperative, cheap switch, ~1M+ max
|
||||
Async/await: Compiler-managed, cooperative, zero-cost, ~1M+ max
|
||||
Event loop: No tasks at all — just fd readiness + callbacks
|
||||
|
||||
Complexity: event loop < fibers < async/await < threads
|
||||
Concurrency: event loop = fibers = async/await >> threads
|
||||
```
|
||||
|
||||
The right choice depends on the workload. For writeonce — an event loop. For a database with millions of queries in flight — fibers or async. For CPU-bound parallel work — kernel threads.
|
||||
248
docs/runtime/garbage-collection.md
Normal file
248
docs/runtime/garbage-collection.md
Normal file
|
|
@ -0,0 +1,248 @@
|
|||
# Why Rust Does Not Use Fibers or Garbage Collection
|
||||
|
||||
## The Question
|
||||
|
||||
Most runtime-heavy languages ship with two things: a garbage collector (Go, Java, Python, C#, Erlang) and fiber-like concurrency (Go goroutines, Erlang processes, Java virtual threads). Rust ships with neither. Why?
|
||||
|
||||
The answer is the same for both: **Rust pushes the cost to compile time so there is zero cost at runtime.**
|
||||
|
||||
## Garbage Collection
|
||||
|
||||
### What a GC Does
|
||||
|
||||
A garbage collector tracks which objects in memory are still reachable from the program. Periodically (or continuously), it scans the heap, finds objects nothing points to, and frees them.
|
||||
|
||||
```
|
||||
Allocate object A
|
||||
Allocate object B
|
||||
A.ref = B // B is reachable through A
|
||||
drop(A) // A is unreachable — GC will free A
|
||||
// B is now also unreachable — GC will free B
|
||||
```
|
||||
|
||||
### How It Works (Simplified)
|
||||
|
||||
**Mark-and-sweep** (Go, Java):
|
||||
|
||||
```
|
||||
1. Pause the program (or run concurrently)
|
||||
2. Start from "roots" (stack variables, globals)
|
||||
3. Mark every object reachable from roots
|
||||
4. Sweep: free every object NOT marked
|
||||
```
|
||||
|
||||
**Reference counting** (Python, Swift, Objective-C):
|
||||
|
||||
```
|
||||
1. Every object has a counter
|
||||
2. When a reference is created: counter++
|
||||
3. When a reference is dropped: counter--
|
||||
4. When counter == 0: free immediately
|
||||
```
|
||||
|
||||
### The Costs
|
||||
|
||||
| Cost | Mark-and-sweep GC | Reference counting |
|
||||
|------|-------------------|-------------------|
|
||||
| Pause time | Stop-the-world pauses (Go: ~1ms, Java: varies) | No pauses, but slower per-operation |
|
||||
| Memory overhead | 2x heap needed (live objects + garbage until collected) | Counter per object (8 bytes) |
|
||||
| CPU overhead | GC thread scanning heap (10-30% throughput loss) | Increment/decrement on every pointer operation |
|
||||
| Predictability | Unpredictable latency spikes | Predictable but cycles leak (need cycle collector) |
|
||||
| Cache impact | GC walks heap → cache pollution | Counters spread across memory → cache misses |
|
||||
|
||||
For a content platform that needs predictable low-latency responses, GC pauses are the enemy. Even Go's ~1ms pauses compound under load — if a GC pause hits during `epoll_wait`, every pending connection stalls.
|
||||
|
||||
### What Rust Does Instead: Ownership
|
||||
|
||||
Rust replaces garbage collection with a compile-time ownership system:
|
||||
|
||||
```rust
|
||||
fn main() {
|
||||
let s = String::from("hello"); // s owns the string, allocated on heap
|
||||
let t = s; // ownership moves to t — s is invalid
|
||||
// println!("{}", s); // compile error: s was moved
|
||||
println!("{}", t); // ok
|
||||
} // t goes out of scope → String freed here. Deterministic. No GC.
|
||||
```
|
||||
|
||||
The rules:
|
||||
|
||||
1. **Every value has exactly one owner**
|
||||
2. **When the owner goes out of scope, the value is dropped (freed)**
|
||||
3. **Ownership can be moved or borrowed, but never duplicated**
|
||||
|
||||
The compiler enforces these rules at compile time. At runtime, there is:
|
||||
- No GC thread
|
||||
- No mark phase
|
||||
- No sweep phase
|
||||
- No reference counters
|
||||
- No heap scanning
|
||||
- No pauses
|
||||
|
||||
Memory is freed at the exact point it is no longer needed — deterministically, at the closing brace.
|
||||
|
||||
### Lifetimes: The Compile-Time GC
|
||||
|
||||
References (borrows) have lifetimes — the compiler tracks how long each reference lives and ensures no reference outlives its data:
|
||||
|
||||
```rust
|
||||
fn longest<'a>(x: &'a str, y: &'a str) -> &'a str {
|
||||
if x.len() > y.len() { x } else { y }
|
||||
}
|
||||
```
|
||||
|
||||
The `'a` lifetime annotation tells the compiler: "the returned reference lives as long as both inputs." If you try to return a reference to a local variable, the compiler rejects it — at compile time, not at runtime.
|
||||
|
||||
```rust
|
||||
fn bad() -> &str {
|
||||
let s = String::from("hello");
|
||||
&s // compile error: s is dropped at end of function, reference would dangle
|
||||
}
|
||||
```
|
||||
|
||||
This is what a GC does at runtime (detect unreachable memory). Rust does it at compile time (detect impossible references). Zero runtime cost.
|
||||
|
||||
### The Tradeoff
|
||||
|
||||
| Aspect | GC languages | Rust |
|
||||
|--------|-------------|------|
|
||||
| Developer effort | Low — just allocate, GC handles cleanup | Higher — must think about ownership and lifetimes |
|
||||
| Compile time | Fast | Slower (borrow checker analysis) |
|
||||
| Runtime cost | GC pauses, heap scanning, memory overhead | Zero — deterministic drop at scope exit |
|
||||
| Latency | Unpredictable (GC can pause anytime) | Predictable — no hidden pauses |
|
||||
| Memory usage | 2x+ (garbage accumulates between collections) | Tight — freed immediately when unused |
|
||||
|
||||
Rust trades developer convenience for runtime performance. For systems software (databases, runtimes, web servers), this is the right trade.
|
||||
|
||||
## Fibers
|
||||
|
||||
### Why Other Languages Use Fibers
|
||||
|
||||
Go has goroutines. Erlang has processes. Java 21 has virtual threads. These are all fibers — lightweight user-space threads that the runtime schedules cooperatively (or semi-preemptively in Go's case).
|
||||
|
||||
They exist because these languages need to:
|
||||
1. Handle millions of concurrent I/O tasks
|
||||
2. Let developers write synchronous-looking code (`result = fetch(url)`) that blocks the fiber, not the OS thread
|
||||
3. Manage scheduling without exposing the event loop
|
||||
|
||||
```go
|
||||
// Go: goroutine blocks on I/O — runtime suspends it and runs another
|
||||
go func() {
|
||||
resp, _ := http.Get("https://example.com") // blocks this goroutine, not the thread
|
||||
fmt.Println(resp.Status)
|
||||
}()
|
||||
```
|
||||
|
||||
### Why Rust Does Not Use Fibers
|
||||
|
||||
**1. Fibers require a runtime that allocates stacks.**
|
||||
|
||||
Each fiber needs its own stack (Go starts at 2-8 KB, grows dynamically). This means:
|
||||
- A heap allocation per fiber
|
||||
- Stack overflow checks on every function call
|
||||
- A runtime that manages stack growth and shrinking
|
||||
- Memory overhead proportional to number of concurrent tasks
|
||||
|
||||
Rust's goal is zero-cost abstractions. Allocating stacks at runtime is a cost.
|
||||
|
||||
**2. Fibers are hard to optimize across FFI boundaries.**
|
||||
|
||||
Rust interoperates with C libraries extensively. Fibers with tiny stacks can't safely call into C code (which expects a full OS stack). Go solves this by switching to a system stack for cgo calls — adding complexity and overhead.
|
||||
|
||||
**3. Async/await achieves the same concurrency without stacks.**
|
||||
|
||||
Rust's async/await compiles each async function into a state machine — a regular struct stored inline, no heap allocation needed:
|
||||
|
||||
```rust
|
||||
async fn fetch_article(title: &str) -> Article {
|
||||
let data = read_from_seg(title).await; // suspend point 1
|
||||
let html = render_markdown(&data).await; // suspend point 2
|
||||
Article { title, html }
|
||||
}
|
||||
```
|
||||
|
||||
The compiler transforms this into something like:
|
||||
|
||||
```rust
|
||||
enum FetchArticle {
|
||||
Start { title: String },
|
||||
AfterRead { title: String, data: Vec<u8> },
|
||||
AfterRender { title: String, html: String },
|
||||
Done,
|
||||
}
|
||||
```
|
||||
|
||||
Each `.await` becomes a variant transition. The "stack" is just the live variables in the current variant — bytes, not kilobytes. No allocation, no stack, no runtime overhead.
|
||||
|
||||
### Fiber vs Async/Await: Memory Per Task
|
||||
|
||||
```
|
||||
Go goroutine: ~2,048 bytes minimum (stack)
|
||||
Erlang process: ~2,688 bytes minimum (stack + heap + mailbox)
|
||||
Rust async task: size_of::<FetchArticle>() — often 32-128 bytes
|
||||
```
|
||||
|
||||
For a million concurrent connections:
|
||||
- Go: ~2 GB just for goroutine stacks
|
||||
- Rust: ~128 MB for state machines (and often less, since the executor batches them)
|
||||
|
||||
### When Fibers Would Be Better
|
||||
|
||||
Fibers have one advantage: **deeply nested call stacks that suspend at arbitrary points.** If a function 20 calls deep needs to yield, a fiber just swaps the stack pointer. With async/await, every function in the chain must be `async` and every call must be `.await`ed — the "async infection" problem.
|
||||
|
||||
```
|
||||
Fibers: yield anywhere in the call stack — transparent to callers
|
||||
Async: yield only at .await points — every caller must be async
|
||||
```
|
||||
|
||||
For database engines with complex query execution plans that suspend mid-evaluation, fibers are compelling. For an HTTP server that suspends at I/O boundaries, async/await is strictly better.
|
||||
|
||||
## How This Applies to writeonce
|
||||
|
||||
writeonce uses neither fibers nor async/await. It uses a plain event loop:
|
||||
|
||||
```rust
|
||||
loop {
|
||||
events = epoll_wait();
|
||||
for event in events {
|
||||
handle(event); // runs to completion, no suspension
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
This is the simplest model — no GC, no fibers, no async state machines. Each handler reads from the `.seg` file, renders a template, writes to the socket, and returns. Nothing suspends mid-handler.
|
||||
|
||||
The memory model:
|
||||
|
||||
| What | How it's managed |
|
||||
|------|-----------------|
|
||||
| Article data in .seg | Owned by `Store`, freed when `Store` drops |
|
||||
| Template ASTs | Owned by `TemplateRegistry`, live for the process lifetime |
|
||||
| HTTP connections | Owned by `HashMap<Token, Connection>`, freed on close/hangup |
|
||||
| Subscription table | Owned by `SubscriptionManager`, entries removed on `EPOLLHUP` |
|
||||
|
||||
No garbage. No fibers. No async. Just ownership, scopes, and the kernel's event notification. The Rust compiler guarantees at compile time that every allocation is freed exactly once, at exactly the right time.
|
||||
|
||||
## Summary
|
||||
|
||||
```
|
||||
GC languages (Go, Java): runtime scans heap → frees unreachable objects
|
||||
cost: pauses, memory overhead, CPU overhead
|
||||
|
||||
Reference counting (Python): counter per object → free at zero
|
||||
cost: per-operation overhead, cycle leaks
|
||||
|
||||
Rust ownership: compiler tracks ownership → free at scope exit
|
||||
cost: zero at runtime, developer thinks harder
|
||||
|
||||
Fibers (Go, Erlang): runtime manages stacks → swap on yield
|
||||
cost: stack allocation, stack checks, runtime
|
||||
|
||||
Async/await (Rust): compiler generates state machines → no stack
|
||||
cost: zero allocation, async must propagate
|
||||
|
||||
Event loop (writeonce): no tasks, no suspension → handlers run to completion
|
||||
cost: nothing — simplest possible model
|
||||
```
|
||||
|
||||
Rust's answer to both GC and fibers is the same: **make the compiler do the work so the runtime doesn't have to.**
|
||||
157
docs/runtime/surreal-case-study.md
Normal file
157
docs/runtime/surreal-case-study.md
Normal file
|
|
@ -0,0 +1,157 @@
|
|||
# SurrealDB — Runtime Case Study
|
||||
|
||||
Reference repository: [github.com/surrealdb/surrealdb](https://github.com/surrealdb/surrealdb)
|
||||
|
||||
```bash
|
||||
git submodule add https://github.com/surrealdb/surrealdb.git references/surrealdb
|
||||
```
|
||||
|
||||
## The Question
|
||||
|
||||
Does SurrealDB only rely on async/await for concurrency?
|
||||
|
||||
**No.** SurrealDB uses a layered concurrency model — async/await is one layer, but it also uses OS thread pools, CPU-affinity-pinned workers, lock-free data structures, and parallel computation frameworks. Each layer serves a different purpose.
|
||||
|
||||
## Architecture Overview
|
||||
|
||||
SurrealDB is a single Rust binary that ships a multi-model database (documents, graphs, key-value) with real-time live queries. It supports multiple deployment modes:
|
||||
|
||||
- **Server**: `surreal start` runs HTTP/WebSocket API via Axum + storage engine
|
||||
- **Embedded**: the library crate embeds directly in Rust applications
|
||||
- **WASM**: runs in the browser with IndexedDB backend
|
||||
|
||||
## Concurrency Layers
|
||||
|
||||
### Layer 1: Tokio — Async I/O and Request Handling
|
||||
|
||||
The primary runtime. Handles:
|
||||
- HTTP/WebSocket connections (via Axum)
|
||||
- Network I/O (accept, read, write)
|
||||
- Timer-based operations
|
||||
- Task scheduling (M:N scheduling of futures onto OS threads)
|
||||
|
||||
```
|
||||
Client request → Axum handler (async) → parse query → execute → respond
|
||||
```
|
||||
|
||||
Every request handler is an async function. Tokio's multi-threaded executor distributes tasks across OS threads using work-stealing.
|
||||
|
||||
### Layer 2: Rayon — Parallel CPU-Bound Computation
|
||||
|
||||
For operations that are compute-heavy, not I/O-bound:
|
||||
- Query plan execution across partitions
|
||||
- Data processing and transformation
|
||||
- Parallel iteration over result sets
|
||||
|
||||
Rayon provides `par_iter()` — automatic parallelism across CPU cores. It has its own thread pool, separate from tokio's.
|
||||
|
||||
### Layer 3: affinitypool — CPU-Pinned Storage I/O
|
||||
|
||||
SurrealDB's custom crate. Runs blocking storage operations on a dedicated thread pool where **each thread is pinned to a specific CPU core** via `libc` CPU affinity syscalls.
|
||||
|
||||
Used by:
|
||||
- RocksDB backend (blocking disk I/O)
|
||||
- SurrealKV embedded storage
|
||||
- In-memory engine for heavy operations
|
||||
|
||||
This bridges the async world (tokio) and the blocking world (disk I/O) without polluting the tokio thread pool with blocking calls.
|
||||
|
||||
### Layer 4: Lock-Free Data Structures
|
||||
|
||||
The hot path in storage engines uses concurrent data structures that avoid locks entirely:
|
||||
|
||||
| Crate | Data Structure | Used For |
|
||||
|-------|---------------|----------|
|
||||
| `crossbeam-skiplist` | Concurrent skip list | Index structures in SurrealKV and surrealmx |
|
||||
| `crossbeam-deque` | Work-stealing deque | Task distribution |
|
||||
| `crossbeam-queue` | Lock-free queue | Message passing |
|
||||
| `papaya` | Concurrent HashMap | In-memory engine (surrealmx) |
|
||||
| `dashmap` | Sharded concurrent map | Pub/sub routing for live queries |
|
||||
| `arc-swap` | Atomic pointer swap | Hot-swapping data structures without locks |
|
||||
| `parking_lot` | Fast mutex/rwlock | Where locking is needed (faster than std) |
|
||||
|
||||
## No Fibers
|
||||
|
||||
SurrealDB does **not** use fibers, green threads, or any custom scheduling mechanism. The concurrency model is:
|
||||
|
||||
```
|
||||
Tokio tasks (async/await) — for I/O-bound work
|
||||
Rayon threads (par_iter) — for CPU-bound work
|
||||
affinitypool threads (pinned) — for blocking storage I/O
|
||||
Lock-free structures — for concurrent data access
|
||||
```
|
||||
|
||||
This is pragmatic — each concurrency mechanism is used where it fits, rather than forcing everything through one model.
|
||||
|
||||
## Live Queries / Real-Time Subscriptions
|
||||
|
||||
SurrealDB's live query system pushes changes to connected clients in real-time:
|
||||
|
||||
1. Client registers a live query via WebSocket: `LIVE SELECT * FROM person WHERE age > 21`
|
||||
2. Server tracks the query in a `dashmap` (concurrent map)
|
||||
3. When a transaction commits changes to `person`, the engine evaluates which live queries are affected
|
||||
4. Matching subscribers receive the diff via their WebSocket connection
|
||||
5. Transport: `tokio-tungstenite` for WebSocket, `async-channel` for internal pub/sub routing
|
||||
|
||||
### Comparison with writeonce Subscriptions
|
||||
|
||||
| Aspect | SurrealDB | writeonce |
|
||||
|--------|-----------|-----------|
|
||||
| Transport | WebSocket (tokio-tungstenite) | Raw socket fd (kernel-level write) |
|
||||
| Query registration | SQL-like live query over WebSocket | `register!` macro binding fd to content pattern |
|
||||
| Change detection | Transaction commit triggers evaluation | inotify detects file change |
|
||||
| Notification routing | `dashmap` + `async-channel` | `SubscriptionManager` HashMap + direct `write(fd)` |
|
||||
| Runtime | Tokio multi-threaded executor | Single-threaded epoll event loop |
|
||||
| Protocol framing | WebSocket frames | Length-prefixed payloads (no protocol) |
|
||||
|
||||
SurrealDB's live queries are the architectural inspiration for writeonce's subscription model (as noted in [03-data.md](../03-data.md)), but the implementation is fundamentally different — SurrealDB uses a full async runtime with WebSocket transport, while writeonce uses kernel fd notifications with no protocol layer.
|
||||
|
||||
## Storage Engine Architecture
|
||||
|
||||
SurrealDB supports 5 backends:
|
||||
|
||||
| Backend | Type | Concurrency |
|
||||
|---------|------|------------|
|
||||
| **surrealmx** | In-memory | Lock-free (papaya, crossbeam-skiplist, arc-swap) |
|
||||
| **surrealkv** | Embedded persistent | Tokio async + crossbeam + parking_lot |
|
||||
| **RocksDB** | Embedded persistent | affinitypool (CPU-pinned blocking threads) |
|
||||
| **TiKV** | Distributed | Async TiKV client over gRPC |
|
||||
| **IndxDB** | Browser/WASM | IndexedDB via wasm-bindgen-futures |
|
||||
|
||||
### Comparison with writeonce Storage
|
||||
|
||||
| Aspect | SurrealDB | writeonce |
|
||||
|--------|-----------|-----------|
|
||||
| Storage format | Key-value entries in LSM trees (RocksDB) or custom B-trees (SurrealKV) | `.seg` files with length-prefixed bincode records |
|
||||
| Index | Built into storage engine | Separate `.idx` files (title hash, date sorted, tags inverted) |
|
||||
| Concurrency | Multi-threaded with locks/lock-free structures | Single-threaded, positional I/O (pread/pwrite) |
|
||||
| Transaction | ACID with MVCC | Full rebuild on change (article count is small) |
|
||||
| Complexity | ~100K+ lines across storage crates | ~300 lines (wo-seg + wo-index) |
|
||||
|
||||
writeonce's storage is intentionally simple — the dataset is small (hundreds of articles, not millions of rows), so a full rebuild on change is fast enough and avoids the complexity of concurrent transactions.
|
||||
|
||||
## Key Takeaways
|
||||
|
||||
1. **Async/await alone is not enough for a database.** SurrealDB uses four concurrency mechanisms, each for a different workload profile.
|
||||
|
||||
2. **Blocking I/O needs its own thread pool.** The affinitypool pattern — CPU-pinned threads for storage operations — keeps blocking work off the async executor. writeonce avoids this entirely by using `pread` (non-blocking positional reads) in a single-threaded loop.
|
||||
|
||||
3. **Lock-free data structures matter at scale.** SurrealDB's hot path avoids mutexes. writeonce doesn't need this — single-threaded access means no contention.
|
||||
|
||||
4. **Live queries are the hard problem.** Both SurrealDB and writeonce solve "push changes to subscribers," but at vastly different scales. SurrealDB handles arbitrary SQL predicates over millions of rows. writeonce handles content queries over hundreds of articles.
|
||||
|
||||
5. **The right amount of complexity depends on the problem.** SurrealDB is a general-purpose database — it needs the complexity. writeonce is a content platform — the event loop model is sufficient and far simpler.
|
||||
|
||||
## Reference
|
||||
|
||||
Add SurrealDB as a submodule for code reference:
|
||||
|
||||
```bash
|
||||
git submodule add https://github.com/surrealdb/surrealdb.git references/surrealdb
|
||||
```
|
||||
|
||||
Key files to study:
|
||||
- `crates/core/src/kvs/` — storage engine abstraction and transaction handling
|
||||
- `crates/sdk/src/api/engine/` — live query subscription routing
|
||||
- `lib/affinitypool/` — CPU-pinned thread pool for blocking I/O
|
||||
- `crates/core/src/sql/` — query parser and execution engine
|
||||
34
infra/deploy.sh
Executable file
34
infra/deploy.sh
Executable file
|
|
@ -0,0 +1,34 @@
|
|||
#!/bin/bash
|
||||
# deploy.sh — run from the development machine to build, copy, and restart
|
||||
set -e
|
||||
|
||||
SERVER="writeonce.de"
|
||||
REMOTE_DIR="/opt/writeonce"
|
||||
|
||||
echo "Building release binary..."
|
||||
cargo build --release -p wo-rt --bin writeonce
|
||||
|
||||
echo "Copying binary to server..."
|
||||
scp target/release/writeonce "$SERVER:$REMOTE_DIR/writeonce.new"
|
||||
|
||||
echo "Syncing content..."
|
||||
rsync -az --delete content/ "$SERVER:$REMOTE_DIR/content/"
|
||||
|
||||
echo "Syncing templates..."
|
||||
rsync -az --delete templates/ "$SERVER:$REMOTE_DIR/templates/"
|
||||
|
||||
echo "Syncing static assets..."
|
||||
rsync -az --delete static/ "$SERVER:$REMOTE_DIR/static/"
|
||||
|
||||
echo "Swapping binary and restarting service..."
|
||||
ssh "$SERVER" "
|
||||
sudo mv $REMOTE_DIR/writeonce.new $REMOTE_DIR/writeonce
|
||||
sudo chmod +x $REMOTE_DIR/writeonce
|
||||
sudo chown writeonce:writeonce $REMOTE_DIR/writeonce
|
||||
sudo systemctl restart writeonce
|
||||
"
|
||||
|
||||
echo "Checking status..."
|
||||
ssh "$SERVER" "sudo systemctl status writeonce --no-pager"
|
||||
|
||||
echo "Deploy complete."
|
||||
36
infra/setup.sh
Executable file
36
infra/setup.sh
Executable file
|
|
@ -0,0 +1,36 @@
|
|||
#!/bin/bash
|
||||
# setup.sh — run once on the server to create user, directory, service, nginx, and SSL
|
||||
set -e
|
||||
|
||||
echo "Creating writeonce user..."
|
||||
sudo useradd -r -s /bin/false writeonce 2>/dev/null || echo "User already exists"
|
||||
|
||||
echo "Creating directory structure..."
|
||||
sudo mkdir -p /opt/writeonce/{content,data,templates,static}
|
||||
sudo chown -R writeonce:writeonce /opt/writeonce
|
||||
|
||||
echo "Installing systemd service..."
|
||||
sudo cp writeonce.service /etc/systemd/system/
|
||||
sudo systemctl daemon-reload
|
||||
sudo systemctl enable writeonce
|
||||
|
||||
echo "Configuring nginx..."
|
||||
sudo cp writeonce.de.nginx /etc/nginx/sites-available/writeonce.de
|
||||
sudo ln -sf /etc/nginx/sites-available/writeonce.de /etc/nginx/sites-enabled/
|
||||
sudo nginx -t
|
||||
sudo systemctl reload nginx
|
||||
|
||||
echo "Setting up SSL with Let's Encrypt..."
|
||||
sudo apt install -y certbot python3-certbot-nginx
|
||||
sudo certbot --nginx -d writeonce.de -d www.writeonce.de
|
||||
|
||||
echo "Installing certbot reload hook..."
|
||||
sudo mkdir -p /etc/letsencrypt/renewal-hooks/deploy
|
||||
sudo tee /etc/letsencrypt/renewal-hooks/deploy/reload-nginx.sh > /dev/null << 'HOOK'
|
||||
#!/bin/bash
|
||||
systemctl reload nginx
|
||||
HOOK
|
||||
sudo chmod +x /etc/letsencrypt/renewal-hooks/deploy/reload-nginx.sh
|
||||
|
||||
echo "Setup complete."
|
||||
echo "Run deploy.sh from your development machine to deploy the binary."
|
||||
40
infra/writeonce.de.nginx
Normal file
40
infra/writeonce.de.nginx
Normal file
|
|
@ -0,0 +1,40 @@
|
|||
server {
|
||||
listen 80;
|
||||
server_name writeonce.de www.writeonce.de;
|
||||
return 301 https://$server_name$request_uri;
|
||||
}
|
||||
|
||||
server {
|
||||
listen 443 ssl http2;
|
||||
server_name writeonce.de www.writeonce.de;
|
||||
|
||||
ssl_certificate /etc/letsencrypt/live/writeonce.de/fullchain.pem;
|
||||
ssl_certificate_key /etc/letsencrypt/live/writeonce.de/privkey.pem;
|
||||
ssl_protocols TLSv1.2 TLSv1.3;
|
||||
ssl_ciphers HIGH:!aNULL:!MD5;
|
||||
ssl_prefer_server_ciphers on;
|
||||
|
||||
# HSTS
|
||||
add_header Strict-Transport-Security "max-age=31536000; includeSubDomains" always;
|
||||
|
||||
location / {
|
||||
proxy_pass http://127.0.0.1:3000;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $scheme;
|
||||
|
||||
# Keep connections open for database subscriptions
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Connection "";
|
||||
proxy_read_timeout 86400s;
|
||||
proxy_send_timeout 86400s;
|
||||
}
|
||||
|
||||
# Static assets — let nginx serve directly for better caching
|
||||
location /static/ {
|
||||
alias /opt/writeonce/static/;
|
||||
expires 1y;
|
||||
add_header Cache-Control "public, immutable";
|
||||
}
|
||||
}
|
||||
24
infra/writeonce.service
Normal file
24
infra/writeonce.service
Normal file
|
|
@ -0,0 +1,24 @@
|
|||
[Unit]
|
||||
Description=writeonce content platform
|
||||
After=network.target
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
User=writeonce
|
||||
Group=writeonce
|
||||
WorkingDirectory=/opt/writeonce
|
||||
ExecStart=/opt/writeonce/writeonce
|
||||
Restart=on-failure
|
||||
RestartSec=5
|
||||
StandardOutput=journal
|
||||
StandardError=journal
|
||||
|
||||
# Security hardening
|
||||
NoNewPrivileges=true
|
||||
ProtectSystem=strict
|
||||
ProtectHome=true
|
||||
ReadWritePaths=/opt/writeonce/data
|
||||
PrivateTmp=true
|
||||
|
||||
[Install]
|
||||
WantedBy=multi-user.target
|
||||
38
reference/writeonce-api/.dockerignore
Normal file
38
reference/writeonce-api/.dockerignore
Normal file
|
|
@ -0,0 +1,38 @@
|
|||
# Ignore target directory (where build artifacts are stored)
|
||||
target/
|
||||
|
||||
# Ignore Cargo.lock if you want to use the latest dependencies
|
||||
# Uncomment the line below if you want to ignore it
|
||||
# Cargo.lock
|
||||
|
||||
# Ignore build files
|
||||
**/*.rs.bk
|
||||
**/*.rlib
|
||||
|
||||
# Ignore any local configuration files
|
||||
*.env
|
||||
*.local
|
||||
|
||||
# Ignore any files that are not part of the Rust project
|
||||
*.md
|
||||
*.txt
|
||||
|
||||
# Ignore version control directories
|
||||
.git/
|
||||
.gitignore
|
||||
|
||||
# Ignore IDE/editor configuration files
|
||||
.vscode/
|
||||
.idea/
|
||||
*.sublime-workspace
|
||||
*.sublime-project
|
||||
*.swp
|
||||
*.DS_Store
|
||||
|
||||
# Ignore any backup files
|
||||
*~
|
||||
|
||||
|
||||
# Diesel CLI generated schema.rs
|
||||
diesel.toml
|
||||
**/migrations/**/diesel_schema.lock
|
||||
4
reference/writeonce-api/.env.sample
Normal file
4
reference/writeonce-api/.env.sample
Normal file
|
|
@ -0,0 +1,4 @@
|
|||
PORT=
|
||||
DATABASE_URL=
|
||||
API_ACCESS_TOKEN=
|
||||
AWS_INFRA_BASE_URL=
|
||||
50
reference/writeonce-api/.gitignore
vendored
Normal file
50
reference/writeonce-api/.gitignore
vendored
Normal file
|
|
@ -0,0 +1,50 @@
|
|||
# Ignore target directory (generated build files)
|
||||
target/
|
||||
|
||||
# Ignore intermediate files from rustc compiler
|
||||
*.rs.bk
|
||||
|
||||
# Cargo-specific files
|
||||
Cargo.lock
|
||||
/.cargo/
|
||||
|
||||
# Output files and logs from Cargo
|
||||
**/*.log
|
||||
|
||||
# Ignore binaries and libraries (compiled executables or libraries)
|
||||
*.exe
|
||||
*.exe~
|
||||
*.out
|
||||
*.o
|
||||
*.a
|
||||
*.so
|
||||
*.dylib
|
||||
*.dll
|
||||
|
||||
# Ignore dependency build artifacts
|
||||
/.rustup/
|
||||
/.git/target/
|
||||
/.git/target/**
|
||||
|
||||
# macOS specific files
|
||||
.DS_Store
|
||||
|
||||
# Windows specific files
|
||||
Thumbs.db
|
||||
ehthumbs.db
|
||||
desktop.ini
|
||||
$RECYCLE.BIN/
|
||||
|
||||
# Linux specific files
|
||||
*~
|
||||
|
||||
# IDE/editor specific files (optional)
|
||||
/.idea/
|
||||
/.vscode/
|
||||
/*.iml
|
||||
|
||||
# Other tooling-specific files (optional)
|
||||
/node_modules/
|
||||
|
||||
.env
|
||||
!.env.sample
|
||||
44
reference/writeonce-api/.gitlab-ci.yml
Normal file
44
reference/writeonce-api/.gitlab-ci.yml
Normal file
|
|
@ -0,0 +1,44 @@
|
|||
stages:
|
||||
- build
|
||||
- push
|
||||
- deploy
|
||||
|
||||
build:
|
||||
stage: build
|
||||
image: docker:20.10.7
|
||||
services:
|
||||
- docker:20.10.7-dind
|
||||
before_script:
|
||||
- echo "Global setup before script"
|
||||
- chmod +x ./infra/00-build-image.sh
|
||||
script:
|
||||
- ./infra/00-build-image.sh
|
||||
|
||||
push:
|
||||
stage: push
|
||||
image: docker:20.10.7
|
||||
services:
|
||||
- docker:20.10.7-dind
|
||||
before_script:
|
||||
- chmod +x ./infra/02-docker-login.sh
|
||||
- chmod +x ./infra/03-upload-image.sh
|
||||
script:
|
||||
- infra/02-docker-login.sh
|
||||
- infra/03-upload-image.sh
|
||||
only:
|
||||
- main
|
||||
|
||||
deploy:
|
||||
stage: deploy
|
||||
image: docker:20.10.7
|
||||
services:
|
||||
- docker:20.10.7-dind
|
||||
before_script:
|
||||
- chmod +x ./infra/04-ssh-setup.sh
|
||||
- chmod +x ./infra/05-deploy.sh
|
||||
script:
|
||||
- echo "Docker set docker auth config"
|
||||
- infra/04-ssh-setup.sh
|
||||
- infra/05-deploy.sh
|
||||
only:
|
||||
- main
|
||||
51
reference/writeonce-api/Cargo.toml
Normal file
51
reference/writeonce-api/Cargo.toml
Normal file
|
|
@ -0,0 +1,51 @@
|
|||
|
||||
#[target.aarch64-unknown-linux-gnu]
|
||||
#image = "ghcr.io/cross-rs/aarch64-unknown-linux-gnu:main"
|
||||
#dockerfile = "Dockerfile.cross"
|
||||
#linker = "aarch64-linux-gnu-gcc"
|
||||
|
||||
[env]
|
||||
#OPENSSL_STATIC = "1"
|
||||
#OPENSSL_DIR = "/usr/aarch64-linux-gnu/"
|
||||
[package]
|
||||
name = "writeonce-manage-article-api"
|
||||
version = "0.1.0"
|
||||
edition = "2021"
|
||||
|
||||
[[bin]]
|
||||
name = "writeonce-manage-article-api"
|
||||
path = "src/main.rs"
|
||||
|
||||
[[bin]]
|
||||
name = "lambda-post-article"
|
||||
path = "src/lambda/post_article.rs"
|
||||
|
||||
[dependencies]
|
||||
actix-web = "4.0"
|
||||
warp = "0.3"
|
||||
serde = { version = "1.0", features = ["derive"] }
|
||||
serde_json = "1.0"
|
||||
dotenv = "0.15"
|
||||
reqwest = { version = "0.11", default-features = false, features = ["json","rustls-tls"] }
|
||||
url = "2"
|
||||
chrono = "0.4"
|
||||
actix-files = "0.6"
|
||||
############ Diesel #################
|
||||
diesel = { version = "2.2.0", default-features = false, features = ["postgres","serde_json","r2d2"] }
|
||||
diesel_migrations = "2.2.0"
|
||||
r2d2 = "0.8"
|
||||
thiserror = "1.0"
|
||||
lazy_static = "1.4"
|
||||
#openssl = { version = "0.10", features = ["vendored"] }
|
||||
##### Lambda ##########
|
||||
tokio = "1.43.0"
|
||||
lambda_runtime = "0.13.0"
|
||||
|
||||
#### AWS ####
|
||||
aws-config = { version = "1", features = ["behavior-version-latest"] }
|
||||
aws-sdk-s3 = "1"
|
||||
aws-sdk-ec2 = "1"
|
||||
aws-smithy-types = "1"
|
||||
|
||||
|
||||
|
||||
5
reference/writeonce-api/Cross.toml
Normal file
5
reference/writeonce-api/Cross.toml
Normal file
|
|
@ -0,0 +1,5 @@
|
|||
[build]
|
||||
default-target = "aarch64-unknown-linux-gnu"
|
||||
|
||||
[target.aarch64-unknown-linux-gnu]
|
||||
image = "cross-aarch64:latest"
|
||||
66
reference/writeonce-api/Dockerfile
Normal file
66
reference/writeonce-api/Dockerfile
Normal file
|
|
@ -0,0 +1,66 @@
|
|||
# Declare ARG in global scope
|
||||
ARG VERSION=1.82.0
|
||||
|
||||
FROM rust:${VERSION} AS base
|
||||
WORKDIR /usr/src/app
|
||||
|
||||
FROM base as build
|
||||
|
||||
# Install build dependencies
|
||||
|
||||
RUN apt-get update && apt-get install -y \
|
||||
curl \
|
||||
build-essential \
|
||||
pkg-config \
|
||||
libssl-dev \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
|
||||
# Install and configure sccache to speed up compilation
|
||||
RUN cargo install sccache
|
||||
ENV RUSTC_WRAPPER="sccache"
|
||||
|
||||
|
||||
# Cache dependencies to speed up builds
|
||||
COPY Cargo.toml ./
|
||||
RUN mkdir src && echo "fn main() {}" > src/main.rs
|
||||
RUN cargo build --release && rm -rf target/release/deps/*
|
||||
|
||||
COPY . .
|
||||
|
||||
RUN cargo build --release --bin writeonce-manage-article-api
|
||||
|
||||
# Stage 2: Create a minimal runtime image
|
||||
FROM debian:bookworm-slim as runtime
|
||||
|
||||
# Install necessary runtime dependencies
|
||||
RUN apt-get update && apt-get install -y \
|
||||
ca-certificates \
|
||||
libc6 \
|
||||
libpq5 \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
|
||||
COPY --from=build /usr/src/app/target/release/writeonce-manage-article-api /usr/local/bin/app
|
||||
COPY doc /usr/local/share/doc
|
||||
|
||||
ARG PORT
|
||||
ARG AWS_INFRA_BASE_URL
|
||||
ARG DATABASE_URL
|
||||
ARG API_ACCESS_TOKEN
|
||||
ARG API_ACCESS_ADMIN_TOKEN
|
||||
RUN echo $PORT
|
||||
|
||||
ENV PORT=$PORT
|
||||
ENV AWS_INFRA_BASE_URL=$AWS_INFRA_BASE_URL
|
||||
ENV API_ACCESS_TOKEN=$API_ACCESS_TOKEN
|
||||
ENV API_ACCESS_ADMIN_TOKEN=$API_ACCESS_ADMIN_TOKEN
|
||||
ENV DATABASE_URL=$DATABASE_URL
|
||||
|
||||
# Expose the application port
|
||||
EXPOSE ${PORT}
|
||||
|
||||
WORKDIR /usr/local/share
|
||||
|
||||
# Run the application
|
||||
CMD ["app"]
|
||||
16
reference/writeonce-api/Dockerfile.cross
Normal file
16
reference/writeonce-api/Dockerfile.cross
Normal file
|
|
@ -0,0 +1,16 @@
|
|||
FROM ghcr.io/cross-rs/aarch64-unknown-linux-gnu:main
|
||||
|
||||
ENV PKG_CONFIG_ALLOW_CROSS=1
|
||||
ENV PKG_CONFIG_PATH=/usr/lib/aarch64-linux-gnu/pkgconfig
|
||||
|
||||
|
||||
RUN dpkg --add-architecture arm64 && \
|
||||
apt-get update && \
|
||||
apt-get install -y \
|
||||
curl \
|
||||
build-essential \
|
||||
pkg-config \
|
||||
libssl-dev \
|
||||
libpq-dev:arm64 \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
38
reference/writeonce-api/DockerfileLambda
Normal file
38
reference/writeonce-api/DockerfileLambda
Normal file
|
|
@ -0,0 +1,38 @@
|
|||
# Stage 1: Build the Rust application using the official Rust image
|
||||
FROM rust:bullseye AS builder
|
||||
|
||||
# Install required dependencies for cargo-lambda and Zig (if needed)
|
||||
RUN apt-get update && apt-get install -y \
|
||||
curl \
|
||||
build-essential \
|
||||
gcc \
|
||||
libc-dev \
|
||||
pkg-config \
|
||||
libssl-dev \
|
||||
gcc-aarch64-linux-gnu \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Install Zig (only if needed for cross-compiling or complex setups)
|
||||
RUN curl -sSf https://ziglang.org/builds/zig-linux-x86_64-0.15.0-dev.377+f01833e03.tar.xz | tar -xJf - -C /usr/local/bin --strip-components=1
|
||||
|
||||
RUN rustup target add aarch64-unknown-linux-gnu
|
||||
# Install cargo-lambda via cargo for building AWS Lambda-compatible binaries
|
||||
RUN cargo install cargo-lambda
|
||||
|
||||
# Create a working directory for the project
|
||||
WORKDIR /app
|
||||
|
||||
# Copy the entire Rust project into the container
|
||||
COPY . .
|
||||
|
||||
# Build the Rust application with cargo-lambda, targeting Lambda-compatible format
|
||||
RUN cargo lambda build --release --bin lambda-post-article --arm64
|
||||
|
||||
# Copy the Lambda-compatible binary from the build stage into the Lambda runtime directory
|
||||
RUN cd target/lambda/lambda-post-article && zip function.zip bootstrap
|
||||
|
||||
# Stage 2: Extract just the zip artifact
|
||||
FROM alpine:latest AS artifact
|
||||
|
||||
# Copy only the zip file out of the builder
|
||||
COPY --from=builder /app/target/lambda/lambda-post-article/function.zip /function.zip
|
||||
27
reference/writeonce-api/Makefile
Normal file
27
reference/writeonce-api/Makefile
Normal file
|
|
@ -0,0 +1,27 @@
|
|||
# Variables
|
||||
CARGO=cargo
|
||||
BUILD_DIR=target/lambda
|
||||
BINARIES=lambda-post-article
|
||||
|
||||
# Default target
|
||||
all: build
|
||||
|
||||
# Build all Lambda functions
|
||||
build:
|
||||
@for bin in $(BINARIES); do \
|
||||
$(CARGO) lambda build --release --bin $$bin --output-format zip; \
|
||||
done
|
||||
|
||||
# Clean up build
|
||||
clean:
|
||||
$(CARGO) clean
|
||||
|
||||
start-emulator:
|
||||
$(CARGO) lambda start &
|
||||
|
||||
|
||||
run-post:
|
||||
$(CARGO) lambda invoke lambda-post-article --data-file events/post.json
|
||||
|
||||
dev-run-post: start-emulator run-post
|
||||
|
||||
25
reference/writeonce-api/README.md
Normal file
25
reference/writeonce-api/README.md
Normal file
|
|
@ -0,0 +1,25 @@
|
|||
# Manage Articles API
|
||||
|
||||
<img src="https://writeonce.de/assets/images/logo.png " width="200" alt="writeonce">
|
||||
|
||||
The **Manage Articles API** allows you to create, retrieve, and manage blog articles. This API is built to help with managing articles on a blog platform, allowing operations like fetching articles by title, pagination, and article creation.
|
||||
|
||||
## Base URL
|
||||
|
||||
api.writeonce.de
|
||||
|
||||
## Tech Stack
|
||||
|
||||
The **Manage Articles API** is built using the following technologies:
|
||||
|
||||
- **Rust**: The backend of the API is built with the Rust programming language for high performance and reliability.
|
||||
- **Diesel**: Diesel is used as the ORM (Object Relational Mapper) to interact with the database.
|
||||
- **PostgreSQL**: The database used for storing articles and related data.
|
||||
|
||||
## Cross build
|
||||
|
||||
CROSS_CONTAINER_ENGINE=docker cross build \
|
||||
--bin writeonce-manage-article-api \
|
||||
--release \
|
||||
--target aarch64-unknown-linux-gnu \
|
||||
--verbose
|
||||
9
reference/writeonce-api/diesel.toml
Normal file
9
reference/writeonce-api/diesel.toml
Normal file
|
|
@ -0,0 +1,9 @@
|
|||
# For documentation on how to configure this file,
|
||||
# see https://diesel.rs/guides/configuring-diesel-cli
|
||||
|
||||
[print_schema]
|
||||
file = "src/schema.rs"
|
||||
custom_type_derives = ["diesel::query_builder::QueryId", "Clone"]
|
||||
|
||||
[migrations_directory]
|
||||
dir = "/home/shoney/manage-articles-api/migrations"
|
||||
42
reference/writeonce-api/doc/index.html
Normal file
42
reference/writeonce-api/doc/index.html
Normal file
|
|
@ -0,0 +1,42 @@
|
|||
<!DOCTYPE html>
|
||||
<html lang="en">
|
||||
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
||||
<title>Swagger UI</title>
|
||||
<link rel="stylesheet" type="text/css"
|
||||
href="https://cdnjs.cloudflare.com/ajax/libs/swagger-ui/4.5.0/swagger-ui.css">
|
||||
<script src="https://cdnjs.cloudflare.com/ajax/libs/swagger-ui/4.5.0/swagger-ui-bundle.js"></script>
|
||||
</head>
|
||||
|
||||
<body>
|
||||
<div id="swagger-ui"></div>
|
||||
<script>
|
||||
const ui = SwaggerUIBundle({
|
||||
url: "./open-api-spec", // Path to the OpenAPI spec
|
||||
dom_id: '#swagger-ui',
|
||||
deepLinking: true,
|
||||
presets: [
|
||||
SwaggerUIBundle.presets.apis,
|
||||
SwaggerUIBundle.presets.topbar
|
||||
],
|
||||
layout: "BaseLayout",
|
||||
requestInterceptor: (request) => {
|
||||
const token = localStorage.getItem("auth_token"); // Optionally save token in localStorage
|
||||
if (token) {
|
||||
request.headers['Authorization'] = 'Bearer ' + token;
|
||||
}
|
||||
return request;
|
||||
},
|
||||
// Add an "Authorize" button to handle token input
|
||||
authActions: {
|
||||
authorize: (authorize) => {
|
||||
// You can handle custom logic for the "Authorize" button here if needed
|
||||
}
|
||||
}
|
||||
});
|
||||
</script>
|
||||
</body>
|
||||
|
||||
</html>
|
||||
520
reference/writeonce-api/doc/openapi.json
Normal file
520
reference/writeonce-api/doc/openapi.json
Normal file
|
|
@ -0,0 +1,520 @@
|
|||
{
|
||||
"openapi": "3.0.3",
|
||||
"info": {
|
||||
"title": "Manage Articles API",
|
||||
"termsOfService": "",
|
||||
"contact": {
|
||||
"email": "shoney.john@outlook.com"
|
||||
},
|
||||
"host": "api.writeonce.de",
|
||||
"version": "0.0.1",
|
||||
"summary": "Blog manage API",
|
||||
"description": "🚀 **WriteOnce API Server**\n\nSource code available on [GitLab](https://git.writeonce.de/shoney/writeonce-manage-article-api).\n\n🔐 **Authorization:**\nYou can use the API key `REDACTED_TOKEN` to test the authorization filters (except for POST requests).\n\n📦 **S3 + AWS Lambda Integration:**\nWhen JSON documents are saved to an S3 bucket, an AWS Lambda function is triggered automatically. This function then makes a POST request to this API endpoint to create a new article.\n\n💡 View the [AWS Lambda function code](https://git.writeonce.de/shoney/writeonce-function)."
|
||||
},
|
||||
"security": [
|
||||
{
|
||||
"BearerAuth": []
|
||||
}
|
||||
],
|
||||
"paths": {
|
||||
"/article/title/{sys_title}": {
|
||||
"get": {
|
||||
"summary": "Fetch Articles by page route blog title name",
|
||||
"operationId": "getArticleByTitle",
|
||||
"security": [
|
||||
{
|
||||
"BearerAuth": []
|
||||
}
|
||||
],
|
||||
"parameters": [
|
||||
{
|
||||
"name": "sys_title",
|
||||
"in": "path",
|
||||
"required": true,
|
||||
"description": "Specify the system title of the article",
|
||||
"schema": {
|
||||
"type": "string",
|
||||
"example": "gitlab-runner-with-kubernetes-executor"
|
||||
}
|
||||
}
|
||||
],
|
||||
"responses": {
|
||||
"200": {
|
||||
"description": "Articles found",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/Article"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"404": {
|
||||
"description": "Article not found",
|
||||
"content": {
|
||||
"text/plain": {
|
||||
"example": "Article not found"
|
||||
}
|
||||
}
|
||||
},
|
||||
"401": {
|
||||
"description": "Unauthorized",
|
||||
"content": {
|
||||
"text/plain": {
|
||||
"example": "Unauthorized"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"/articles/{skip}/{limit}": {
|
||||
"get": {
|
||||
"summary": "Fetch Articles Pagination",
|
||||
"operationId": "getArticleByPage",
|
||||
"security": [
|
||||
{
|
||||
"BearerAuth": []
|
||||
}
|
||||
],
|
||||
"parameters": [
|
||||
{
|
||||
"name": "skip",
|
||||
"in": "path",
|
||||
"required": true,
|
||||
"description": "The lower range of Article pages",
|
||||
"schema": {
|
||||
"type": "integer",
|
||||
"example": 0
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "limit",
|
||||
"in": "path",
|
||||
"required": true,
|
||||
"description": "The limit range of Article pages",
|
||||
"schema": {
|
||||
"type": "integer",
|
||||
"example": 0
|
||||
}
|
||||
}
|
||||
],
|
||||
"responses": {
|
||||
"200": {
|
||||
"description": "Articles found",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/Article"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"404": {
|
||||
"description": "Article not found",
|
||||
"content": {
|
||||
"text/plain": {
|
||||
"example": "Article not found"
|
||||
}
|
||||
}
|
||||
},
|
||||
"401": {
|
||||
"description": "Unauthorized",
|
||||
"content": {
|
||||
"text/plain": {
|
||||
"example": "Unauthorized"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"/article/{id}": {
|
||||
"get": {
|
||||
"summary": "Get article by ID",
|
||||
"operationId": "getArticleById",
|
||||
"security": [
|
||||
{
|
||||
"BearerAuth": []
|
||||
}
|
||||
],
|
||||
"parameters": [
|
||||
{
|
||||
"name": "id",
|
||||
"in": "path",
|
||||
"required": true,
|
||||
"description": "The ID of the article to retrieve.",
|
||||
"schema": {
|
||||
"type": "integer",
|
||||
"example": 1
|
||||
}
|
||||
}
|
||||
],
|
||||
"responses": {
|
||||
"200": {
|
||||
"description": "Article found",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/Article"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"404": {
|
||||
"description": "Article not found",
|
||||
"content": {
|
||||
"text/plain": {
|
||||
"example": "Article not found"
|
||||
}
|
||||
}
|
||||
},
|
||||
"401": {
|
||||
"description": "Unauthorized",
|
||||
"content": {
|
||||
"text/plain": {
|
||||
"example": "Unauthorized"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"/article/": {
|
||||
"post": {
|
||||
"summary": "Create a new Article",
|
||||
"operationId": "saveArticle",
|
||||
"requestBody": {
|
||||
"description": "Save a new article",
|
||||
"required": true,
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"title": {
|
||||
"type": "string",
|
||||
"example": "Gitlab Runner with Kubernetes Executor"
|
||||
},
|
||||
"sys_title": {
|
||||
"type": "string",
|
||||
"example": "gitlab-runner-with-kubernetes-executor"
|
||||
},
|
||||
"published": {
|
||||
"type": "boolean",
|
||||
"example": true
|
||||
},
|
||||
"content": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"author": {
|
||||
"type": "string",
|
||||
"example": "Shoney Arickathil"
|
||||
},
|
||||
"content": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"codes": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"language": {
|
||||
"type": "string",
|
||||
"example": "yaml"
|
||||
},
|
||||
"paragraphIndex": {
|
||||
"type": "integer",
|
||||
"example": 0
|
||||
},
|
||||
"sectionIndex": {
|
||||
"type": "integer",
|
||||
"example": 1
|
||||
},
|
||||
"snippet": {
|
||||
"type": "string",
|
||||
"example": "gitlab-runnergitlab-runner-deploy/gitlab-runner-with-kubernetes-executor"
|
||||
},
|
||||
"title": {
|
||||
"type": "string",
|
||||
"example": "gitlab runner deploy"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"images": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"properties": {}
|
||||
}
|
||||
},
|
||||
"img": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"caption": {
|
||||
"type": "string",
|
||||
"example": "Gitlab runner with kubernetes executor."
|
||||
},
|
||||
"path": {
|
||||
"type": "string",
|
||||
"example": "https://writeonce-filecloud-bucket.s3.eu-central-1.amazonaws.com/images/kubernetes.png"
|
||||
}
|
||||
}
|
||||
},
|
||||
"sections": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"heading": {
|
||||
"type": "string",
|
||||
"example": "Introduction"
|
||||
},
|
||||
"paragraphs": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "string",
|
||||
"example": "Deploying a Gitlab runner using kubernetes is a great option to overcome the limitations of other gitlab runner executor such as docker and docker machine."
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"publishedOn": {
|
||||
"type": "integer",
|
||||
"example": 1737414886
|
||||
},
|
||||
"references": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"dateAccessed": {
|
||||
"type": "integer",
|
||||
"example": 1737414886
|
||||
},
|
||||
"title": {
|
||||
"type": "string",
|
||||
"example": "Gitlab Kubernetes executor"
|
||||
},
|
||||
"url": {
|
||||
"type": "string",
|
||||
"example": "https://docs.gitlab.com/runner/executors/kubernetes/"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"systitle": {
|
||||
"type": "string",
|
||||
"example": "gitlab-runner-with-kubernetes-executor"
|
||||
},
|
||||
"tags": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "string",
|
||||
"example": ["Kubernetes", "Gitlab runner"]
|
||||
}
|
||||
},
|
||||
"title": {
|
||||
"type": "string",
|
||||
"example": "Gitlab Runner with Kubernetes Executor"
|
||||
}
|
||||
}
|
||||
},
|
||||
"do_aws_sync": {
|
||||
"type": "boolean",
|
||||
"example": false
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"responses": {
|
||||
"201": {
|
||||
"description": "created successfully",
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"message": {
|
||||
"type": "string",
|
||||
"example": ""
|
||||
},
|
||||
"articleId": {
|
||||
"type": "integer",
|
||||
"example": "2"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"400": {
|
||||
"description": "Bad request"
|
||||
},
|
||||
"500": {
|
||||
"description": "Internal server error"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"components": {
|
||||
"securitySchemes": {
|
||||
"BearerAuth": {
|
||||
"type": "http",
|
||||
"scheme": "bearer",
|
||||
"bearerFormat": "JWT"
|
||||
}
|
||||
},
|
||||
"schemas": {
|
||||
"Article": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"title": {
|
||||
"type": "string",
|
||||
"example": "Gitlab Runner with Kubernetes Executor"
|
||||
},
|
||||
"sys_title": {
|
||||
"type": "string",
|
||||
"example": "gitlab-runner-with-kubernetes-executor"
|
||||
},
|
||||
"published": {
|
||||
"type": "boolean",
|
||||
"example": true
|
||||
},
|
||||
"content": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"author": {
|
||||
"type": "string",
|
||||
"example": "Shoney Arickathil"
|
||||
},
|
||||
"content": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"codes": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"language": {
|
||||
"type": "string",
|
||||
"example": "yaml"
|
||||
},
|
||||
"paragraphIndex": {
|
||||
"type": "integer",
|
||||
"example": 0
|
||||
},
|
||||
"sectionIndex": {
|
||||
"type": "integer",
|
||||
"example": 1
|
||||
},
|
||||
"snippet": {
|
||||
"type": "string",
|
||||
"example": "gitlab-runnergitlab-runner-deploy/gitlab-runner-with-kubernetes-executor"
|
||||
},
|
||||
"title": {
|
||||
"type": "string",
|
||||
"example": "gitlab runner deploy"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"images": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"properties": {}
|
||||
}
|
||||
},
|
||||
"img": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"caption": {
|
||||
"type": "string",
|
||||
"example": "Gitlab runner with kubernetes executor."
|
||||
},
|
||||
"path": {
|
||||
"type": "string",
|
||||
"example": "https://writeonce-filecloud-bucket.s3.eu-central-1.amazonaws.com/images/kubernetes.png"
|
||||
}
|
||||
}
|
||||
},
|
||||
"sections": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"heading": {
|
||||
"type": "string",
|
||||
"example": "Introduction"
|
||||
},
|
||||
"paragraphs": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "string",
|
||||
"example": "Deploying a Gitlab runner using kubernetes is a great option to overcome the limitations of other gitlab runner executor such as docker and docker machine."
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"publishedOn": {
|
||||
"type": "integer",
|
||||
"example": 1737414886
|
||||
},
|
||||
"references": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"dateAccessed": {
|
||||
"type": "integer",
|
||||
"example": 1737414886
|
||||
},
|
||||
"title": {
|
||||
"type": "string",
|
||||
"example": "Gitlab Kubernetes executor"
|
||||
},
|
||||
"url": {
|
||||
"type": "string",
|
||||
"example": "https://docs.gitlab.com/runner/executors/kubernetes/"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"systitle": {
|
||||
"type": "string",
|
||||
"example": "gitlab-runner-with-kubernetes-executor"
|
||||
},
|
||||
"tags": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "string",
|
||||
"example": ["Kubernetes", "Gitlab runner"]
|
||||
}
|
||||
},
|
||||
"title": {
|
||||
"type": "string",
|
||||
"example": "Gitlab Runner with Kubernetes Executor"
|
||||
}
|
||||
}
|
||||
},
|
||||
"do_aws_sync": {
|
||||
"type": "boolean",
|
||||
"example": false
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
53
reference/writeonce-api/docker-compose.yml
Normal file
53
reference/writeonce-api/docker-compose.yml
Normal file
|
|
@ -0,0 +1,53 @@
|
|||
services:
|
||||
writeonce-postgress-db:
|
||||
image: postgres:17.0
|
||||
|
||||
container_name: writeonce-postgress-db
|
||||
restart: always
|
||||
# set shared memory limit when using docker-compose
|
||||
shm_size: 128mb
|
||||
# or set shared memory limit when deploy via swarm stack
|
||||
#volumes:
|
||||
# - type: tmpfs
|
||||
# target: /dev/shm
|
||||
# tmpfs:
|
||||
# size: 134217728 # 128*2^20 bytes = 128Mb
|
||||
environment:
|
||||
POSTGRES_USER: ${POSTGRES_USER}
|
||||
POSTGRES_PASSWORD: ${POSTGRES_PASSWORD}
|
||||
POSTGRES_DB: ${POSTGRES_DB}
|
||||
DB_USER: ${DB_USER}
|
||||
DB_PASSWORD: ${DB_PASSWORD}
|
||||
DB_NAME: ${DB_NAME}
|
||||
volumes:
|
||||
- ./postgress/:/docker-entrypoint-initdb.d
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pg_isready -U ${POSTGRES_USER}"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
networks:
|
||||
- writeonce-network
|
||||
|
||||
writeonce-manage-article-api:
|
||||
image: registry.writeonce.de/shoney/writeonce-manage-article-api
|
||||
|
||||
container_name: writeonce-manage-article-api
|
||||
ports:
|
||||
- ${PORT}:${PORT}
|
||||
depends_on:
|
||||
writeonce-postgress-db:
|
||||
condition: service_healthy
|
||||
environment:
|
||||
- AWS_INFRA_BASE_URL=${AWS_INFRA_BASE_URL}
|
||||
- PORT=${PORT}
|
||||
- API_ACCESS_TOKEN=${API_ACCESS_TOKEN}
|
||||
- API_ACCESS_ADMIN_TOKEN=${API_ACCESS_ADMIN_TOKEN}
|
||||
- DATABASE_URL=${DATABASE_URL}
|
||||
restart: always
|
||||
networks:
|
||||
- writeonce-network
|
||||
|
||||
networks:
|
||||
writeonce-network:
|
||||
external: true
|
||||
51
reference/writeonce-api/events/post.json
Normal file
51
reference/writeonce-api/events/post.json
Normal file
|
|
@ -0,0 +1,51 @@
|
|||
{
|
||||
"content": {
|
||||
"author": "Shoney Arickathil",
|
||||
"content": {
|
||||
"codes": [
|
||||
{
|
||||
"language": "yaml",
|
||||
"paragraphIndex": 0,
|
||||
"sectionIndex": 1,
|
||||
"snippet": "gitlab-runnergitlab-runner-deploy/gitlab-runner-with-kubernetes-executor",
|
||||
"title": "gitlab runner deploy"
|
||||
}
|
||||
],
|
||||
"images": [
|
||||
{}
|
||||
],
|
||||
"img": {
|
||||
"caption": "Gitlab runner with kubernetes executor.",
|
||||
"path": "https://writeonce-filecloud-bucket.s3.eu-central-1.amazonaws.com/images/kubernetes.png"
|
||||
},
|
||||
"sections": [
|
||||
{
|
||||
"heading": "Introduction",
|
||||
"paragraphs": [
|
||||
"Deploying a Gitlab runner using kubernetes is a great option to overcome the limitations of other gitlab runner executor such as docker and docker machine."
|
||||
]
|
||||
}
|
||||
]
|
||||
},
|
||||
"publishedOn": 1737414886,
|
||||
"references": [
|
||||
{
|
||||
"dateAccessed": 1737414886,
|
||||
"title": "Gitlab Kubernetes executor",
|
||||
"url": "https://docs.gitlab.com/runner/executors/kubernetes/"
|
||||
}
|
||||
],
|
||||
"systitle": "gitlab-runner-with-kubernetes-executor",
|
||||
"tags": [
|
||||
[
|
||||
"Kubernetes",
|
||||
"Gitlab runner"
|
||||
]
|
||||
],
|
||||
"title": "Gitlab Runner with Kubernetes Executor"
|
||||
},
|
||||
"do_aws_sync": false,
|
||||
"published": true,
|
||||
"sys_title": "gitlab-runner-with-kubernetes-executor",
|
||||
"title": "Gitlab Runner with Kubernetes Executor"
|
||||
}
|
||||
16
reference/writeonce-api/infra/00-build-image.sh
Normal file
16
reference/writeonce-api/infra/00-build-image.sh
Normal file
|
|
@ -0,0 +1,16 @@
|
|||
#!/bin/sh
|
||||
|
||||
if [ -z "$API_PORT" ] || [ -z "$AWS_INFRA_BASE_URL" ] || [ -z "$DATABASE_URL" ] || [ -z "$API_ACCESS_TOKEN" ] || [ -z "$API_ACCESS_ADMIN_TOKEN" ]; then
|
||||
echo "Error: Missing required environment variables"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo PORT="$API_PORT"
|
||||
|
||||
docker build \
|
||||
--build-arg PORT="$API_PORT" \
|
||||
--build-arg AWS_INFRA_BASE_URL="$AWS_INFRA_BASE_URL" \
|
||||
--build-arg DATABASE_URL="$DATABASE_URL" \
|
||||
--build-arg API_ACCESS_TOKEN="$API_ACCESS_TOKEN" \
|
||||
--build-arg API_ACCESS_ADMIN_TOKEN="$API_ACCESS_ADMIN_TOKEN" \
|
||||
-t "$CI_REGISTRY_IMAGE:$CI_COMMIT_SHORT_SHA" .
|
||||
3
reference/writeonce-api/infra/02-docker-login.sh
Normal file
3
reference/writeonce-api/infra/02-docker-login.sh
Normal file
|
|
@ -0,0 +1,3 @@
|
|||
#!/bin/sh
|
||||
|
||||
docker login -u $CI_REGISTRY_USER -p $CI_REGISTRY_PASSWORD $CI_REGISTRY
|
||||
2
reference/writeonce-api/infra/03-upload-image.sh
Normal file
2
reference/writeonce-api/infra/03-upload-image.sh
Normal file
|
|
@ -0,0 +1,2 @@
|
|||
#!/bin/sh
|
||||
docker push "$CI_REGISTRY_IMAGE:$CI_COMMIT_SHORT_SHA"
|
||||
Some files were not shown because too many files have changed in this diff Show more
Loading…
Reference in a new issue