From d0a4e19c05bf577030c07bf2e33021c61beccb55 Mon Sep 17 00:00:00 2001 From: alangou Date: Fri, 2 Oct 2026 21:04:23 +0000 Subject: [PATCH 01/13] fix(deps): upgrade ratatui to resolve lru vulnerability (#4131) Signed-off-by: Adrien Langou --- .agents/skills/tui-development/SKILL.md | 2 +- Cargo.lock | 924 +++++++++++++++----- Cargo.toml | 2 +- crates/openshell-tui/src/app.rs | 2 +- crates/openshell-tui/src/ui/mod.rs | 26 +- crates/openshell-tui/src/ui/sandbox_logs.rs | 2 +- 6 files changed, 741 insertions(+), 217 deletions(-) diff --git a/.agents/skills/tui-development/SKILL.md b/.agents/skills/tui-development/SKILL.md index 832bb75144..b0d546be0b 100644 --- a/.agents/skills/tui-development/SKILL.md +++ b/.agents/skills/tui-development/SKILL.md @@ -16,7 +16,7 @@ The OpenShell TUI is a ratatui-based terminal UI for the OpenShell platform. It - **Launched via:** `openshell term` or `mise run term` - **Crate:** `crates/openshell-tui/` - **Key dependencies:** - - `ratatui` (workspace version) — uses `frame.size()` (not `frame.area()`) + - `ratatui` (workspace version) — uses `frame.area()` for the drawable terminal area - `crossterm` (workspace version) — terminal backend and event polling - `tonic` with TLS — gRPC client for the OpenShell gateway - `tokio` — async runtime for event loop, spawned tasks, and mpsc channels diff --git a/Cargo.lock b/Cargo.lock index fc0b173702..180617e40d 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -158,6 +158,15 @@ dependencies = [ "thiserror 2.0.20", ] +[[package]] +name = "approx" +version = "0.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cab112f0a86d568ea0e627cc1d6be74a1e9cd55214684db5561995f6dad897c6" +dependencies = [ + "num-traits", +] + [[package]] name = "arc-swap" version = "1.9.1" @@ -282,6 +291,15 @@ dependencies = [ "num-traits", ] +[[package]] +name = "atomic" +version = "0.6.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a89cbf775b137e9b968e67227ef7f775587cde3fd31b0d8599dbd0f598a48340" +dependencies = [ + "bytemuck", +] + [[package]] name = "atomic-waker" version = "1.1.2" @@ -770,6 +788,27 @@ dependencies = [ "sha2 0.11.0", ] +[[package]] +name = "bit-set" +version = "0.5.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0700ddab506f33b20a03b13996eccd309a48e5ff77d0d95926aa0210fb4e95f1" +dependencies = [ + "bit-vec", +] + +[[package]] +name = "bit-vec" +version = "0.6.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "349f9b6a179ed607305526ca489b34ad0a41aed5f7980fa90eb03160b69598fb" + +[[package]] +name = "bitflags" +version = "1.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bef38d45163c2f1dde094a7dfd33ccf595c92905c8f8f4fdc18d06fb1037718a" + [[package]] name = "bitflags" version = "2.13.2" @@ -886,6 +925,18 @@ version = "3.20.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "5d20789868f4b01b2f2caec9f5c4e0213b41e3e5702a50157d699ae31ced2fcb" +[[package]] +name = "by_address" +version = "1.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "64fa3c856b712db6612c019f14756e64e4bcea13337a6b33b696333a9eaa2d06" + +[[package]] +name = "bytemuck" +version = "1.25.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "95832e849adfb21180ccb6826a99da14e5d266ae5c2e668e1602cf234f153797" + [[package]] name = "byteorder" version = "1.5.0" @@ -917,12 +968,6 @@ dependencies = [ "libbz2-rs-sys", ] -[[package]] -name = "cassowary" -version = "0.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "df8670b8c7b9dae1793364eafadf7239c40d669904660c5960d74cfd80b46a53" - [[package]] name = "castaway" version = "0.2.4" @@ -1095,13 +1140,14 @@ dependencies = [ [[package]] name = "compact_str" -version = "0.7.1" +version = "0.9.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f86b9c4c00838774a6d902ef931eff7470720c51d90c2e32cfe15dc304737b3f" +checksum = "9dfdd1c2274d9aa354115b09dc9a901d6c5576818cdf70d14cae2bdb47df00ab" dependencies = [ "castaway", "cfg-if", "itoa", + "rustversion", "ryu", "static_assertions", ] @@ -1167,6 +1213,15 @@ version = "0.4.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "3d52eff69cd5e647efe296129160853a42795992097e8af39800e1060caeea9b" +[[package]] +name = "convert_case" +version = "0.10.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "633458d4ef8c78b72454de2d54fd6ab2e60f9e02be22f3c6104cdc8a4e0fceb9" +dependencies = [ + "unicode-segmentation", +] + [[package]] name = "core-foundation" version = "0.10.1" @@ -1278,15 +1333,15 @@ checksum = "d0a5c400df2834b80a4c3327b3aad3a4c4cd4de0629063962b03235697506a28" [[package]] name = "crossterm" -version = "0.27.0" +version = "0.28.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f476fe445d41c9e991fd07515a6f463074b782242ccf4a5b7b1d1012e70824df" +checksum = "829d955a0bb380ef178a640b91779e3987da38c9aea133b20614cfed8cdea9c6" dependencies = [ - "bitflags", + "bitflags 2.13.2", "crossterm_winapi", - "libc", - "mio 0.8.11", + "mio", "parking_lot", + "rustix 0.38.44", "signal-hook", "signal-hook-mio", "winapi", @@ -1294,15 +1349,17 @@ dependencies = [ [[package]] name = "crossterm" -version = "0.28.1" +version = "0.29.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "829d955a0bb380ef178a640b91779e3987da38c9aea133b20614cfed8cdea9c6" +checksum = "d8b9f2e4c67f833b660cdb0a3523065869fb35570177239812ed4c905aeff87b" dependencies = [ - "bitflags", + "bitflags 2.13.2", "crossterm_winapi", - "mio 1.2.0", + "derive_more", + "document-features", + "mio", "parking_lot", - "rustix 0.38.44", + "rustix 1.1.4", "signal-hook", "signal-hook-mio", "winapi", @@ -1377,6 +1434,16 @@ dependencies = [ "rand_core 0.10.1", ] +[[package]] +name = "csscolorparser" +version = "0.6.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "eb2a7d3066da2de787b7f032c736763eb7ae5d355f81a68bab2675a96008b0bf" +dependencies = [ + "lab", + "phf", +] + [[package]] name = "ctr" version = "0.10.1" @@ -1446,8 +1513,18 @@ version = "0.20.11" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "fc7f46116c46ff9ab3eb1597a45688b6715c6e628b5c133e288e709a29bcb4ee" dependencies = [ - "darling_core", - "darling_macro", + "darling_core 0.20.11", + "darling_macro 0.20.11", +] + +[[package]] +name = "darling" +version = "0.24.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ed17f5901b6630b993ca003def43f2f8ef4014fc13b047b57aad617ff32bc2ec" +dependencies = [ + "darling_core 0.24.1", + "darling_macro 0.24.1", ] [[package]] @@ -1464,17 +1541,41 @@ dependencies = [ "syn 2.0.117", ] +[[package]] +name = "darling_core" +version = "0.24.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6837e2cf7485aaae18f86181d2f0e9a7ed297a025e220aeabf63fdebd3a2ddff" +dependencies = [ + "ident_case", + "proc-macro2", + "quote", + "strsim", + "syn 3.0.5", +] + [[package]] name = "darling_macro" version = "0.20.11" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "fc34b93ccb385b40dc71c6fceac4b2ad23662c7eeb248cf10d529b7e055b6ead" dependencies = [ - "darling_core", + "darling_core 0.20.11", "quote", "syn 2.0.117", ] +[[package]] +name = "darling_macro" +version = "0.24.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2ac7135c3ef02b2f7833bbeb1be5ba7f966dcde8a87c6b87f65a778d71a02785" +dependencies = [ + "darling_core 0.24.1", + "quote", + "syn 3.0.5", +] + [[package]] name = "dashmap" version = "6.2.1" @@ -1530,6 +1631,12 @@ dependencies = [ "syn 2.0.117", ] +[[package]] +name = "deltae" +version = "0.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5729f5117e208430e437df2f4843f5e5952997175992d1414f94c57d61e270b4" + [[package]] name = "der" version = "0.7.10" @@ -1590,7 +1697,7 @@ version = "0.20.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "2d5bcf7b024d6835cfb3d473887cd966994907effbe9227e8c8219824d06c4e8" dependencies = [ - "darling", + "darling 0.20.11", "proc-macro2", "quote", "syn 2.0.117", @@ -1606,6 +1713,28 @@ dependencies = [ "syn 2.0.117", ] +[[package]] +name = "derive_more" +version = "2.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d751e9e49156b02b44f9c1815bcb94b984cdcc4396ecc32521c739452808b134" +dependencies = [ + "derive_more-impl", +] + +[[package]] +name = "derive_more-impl" +version = "2.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "799a97264921d8623a957f6c3b9011f3b5492f557bbb7a5a19b7fa6d06ba8dcb" +dependencies = [ + "convert_case", + "proc-macro2", + "quote", + "rustc_version", + "syn 2.0.117", +] + [[package]] name = "des" version = "0.9.0" @@ -1664,6 +1793,15 @@ dependencies = [ "syn 2.0.117", ] +[[package]] +name = "document-features" +version = "0.2.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d4b8a88685455ed29a21542a33abd9cb6510b6b129abadabdcef0f4c55bc8f61" +dependencies = [ + "litrs", +] + [[package]] name = "dotenvy" version = "0.15.7" @@ -1909,6 +2047,15 @@ dependencies = [ "windows-sys 0.61.2", ] +[[package]] +name = "euclid" +version = "0.22.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f1a05365e3b1c6d1650318537c7460c6923f1abdd272ad6842baa2b509957a06" +dependencies = [ + "num-traits", +] + [[package]] name = "event-listener" version = "5.4.1" @@ -1930,6 +2077,16 @@ dependencies = [ "pin-project-lite", ] +[[package]] +name = "fancy-regex" +version = "0.11.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b95f7c0680e4142284cf8b22c14a476e87d61b004a3a0861872b32ef7ead40a2" +dependencies = [ + "bit-set", + "regex", +] + [[package]] name = "fastrand" version = "2.4.1" @@ -1968,6 +2125,17 @@ version = "0.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "64cd1e32ddd350061ae6edb1b082d7c54915b5c672c389143b9a63403a109f24" +[[package]] +name = "filedescriptor" +version = "0.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e40758ed24c9b2eeb76c35fb0aebc66c626084edd827e07e1552279814c6682d" +dependencies = [ + "libc", + "thiserror 1.0.69", + "winapi", +] + [[package]] name = "filetime" version = "0.2.27" @@ -1985,6 +2153,18 @@ version = "0.1.9" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "5baebc0774151f905a1a2cc41989300b1e6fbb29aff0ceffa1064fdd3088d582" +[[package]] +name = "finl_unicode" +version = "1.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "80bb028c8b4148c9ee0cca68fcd9add6044e81d3619f48577ddf13a263d047a2" + +[[package]] +name = "fixedbitset" +version = "0.4.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0ce7134b9999ecaf8bcd65542e436736ef32ddca1b3e06094cb6ec5755203b80" + [[package]] name = "fixedbitset" version = "0.5.7" @@ -2360,6 +2540,11 @@ name = "hashbrown" version = "0.17.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "4f467dd6dccf739c208452f8014c75c18bb8301b050ad1cfb27153803edb0f51" +dependencies = [ + "allocator-api2", + "equivalent", + "foldhash 0.2.0", +] [[package]] name = "hashlink" @@ -2889,13 +3074,22 @@ dependencies = [ "web-time", ] +[[package]] +name = "indoc" +version = "2.0.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "79cf5c93f93228cf8efb3ba362535fb11199ac548a09ce117c9b1adc3030d706" +dependencies = [ + "rustversion", +] + [[package]] name = "inotify" version = "0.11.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "533e68a5842e734946fe159fb03fc9bbbb254f590dd0d8ad321ae5ff7beca2c1" dependencies = [ - "bitflags", + "bitflags 2.13.2", "inotify-sys", "libc", ] @@ -2919,6 +3113,19 @@ dependencies = [ "hybrid-array", ] +[[package]] +name = "instability" +version = "0.3.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4c3b5acc1e2fd9375041a388da33d1eb8aed5f7a8c0dd3543e3ea2805adfbe20" +dependencies = [ + "darling 0.24.1", + "indoc", + "proc-macro2", + "quote", + "syn 3.0.5", +] + [[package]] name = "ipnet" version = "2.12.0" @@ -2956,24 +3163,6 @@ version = "1.70.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "a6cb138bb79a146c1bd460005623e142ef0181e3d0219cb493e02f7d08a35695" -[[package]] -name = "itertools" -version = "0.12.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ba291022dbbd398a455acf126c1e341954079855bc60dfdda641363bd6922569" -dependencies = [ - "either", -] - -[[package]] -name = "itertools" -version = "0.13.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "413ee7dfc52ee1a4949ceeb7dbc8a33f2d6c088194d9f922fb8318faf1f01186" -dependencies = [ - "either", -] - [[package]] name = "itertools" version = "0.14.0" @@ -3167,6 +3356,17 @@ dependencies = [ "serde_json", ] +[[package]] +name = "kasuari" +version = "0.4.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bde5057d6143cc94e861d90f591b9303d6716c6b9602309150bd068853c10899" +dependencies = [ + "hashbrown 0.16.1", + "portable-atomic", + "thiserror 2.0.20", +] + [[package]] name = "keccak" version = "0.2.0" @@ -3218,7 +3418,7 @@ version = "1.1.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "07293a4e297ac234359b510362495713f75ea345d5307140414f20c69ffeb087" dependencies = [ - "bitflags", + "bitflags 2.13.2", "libc", ] @@ -3296,7 +3496,7 @@ version = "0.99.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "c562f58dc9f7ca5feac8a6ee5850ca221edd6f04ce0dd2ee873202a88cd494c9" dependencies = [ - "darling", + "darling 0.20.11", "proc-macro2", "quote", "serde", @@ -3332,6 +3532,12 @@ dependencies = [ "tracing", ] +[[package]] +name = "lab" +version = "0.11.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bf36173d4167ed999940f804952e6b08197cae5ad5d572eb4db150ce8ad5d58f" + [[package]] name = "landlock" version = "0.4.4" @@ -3392,7 +3598,7 @@ version = "0.1.16" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e02f3bb43d335493c96bf3fd3a321600bf6bd07ed34bc64118e9293bdffea46c" dependencies = [ - "bitflags", + "bitflags 2.13.2", "libc", "plain", "redox_syscall 0.7.4", @@ -3409,6 +3615,15 @@ dependencies = [ "vcpkg", ] +[[package]] +name = "line-clipping" +version = "0.3.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e752191d037c44ad111a8caa762921926658402f01cc1253f7bef2020ece4f5e" +dependencies = [ + "bitflags 2.13.2", +] + [[package]] name = "linux-raw-sys" version = "0.4.15" @@ -3427,6 +3642,12 @@ version = "0.8.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "92daf443525c4cce67b150400bc2316076100ce0b3686209eb8cf3c31612e6f0" +[[package]] +name = "litrs" +version = "1.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "11d3d7f243d5c5a8b9bb5d6dd2b1602c0cb0b9db1621bafc7ed66e35ff9fe092" + [[package]] name = "lock_api" version = "0.4.14" @@ -3444,11 +3665,11 @@ checksum = "f9f8bd3e56ce4dfc153cf470fffbfa98c7620958b312ca5c3a4b8d5181fd13c6" [[package]] name = "lru" -version = "0.12.5" +version = "0.18.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "234cf4f4a04dc1f57e24b96cc0cd600cf2af460d4161ac5ecdd0af8e1f3b2a38" +checksum = "ef9ac18847474e638e3702b76c65d4eb93428471a74778ef0f1be711717f89b5" dependencies = [ - "hashbrown 0.15.5", + "hashbrown 0.17.0", ] [[package]] @@ -3466,6 +3687,16 @@ dependencies = [ "sha2 0.11.0", ] +[[package]] +name = "mac_address" +version = "1.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b367b50a9be9a5d3b5718c1da74e5d6b9c17647f89752ee2562a470cc3c4eb1a" +dependencies = [ + "nix 0.30.1", + "windows-sys 0.61.2", +] + [[package]] name = "matchers" version = "0.2.0" @@ -3503,6 +3734,21 @@ version = "2.8.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f8ca58f447f06ed17d5fc4043ce1b10dd205e060fb3ce5b979b8ed8e59ff3f79" +[[package]] +name = "memmem" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a64a92489e2744ce060c349162be1c5f33c6969234104dbd99ddb5feb08b8c15" + +[[package]] +name = "memoffset" +version = "0.9.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "488016bfae457b036d996092f6cb448677611ce4449e970ceaf42695203f218a" +dependencies = [ + "autocfg", +] + [[package]] name = "metrics" version = "0.24.3" @@ -3601,18 +3847,6 @@ dependencies = [ "simd-adler32", ] -[[package]] -name = "mio" -version = "0.8.11" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a4a650543ca06a924e8b371db273b2756685faae30f8487da1b56505a8f78b0c" -dependencies = [ - "libc", - "log", - "wasi", - "windows-sys 0.48.0", -] - [[package]] name = "mio" version = "1.2.0" @@ -3671,10 +3905,23 @@ version = "0.29.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "71e2746dc3a24dd78b3cfcb7be93368c6de9963d30f43a6a73998a9cf4b17b46" dependencies = [ - "bitflags", + "bitflags 2.13.2", + "cfg-if", + "cfg_aliases", + "libc", +] + +[[package]] +name = "nix" +version = "0.30.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "74523f3a35e05aba87a1d978330aef40f67b0304ac79c1c00b294c9830543db6" +dependencies = [ + "bitflags 2.13.2", "cfg-if", "cfg_aliases", "libc", + "memoffset", ] [[package]] @@ -3683,7 +3930,7 @@ version = "0.31.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "cf20d2fde8ff38632c426f1165ed7436270b44f199fc55284c38276f9db47c3d" dependencies = [ - "bitflags", + "bitflags 2.13.2", "cfg-if", "cfg_aliases", "libc", @@ -3705,13 +3952,13 @@ version = "8.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "4d3d07927151ff8575b7087f245456e549fea62edf0ec4e565a5ee50c8402bc3" dependencies = [ - "bitflags", + "bitflags 2.13.2", "fsevent-sys", "inotify", "kqueue", "libc", "log", - "mio 1.2.0", + "mio", "notify-types", "walkdir", "windows-sys 0.60.2", @@ -3723,7 +3970,7 @@ version = "2.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "42b8cfee0e339a0337359f3c88165702ac6e600dc01c0cc9579a92d62b08477a" dependencies = [ - "bitflags", + "bitflags 2.13.2", ] [[package]] @@ -3795,6 +4042,17 @@ version = "0.2.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "c6673768db2d862beb9b39a78fdcb1a69439615d5794a1be50caa9bc92c81967" +[[package]] +name = "num-derive" +version = "0.4.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ed3955f1a9c7c0c15e092f9c887db08b1fc683305fdf6eb6684f22555355e202" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.117", +] + [[package]] name = "num-integer" version = "0.1.46" @@ -3836,12 +4094,21 @@ dependencies = [ ] [[package]] -name = "number_prefix" -version = "0.4.0" +name = "num_threads" +version = "0.1.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "830b246a0e5f20af87141b25c173cd1b609bd7779a4617d6ec582abaf90870f3" - -[[package]] +checksum = "5c7398b9c8b70908f6371f47ed36737907c87c52af34c268fed0bf0ceb92ead9" +dependencies = [ + "libc", +] + +[[package]] +name = "number_prefix" +version = "0.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "830b246a0e5f20af87141b25c173cd1b609bd7779a4617d6ec582abaf90870f3" + +[[package]] name = "oauth2" version = "5.0.0" source = "registry+https://github.com/rust-lang/crates.io-index" @@ -5006,6 +5273,15 @@ dependencies = [ "num-traits", ] +[[package]] +name = "ordered-float" +version = "4.6.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7bb71e1b3fa6ca1c61f383464aaf2bb0e2f8e772a1f01d486832464de363b951" +dependencies = [ + "num-traits", +] + [[package]] name = "outref" version = "0.5.2" @@ -5102,6 +5378,39 @@ dependencies = [ "windows-strings", ] +[[package]] +name = "palette" +version = "0.7.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ddeed8580d347d2abf3dcf06a5f0b3dc020258338526b277847cd4248a70fc64" +dependencies = [ + "approx", + "libm", + "palette_derive", + "palette_math", +] + +[[package]] +name = "palette_derive" +version = "0.7.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "88537020289b719d81be994ccf1bbf4990f477e2f69ee52fe3e45f43a02e56be" +dependencies = [ + "by_address", + "proc-macro2", + "quote", + "syn 2.0.117", +] + +[[package]] +name = "palette_math" +version = "0.7.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6e6eb142958d64335fb0e345c5b9ead2ecd6fc438c307e9d7d3c4fd428dbaf12" +dependencies = [ + "libm", +] + [[package]] name = "parking" version = "2.2.1" @@ -5140,12 +5449,6 @@ dependencies = [ "phc", ] -[[package]] -name = "paste" -version = "1.0.15" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "57c0d7b74b563b49d38dae00a0c37d4d6de9b432382b2892f0574ddcae73fd0a" - [[package]] name = "pbkdf2" version = "0.13.0" @@ -5239,7 +5542,7 @@ version = "0.8.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "8701b58ea97060d5e5b155d383a69952a60943f0e6dfe30b04c287beb0b27455" dependencies = [ - "fixedbitset", + "fixedbitset 0.5.7", "hashbrown 0.15.5", "indexmap", ] @@ -5252,7 +5555,7 @@ checksum = "9cd31dcfdbbd7431a807ef4df6edd6473228e94d5c805e8cf671227a21bad068" dependencies = [ "anyhow", "clap", - "itertools 0.14.0", + "itertools", "proc-macro2", "quote", "rand 0.8.6", @@ -5268,6 +5571,58 @@ dependencies = [ "ctutils", ] +[[package]] +name = "phf" +version = "0.11.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1fd6780a80ae0c52cc120a26a1a42c1ae51b247a253e4e06113d23d2c2edd078" +dependencies = [ + "phf_macros", + "phf_shared", +] + +[[package]] +name = "phf_codegen" +version = "0.11.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "aef8048c789fa5e851558d709946d6d79a8ff88c0440c587967f8e94bfb1216a" +dependencies = [ + "phf_generator", + "phf_shared", +] + +[[package]] +name = "phf_generator" +version = "0.11.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3c80231409c20246a13fddb31776fb942c38553c51e871f8cbd687a4cfb5843d" +dependencies = [ + "phf_shared", + "rand 0.8.6", +] + +[[package]] +name = "phf_macros" +version = "0.11.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f84ac04429c13a7ff43785d75ad27569f2951ce0ffd30a3321230db2fc727216" +dependencies = [ + "phf_generator", + "phf_shared", + "proc-macro2", + "quote", + "syn 2.0.117", +] + +[[package]] +name = "phf_shared" +version = "0.11.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "67eabc2ef2a60eb7faa00097bd1ffdb5bd28e62bf39990626a582201b7a754e5" +dependencies = [ + "siphasher", +] + [[package]] name = "pin-project" version = "1.1.11" @@ -5538,7 +5893,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "343d3bd7056eda839b03204e68deff7d1b13aba7af2b2fd16890697274262ee7" dependencies = [ "heck", - "itertools 0.14.0", + "itertools", "log", "multimap", "petgraph", @@ -5559,7 +5914,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "27c6023962132f4b30eb4c172c91ce92d933da334c59c23cddee82358ddafb0b" dependencies = [ "anyhow", - "itertools 0.14.0", + "itertools", "proc-macro2", "quote", "syn 2.0.117", @@ -5657,7 +6012,7 @@ version = "0.13.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7c3a14896dfa883796f1cb410461aef38810ea05f2b2c33c5aded3649095fdad" dependencies = [ - "bitflags", + "bitflags 2.13.2", "memchr", "unicase", ] @@ -5860,22 +6215,92 @@ dependencies = [ [[package]] name = "ratatui" -version = "0.26.3" +version = "0.30.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f44c9e68fd46eda15c646fbb85e1040b657a58cdc8c98db1d97a55930d991eef" +checksum = "3274ba0a2c5e1bcad2a2005d20f4dc59dad26b2eb0940fb094500dba4099d57d" dependencies = [ - "bitflags", - "cassowary", + "instability", + "ratatui-core", + "ratatui-crossterm", + "ratatui-termina", + "ratatui-termwiz", + "ratatui-widgets", + "serde", +] + +[[package]] +name = "ratatui-core" +version = "0.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cbb175c433c8e28a809d1f5773a2ae96e68c0ce40db865cbab1020bf33ae479c" +dependencies = [ + "bitflags 2.13.2", "compact_str", - "crossterm 0.27.0", - "itertools 0.12.1", + "critical-section", + "hashbrown 0.17.0", + "itertools", + "kasuari", "lru", - "paste", - "stability", - "strum 0.26.3", + "palette", + "serde", + "strum 0.28.0", + "thiserror 2.0.20", "unicode-segmentation", "unicode-truncate", - "unicode-width 0.1.14", + "unicode-width 0.2.2", +] + +[[package]] +name = "ratatui-crossterm" +version = "0.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "567584a3b0e6a8203c23de40b4861497266725eb5363dbfd18a1edd603cca9f0" +dependencies = [ + "cfg-if", + "crossterm 0.29.0", + "instability", + "ratatui-core", +] + +[[package]] +name = "ratatui-termina" +version = "0.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c0bf912d9e66f057a759d92e386a280ea886b352ab757d6ac4d653c7ed2c43c2" +dependencies = [ + "instability", + "ratatui-core", + "termina", +] + +[[package]] +name = "ratatui-termwiz" +version = "0.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "faf03e0380b7744054d6cb74224fe3adf062a029754933f575ca1e3b4c2ce977" +dependencies = [ + "ratatui-core", + "termwiz", +] + +[[package]] +name = "ratatui-widgets" +version = "0.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "66e3d19bcc9130ca376277d93b60767ff121ace3be06f5f95f81dd68956407d1" +dependencies = [ + "bitflags 2.13.2", + "hashbrown 0.17.0", + "indoc", + "instability", + "itertools", + "line-clipping", + "ratatui-core", + "serde", + "strum 0.28.0", + "time", + "unicode-segmentation", + "unicode-width 0.2.2", ] [[package]] @@ -5884,7 +6309,7 @@ version = "11.6.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "498cd0dc59d73224351ee52a95fee0f1a617a2eae0e7d9d720cc622c73a54186" dependencies = [ - "bitflags", + "bitflags 2.13.2", ] [[package]] @@ -5907,7 +6332,7 @@ version = "0.5.18" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ed2bf2547551a7053d6fdfafda3f938979645c44812fbfcda098faae3f1a362d" dependencies = [ - "bitflags", + "bitflags 2.13.2", ] [[package]] @@ -5916,7 +6341,7 @@ version = "0.7.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f450ad9c3b1da563fb6948a8e0fb0fb9269711c9c73d9ea1de5058c79c8d643a" dependencies = [ - "bitflags", + "bitflags 2.13.2", ] [[package]] @@ -6149,7 +6574,7 @@ checksum = "036204edbd199552a5b3832f63c60dcdf395dc44c7f06b4af1c0e8139cc11bce" dependencies = [ "aes", "aws-lc-rs", - "bitflags", + "bitflags 2.13.2", "block-padding", "byteorder", "bytes", @@ -6230,7 +6655,7 @@ version = "3.0.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "093197e526668d92bba562e2bbbe98d1af9831bf080b619c736316ca1fa35101" dependencies = [ - "bitflags", + "bitflags 2.13.2", "bytes", "chrono", "dashmap", @@ -6298,7 +6723,7 @@ version = "0.38.44" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "fdb5bc1ae2baa591800df16c9ca78619bf65c0488b41b96ccec5d11220d8c154" dependencies = [ - "bitflags", + "bitflags 2.13.2", "errno", "libc", "linux-raw-sys 0.4.15", @@ -6311,7 +6736,7 @@ version = "1.1.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b6fe4565b9518b83ef4f91bb47ce29620ca828bd32cb7e408f0062e9930ba190" dependencies = [ - "bitflags", + "bitflags 2.13.2", "errno", "libc", "linux-raw-sys 0.12.1", @@ -6538,7 +6963,7 @@ version = "3.7.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b7f4bc775c73d9a02cde8bf7b2ec4c9d12743edf609006c7facc23998404cd1d" dependencies = [ - "bitflags", + "bitflags 2.13.2", "core-foundation", "core-foundation-sys", "libc", @@ -6577,7 +7002,7 @@ version = "0.7.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f3a1a3341211875ef120e117ea7fd5228530ae7e7036a779fdc9117be6b3282c" dependencies = [ - "ordered-float", + "ordered-float 2.10.1", "serde", ] @@ -6811,8 +7236,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b75a19a7a740b25bc7944bdee6172368f988763b744e3d4dfe753f6b4ece40cc" dependencies = [ "libc", - "mio 0.8.11", - "mio 1.2.0", + "mio", "signal-hook", ] @@ -6880,6 +7304,12 @@ dependencies = [ "time", ] +[[package]] +name = "siphasher" +version = "1.0.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "33f4fe9184a62d842c9ef383018f3306d8ba224fd9d836f56d7288308847c256" + [[package]] name = "sketches-ddsketch" version = "0.3.1" @@ -7069,7 +7499,7 @@ version = "0.9.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "90b8020fe17c5f2c245bfa2505d7ef59c5604839527c740266ad2214acebea27" dependencies = [ - "bitflags", + "bitflags 2.13.2", "byteorder", "bytes", "crc", @@ -7097,7 +7527,7 @@ checksum = "87a2bdd6e83f6b3ea525ca9fee568030508b58355a43d0b2c1674d5f79dcd65e" dependencies = [ "atoi", "base64 0.22.1", - "bitflags", + "bitflags 2.13.2", "byteorder", "crc", "dotenvy", @@ -7207,16 +7637,6 @@ dependencies = [ "zeroize", ] -[[package]] -name = "stability" -version = "0.2.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d904e7009df136af5297832a3ace3370cd14ff1546a232f4f185036c2736fcac" -dependencies = [ - "quote", - "syn 2.0.117", -] - [[package]] name = "stable_deref_trait" version = "1.2.1" @@ -7248,37 +7668,36 @@ checksum = "7da8b5736845d9f2fcb837ea5d9e2628564b3b043a70948a3f0b778838c5fb4f" [[package]] name = "strum" -version = "0.26.3" +version = "0.27.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8fec0f0aef304996cf250b31b5a10dee7980c85da9d759361292b8bca5a18f06" -dependencies = [ - "strum_macros 0.26.4", -] +checksum = "af23d6f6c1a224baef9d3f61e287d2761385a5b88fdab4eb4c6f11aeb54c4bcf" [[package]] name = "strum" -version = "0.27.2" +version = "0.28.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "af23d6f6c1a224baef9d3f61e287d2761385a5b88fdab4eb4c6f11aeb54c4bcf" +checksum = "9628de9b8791db39ceda2b119bbe13134770b56c138ec1d3af810d045c04f9bd" +dependencies = [ + "strum_macros 0.28.0", +] [[package]] name = "strum_macros" -version = "0.26.4" +version = "0.27.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4c6bee85a5a24955dc440386795aa378cd9cf82acd5f764469152d2270e581be" +checksum = "7695ce3845ea4b33927c055a39dc438a45b059f7c1b3d91d38d10355fb8cbca7" dependencies = [ "heck", "proc-macro2", "quote", - "rustversion", "syn 2.0.117", ] [[package]] name = "strum_macros" -version = "0.27.2" +version = "0.28.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7695ce3845ea4b33927c055a39dc438a45b059f7c1b3d91d38d10355fb8cbca7" +checksum = "ab85eea0270ee17587ed4156089e10b9e6880ee688791d45a905f5b1ca36f664" dependencies = [ "heck", "proc-macro2", @@ -7319,6 +7738,17 @@ version = "0.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "a7973cce6668464ea31f176d85b13c7ab3bba2cb3b77a2ed26abd7801688010a" +[[package]] +name = "syn" +version = "1.0.109" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "72b64191b275b66ffe2469e8af2c1cfe3bafa67b529ead792a6d0160888b4237" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + [[package]] name = "syn" version = "2.0.117" @@ -7394,6 +7824,19 @@ dependencies = [ "windows-sys 0.61.2", ] +[[package]] +name = "termina" +version = "0.3.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9048a889effe34a5cddee0af7f53285198b16dca3be510858d38dfdb3e62a04e" +dependencies = [ + "bitflags 2.13.2", + "parking_lot", + "rustix 1.1.4", + "signal-hook", + "windows-sys 0.60.2", +] + [[package]] name = "terminal-colorsaurus" version = "1.0.3" @@ -7403,7 +7846,7 @@ dependencies = [ "cfg-if", "libc", "memchr", - "mio 1.2.0", + "mio", "terminal-trx", "windows-sys 0.61.2", "xterm-color", @@ -7430,6 +7873,69 @@ dependencies = [ "windows-sys 0.61.2", ] +[[package]] +name = "terminfo" +version = "0.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d4ea810f0692f9f51b382fff5893887bb4580f5fa246fde546e0b13e7fcee662" +dependencies = [ + "fnv", + "nom", + "phf", + "phf_codegen", +] + +[[package]] +name = "termios" +version = "0.3.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "411c5bf740737c7918b8b1fe232dca4dc9f8e754b8ad5e20966814001ed0ac6b" +dependencies = [ + "libc", +] + +[[package]] +name = "termwiz" +version = "0.23.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4676b37242ccbd1aabf56edb093a4827dc49086c0ffd764a5705899e0f35f8f7" +dependencies = [ + "anyhow", + "base64 0.22.1", + "bitflags 2.13.2", + "fancy-regex", + "filedescriptor", + "finl_unicode", + "fixedbitset 0.4.2", + "hex", + "lazy_static", + "libc", + "log", + "memmem", + "nix 0.29.0", + "num-derive", + "num-traits", + "ordered-float 4.6.0", + "pest", + "pest_derive", + "phf", + "sha2 0.10.9", + "signal-hook", + "siphasher", + "terminfo", + "termios", + "thiserror 1.0.69", + "ucd-trie", + "unicode-segmentation", + "vtparse", + "wezterm-bidi", + "wezterm-blob-leases", + "wezterm-color-types", + "wezterm-dynamic", + "wezterm-input-types", + "winapi", +] + [[package]] name = "text-size" version = "1.1.1" @@ -7504,7 +8010,9 @@ dependencies = [ "deranged", "itoa", "js-sys", + "libc", "num-conv", + "num_threads", "powerfmt", "serde_core", "time-core", @@ -7560,7 +8068,7 @@ checksum = "b67dee974fe86fd92cc45b7a95fdd2f99a36a6d7b0d431a231178d3d670bbcc6" dependencies = [ "bytes", "libc", - "mio 1.2.0", + "mio", "parking_lot", "pin-project-lite", "signal-hook-registry", @@ -7793,7 +8301,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "d4e6559d53cc268e5031cd8429d05415bc4cb4aefc4aa5d6cc35fbf5b924a1f8" dependencies = [ "base64 0.22.1", - "bitflags", + "bitflags 2.13.2", "bytes", "futures-util", "http 1.4.0", @@ -8042,13 +8550,13 @@ checksum = "9629274872b2bfaf8d66f5f15725007f635594914870f65218920345aa11aa8c" [[package]] name = "unicode-truncate" -version = "1.1.0" +version = "2.0.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b3644627a5af5fa321c95b9b235a72fd24cd29c648c2c379431e6628655627bf" +checksum = "16b380a1238663e5f8a691f9039c73e1cdae598a30e9855f541d29b08b53e9a5" dependencies = [ - "itertools 0.13.0", + "itertools", "unicode-segmentation", - "unicode-width 0.1.14", + "unicode-width 0.2.2", ] [[package]] @@ -8140,6 +8648,7 @@ version = "1.23.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ddd74a9687298c6858e9b88ec8935ec45d22e8fd5e6394fa1bd4e99a87789c76" dependencies = [ + "atomic", "getrandom 0.4.2", "js-sys", "wasm-bindgen", @@ -8169,6 +8678,15 @@ version = "0.8.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "5c3082ca00d5a5ef149bb8b555a72ae84c9c59f7250f013ac822ac2e49b19c64" +[[package]] +name = "vtparse" +version = "0.6.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6d9b2acfb050df409c972a37d3b8e08cdea3bddb0c09db9d53137e504cfabed0" +dependencies = [ + "utf8parse", +] + [[package]] name = "walkdir" version = "2.5.0" @@ -8308,7 +8826,7 @@ version = "0.244.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "47b807c72e1bac69382b3a6fb3dbe8ea4c0ed87ff5629b8685ae6b9a611028fe" dependencies = [ - "bitflags", + "bitflags 2.13.2", "hashbrown 0.15.5", "indexmap", "semver", @@ -8352,6 +8870,78 @@ dependencies = [ "rustls-pki-types", ] +[[package]] +name = "wezterm-bidi" +version = "0.2.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0c0a6e355560527dd2d1cf7890652f4f09bb3433b6aadade4c9b5ed76de5f3ec" +dependencies = [ + "log", + "wezterm-dynamic", +] + +[[package]] +name = "wezterm-blob-leases" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "692daff6d93d94e29e4114544ef6d5c942a7ed998b37abdc19b17136ea428eb7" +dependencies = [ + "getrandom 0.3.4", + "mac_address", + "sha2 0.10.9", + "thiserror 1.0.69", + "uuid", +] + +[[package]] +name = "wezterm-color-types" +version = "0.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7de81ef35c9010270d63772bebef2f2d6d1f2d20a983d27505ac850b8c4b4296" +dependencies = [ + "csscolorparser", + "deltae", + "lazy_static", + "wezterm-dynamic", +] + +[[package]] +name = "wezterm-dynamic" +version = "0.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5f2ab60e120fd6eaa68d9567f3226e876684639d22a4219b313ff69ec0ccd5ac" +dependencies = [ + "log", + "ordered-float 4.6.0", + "strsim", + "thiserror 1.0.69", + "wezterm-dynamic-derive", +] + +[[package]] +name = "wezterm-dynamic-derive" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "46c0cf2d539c645b448eaffec9ec494b8b19bd5077d9e58cb1ae7efece8d575b" +dependencies = [ + "proc-macro2", + "quote", + "syn 1.0.109", +] + +[[package]] +name = "wezterm-input-types" +version = "0.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7012add459f951456ec9d6c7e6fc340b1ce15d6fc9629f8c42853412c029e57e" +dependencies = [ + "bitflags 1.3.2", + "euclid", + "lazy_static", + "serde", + "wezterm-dynamic", +] + [[package]] name = "whoami" version = "2.1.3" @@ -8499,15 +9089,6 @@ dependencies = [ "windows-targets 0.42.2", ] -[[package]] -name = "windows-sys" -version = "0.48.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "677d2418bec65e3338edb076e806bc1ec15693c5d0104683f2efe857f61056a9" -dependencies = [ - "windows-targets 0.48.5", -] - [[package]] name = "windows-sys" version = "0.52.0" @@ -8559,21 +9140,6 @@ dependencies = [ "windows_x86_64_msvc 0.42.2", ] -[[package]] -name = "windows-targets" -version = "0.48.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9a2fa6e2155d7247be68c096456083145c183cbbbc2764150dda45a87197940c" -dependencies = [ - "windows_aarch64_gnullvm 0.48.5", - "windows_aarch64_msvc 0.48.5", - "windows_i686_gnu 0.48.5", - "windows_i686_msvc 0.48.5", - "windows_x86_64_gnu 0.48.5", - "windows_x86_64_gnullvm 0.48.5", - "windows_x86_64_msvc 0.48.5", -] - [[package]] name = "windows-targets" version = "0.52.6" @@ -8622,12 +9188,6 @@ version = "0.42.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "597a5118570b68bc08d8d59125332c54f1ba9d9adeedeef5b99b02ba2b0698f8" -[[package]] -name = "windows_aarch64_gnullvm" -version = "0.48.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2b38e32f0abccf9987a4e3079dfb67dcd799fb61361e53e2882c3cbaf0d905d8" - [[package]] name = "windows_aarch64_gnullvm" version = "0.52.6" @@ -8646,12 +9206,6 @@ version = "0.42.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e08e8864a60f06ef0d0ff4ba04124db8b0fb3be5776a5cd47641e942e58c4d43" -[[package]] -name = "windows_aarch64_msvc" -version = "0.48.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dc35310971f3b2dbbf3f0690a219f40e2d9afcf64f9ab7cc1be722937c26b4bc" - [[package]] name = "windows_aarch64_msvc" version = "0.52.6" @@ -8670,12 +9224,6 @@ version = "0.42.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "c61d927d8da41da96a81f029489353e68739737d3beca43145c8afec9a31a84f" -[[package]] -name = "windows_i686_gnu" -version = "0.48.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a75915e7def60c94dcef72200b9a8e58e5091744960da64ec734a6c6e9b3743e" - [[package]] name = "windows_i686_gnu" version = "0.52.6" @@ -8706,12 +9254,6 @@ version = "0.42.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "44d840b6ec649f480a41c8d80f9c65108b92d89345dd94027bfe06ac444d1060" -[[package]] -name = "windows_i686_msvc" -version = "0.48.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8f55c233f70c4b27f66c523580f78f1004e8b5a8b659e05a4eb49d4166cca406" - [[package]] name = "windows_i686_msvc" version = "0.52.6" @@ -8730,12 +9272,6 @@ version = "0.42.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "8de912b8b8feb55c064867cf047dda097f92d51efad5b491dfb98f6bbb70cb36" -[[package]] -name = "windows_x86_64_gnu" -version = "0.48.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "53d40abd2583d23e4718fddf1ebec84dbff8381c07cae67ff7768bbf19c6718e" - [[package]] name = "windows_x86_64_gnu" version = "0.52.6" @@ -8754,12 +9290,6 @@ version = "0.42.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "26d41b46a36d453748aedef1486d5c7a85db22e56aff34643984ea85514e94a3" -[[package]] -name = "windows_x86_64_gnullvm" -version = "0.48.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0b7b52767868a23d5bab768e390dc5f5c55825b6d30b86c844ff2dc7414044cc" - [[package]] name = "windows_x86_64_gnullvm" version = "0.52.6" @@ -8778,12 +9308,6 @@ version = "0.42.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9aec5da331524158c6d1a4ac0ab1541149c0b9505fde06423b02f5ef0106b9f0" -[[package]] -name = "windows_x86_64_msvc" -version = "0.48.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ed94fce61571a4006852b7389a063ab983c02eb1bb37b47f8272ce92d06d9538" - [[package]] name = "windows_x86_64_msvc" version = "0.52.6" @@ -8892,7 +9416,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9d66ea20e9553b30172b5e831994e35fbde2d165325bec84fc43dbf6f4eb9cb2" dependencies = [ "anyhow", - "bitflags", + "bitflags 2.13.2", "indexmap", "log", "serde", diff --git a/Cargo.toml b/Cargo.toml index a50457f136..754a72c3f6 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -49,7 +49,7 @@ clap = { version = "4.5", features = ["derive", "env"] } clap_complete = { version = "4.5", features = ["unstable-dynamic"] } indicatif = "0.17" owo-colors = "4" -ratatui = "0.26" +ratatui = { version = "0.30.2", default-features = false, features = ["crossterm", "layout-cache", "underline-color"] } crossterm = "0.28" terminal-colorsaurus = "1.0" diff --git a/crates/openshell-tui/src/app.rs b/crates/openshell-tui/src/app.rs index 4b218a37cf..2cbd7b2946 100644 --- a/crates/openshell-tui/src/app.rs +++ b/crates/openshell-tui/src/app.rs @@ -3725,7 +3725,7 @@ mod tests { let mut terminal = ratatui::Terminal::new(ratatui::backend::TestBackend::new(100, 36)) .expect("test terminal"); terminal - .draw(|frame| crate::ui::create_provider::draw_detail(frame, app, frame.size())) + .draw(|frame| crate::ui::create_provider::draw_detail(frame, app, frame.area())) .expect("provider detail renders"); terminal .backend() diff --git a/crates/openshell-tui/src/ui/mod.rs b/crates/openshell-tui/src/ui/mod.rs index 9b33146179..749fddc1a5 100644 --- a/crates/openshell-tui/src/ui/mod.rs +++ b/crates/openshell-tui/src/ui/mod.rs @@ -26,7 +26,7 @@ use crate::theme::Theme; pub fn draw(frame: &mut Frame<'_>, app: &mut App) { // Splash screen is a full-screen takeover — no chrome. if app.screen == Screen::Splash { - splash::draw(frame, frame.size(), &app.theme); + splash::draw(frame, frame.area(), &app.theme); return; } @@ -38,7 +38,7 @@ pub fn draw(frame: &mut Frame<'_>, app: &mut App) { Constraint::Length(1), // nav bar Constraint::Length(1), // command bar ]) - .split(frame.size()); + .split(frame.area()); draw_title_bar(frame, app, chunks[0]); @@ -53,16 +53,16 @@ pub fn draw(frame: &mut Frame<'_>, app: &mut App) { // Modal overlays (drawn last so they're on top). if app.create_form.is_some() { - create_sandbox::draw(frame, app, frame.size()); + create_sandbox::draw(frame, app, frame.area()); } if app.create_provider_form.is_some() { - create_provider::draw(frame, app, frame.size()); + create_provider::draw(frame, app, frame.area()); } if app.provider_detail.is_some() { - create_provider::draw_detail(frame, app, frame.size()); + create_provider::draw_detail(frame, app, frame.area()); } if app.update_provider_form.is_some() { - create_provider::draw_update(frame, app, frame.size()); + create_provider::draw_update(frame, app, frame.area()); } } @@ -96,7 +96,7 @@ fn draw_sandbox_screen(frame: &mut Frame<'_>, app: &mut App, area: Rect) { { let filtered: Vec<&app::LogLine> = app.filtered_log_lines(); if let Some(log) = filtered.get(detail_idx) { - sandbox_logs::draw_detail_popup(frame, log, frame.size(), &app.theme); + sandbox_logs::draw_detail_popup(frame, log, frame.area(), &app.theme); } } @@ -105,7 +105,7 @@ fn draw_sandbox_screen(frame: &mut Frame<'_>, app: &mut App, area: Rect) { let abs = app.draft_scroll + app.draft_selected; let scroll = app.draft_detail_scroll; let metrics = app.draft_chunks.get(abs).map(|chunk| { - sandbox_draft::draw_detail_popup(frame, chunk, frame.size(), &app.theme, scroll) + sandbox_draft::draw_detail_popup(frame, chunk, frame.area(), &app.theme, scroll) }); if let Some(metrics) = metrics { app.draft_detail_rows = metrics.total_rows; @@ -118,7 +118,7 @@ fn draw_sandbox_screen(frame: &mut Frame<'_>, app: &mut App, area: Rect) { sandbox_draft::draw_approve_all_popup( frame, &app.approve_all_confirm_chunks, - frame.size(), + frame.area(), &app.theme, ); } @@ -701,7 +701,7 @@ mod tests { let mut terminal = Terminal::new(TestBackend::new(180, 1)).unwrap(); terminal - .draw(|frame| draw_nav_bar(frame, &app, frame.size())) + .draw(|frame| draw_nav_bar(frame, &app, frame.area())) .unwrap(); let text: String = terminal @@ -725,14 +725,14 @@ mod tests { .draw(|frame| { frame.render_widget( Paragraph::new(Line::from(title_bar_brand_spans(&Theme::dark()))), - frame.size(), + frame.area(), ); }) .unwrap(); let buffer = terminal.backend().buffer(); let rendered = (0..width) - .map(|x| buffer.get(x, 0).symbol()) + .map(|x| buffer[(x, 0)].symbol()) .collect::(); assert_eq!(rendered, expected); assert!(!rendered.contains("ALPHA")); @@ -752,7 +752,7 @@ mod tests { let mut terminal = Terminal::new(TestBackend::new(width, 24)).unwrap(); terminal .draw(|frame| { - sandboxes::draw(frame, &app, frame.size(), true); + sandboxes::draw(frame, &app, frame.area(), true); }) .unwrap(); let text: String = terminal diff --git a/crates/openshell-tui/src/ui/sandbox_logs.rs b/crates/openshell-tui/src/ui/sandbox_logs.rs index 45f548e8ba..d2dda73309 100644 --- a/crates/openshell-tui/src/ui/sandbox_logs.rs +++ b/crates/openshell-tui/src/ui/sandbox_logs.rs @@ -116,7 +116,7 @@ pub fn draw(frame: &mut Frame<'_>, app: &mut App, area: Rect) { frame.render_widget(Paragraph::new(lines).block(block), area); // NOTE: Detail popup overlay is now rendered by draw_sandbox_screen() in - // mod.rs using frame.size() so it renders over the full screen, not + // mod.rs using frame.area() so it renders over the full screen, not // constrained to this pane. } From 121930be076427acc408e58ac5e1b31a6202e55f Mon Sep 17 00:00:00 2001 From: Drew Newberry Date: Fri, 2 Oct 2026 21:31:40 +0000 Subject: [PATCH 02/13] fix(sandbox): replace lifetime exec cap with retry deadlines (#4105) * fix(sandbox): replace lifetime exec cap with retry deadlines Signed-off-by: Drew Newberry * docs(sandboxes): remove exec recovery overview change Signed-off-by: Drew Newberry * docs(skills): remove exec recovery CLI skill change Signed-off-by: Drew Newberry --------- Signed-off-by: Drew Newberry --- crates/openshell-sandbox-backend/README.md | 26 +++ .../src/boundary_protocol.rs | 89 +++++++- .../openshell-sandbox-backend/src/runtime.rs | 17 +- .../openshell-sandbox/src/boundary_server.rs | 209 +++++++++++++----- e2e/rust/tests/sandbox_lifecycle.rs | 22 ++ 5 files changed, 298 insertions(+), 65 deletions(-) create mode 100644 crates/openshell-sandbox-backend/README.md diff --git a/crates/openshell-sandbox-backend/README.md b/crates/openshell-sandbox-backend/README.md new file mode 100644 index 0000000000..36a3933b28 --- /dev/null +++ b/crates/openshell-sandbox-backend/README.md @@ -0,0 +1,26 @@ +# OpenShell sandbox backend + +This crate implements the supervisor-side isolation backend and the private +control protocol shared with `openshell-sandbox`. + +## Exec recovery + +An exec envelope carries a UUID request ID and an absolute expiration time in +Unix milliseconds, set to 30 seconds after creation. The payload digest includes +the expiration time, and transport recovery preserves the entire envelope. +The supervisor and sandbox need synchronized wall clocks. + +The sandbox checks expiration before starting or reattaching an exec. It keeps +each admitted request ID until that deadline, independently of process and I/O +retention. An expired request, including a delayed first attempt, is rejected +even after its ID has been discarded. The sandbox never moves its observed +admission clock backwards, so a clock adjustment cannot resurrect discarded +requests. Expired IDs are reclaimed when the next exec request arrives. + +There is no lifetime exec request count limit. The existing concurrent process +retention limit still applies. The deadline governs admission and recovery; +it does not terminate an already running command. A recovery timeout can leave +the execution outcome unknown. + +Exec envelopes without an expiration time are rejected. Update the supervisor +and sandbox runtime together when deploying this protocol change. diff --git a/crates/openshell-sandbox-backend/src/boundary_protocol.rs b/crates/openshell-sandbox-backend/src/boundary_protocol.rs index 6c6c66e76c..b7b32f72bd 100644 --- a/crates/openshell-sandbox-backend/src/boundary_protocol.rs +++ b/crates/openshell-sandbox-backend/src/boundary_protocol.rs @@ -14,6 +14,7 @@ use std::io; use std::io::{Read, Write}; use std::path::PathBuf; use std::str::FromStr; +use std::time::{Duration, SystemTime, UNIX_EPOCH}; use openshell_core::SandboxSessionId; use openshell_core::policy::{ @@ -34,6 +35,8 @@ use sha2::{Digest as _, Sha256}; use tokio::io::{AsyncRead, AsyncReadExt, AsyncWrite, AsyncWriteExt}; pub const MAX_CONTROL_FRAME_BYTES: usize = 1024 * 1024; +/// Exec admission and recovery share one deadline; it never limits process runtime. +pub const EXEC_REQUEST_RETRY_WINDOW: Duration = Duration::from_secs(30); pub const STREAM_STDIN: u8 = 0; pub const STREAM_STDOUT: u8 = 1; pub const STREAM_STDERR: u8 = 2; @@ -625,6 +628,10 @@ pub struct RequestEnvelope { pub request_id: String, /// SHA-256 of the canonically serialized request payload. pub payload_digest: String, + /// Absolute exec admission deadline, preserved across retries and bound to + /// the payload digest. Other request kinds do not expire through this field. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub exec_expires_at_unix_ms: Option, pub request: Request, } @@ -632,10 +639,18 @@ impl RequestEnvelope { /// Build a request envelope with a fresh idempotency key and normalized /// payload digest. pub fn new(request: Request) -> Result { - let payload_digest = request_payload_digest(&request)?; + let exec_expires_at_unix_ms = if matches!(request, Request::Exec { .. }) { + let window_ms = + u64::try_from(EXEC_REQUEST_RETRY_WINDOW.as_millis()).map_err(io::Error::other)?; + Some(unix_time_millis()?.saturating_add(window_ms)) + } else { + None + }; + let payload_digest = request_envelope_digest(&request, exec_expires_at_unix_ms)?; Ok(Self { request_id: uuid::Uuid::new_v4().to_string(), payload_digest, + exec_expires_at_unix_ms, request, }) } @@ -643,7 +658,7 @@ impl RequestEnvelope { /// Verify that the request body still matches the immutable digest bound /// to this idempotency key. pub fn validate_payload_digest(&self) -> Result<(), FrameError> { - let actual = request_payload_digest(&self.request)?; + let actual = request_envelope_digest(&self.request, self.exec_expires_at_unix_ms)?; if actual == self.payload_digest { Ok(()) } else { @@ -652,11 +667,25 @@ impl RequestEnvelope { } } -fn request_payload_digest(request: &Request) -> Result { +/// Shared wall clock for the host's deadline and the boundary's admission check. +pub fn unix_time_millis() -> Result { + let elapsed = SystemTime::now() + .duration_since(UNIX_EPOCH) + .map_err(io::Error::other)?; + u64::try_from(elapsed.as_millis()).map_err(|error| FrameError::Io(io::Error::other(error))) +} + +fn request_envelope_digest( + request: &Request, + exec_expires_at_unix_ms: Option, +) -> Result { // Sort every object explicitly: dependency features may make Value retain // insertion order. Provider environments must hash identically after // deserialization and across independently serialized retries. let mut normalized = serde_json::to_value(request).map_err(FrameError::Serialize)?; + if let Some(expires_at) = exec_expires_at_unix_ms { + normalized["exec_expires_at_unix_ms"] = expires_at.into(); + } normalized.sort_all_objects(); let payload = serde_json::to_vec(&normalized).map_err(FrameError::Serialize)?; let digest = Sha256::digest(payload); @@ -1574,6 +1603,7 @@ mod tests { let request = RequestEnvelope { request_id: "4e94636d-54f8-4d85-8e4e-58954fb5af0a".to_string(), payload_digest: String::new(), + exec_expires_at_unix_ms: None, request: Request::StartAgent { provider_files: std::collections::HashMap::new(), sandbox_id: "sandbox-1".to_string(), @@ -1601,7 +1631,8 @@ mod tests { }, }; let request = RequestEnvelope { - payload_digest: request_payload_digest(&request.request).expect("request digest"), + payload_digest: request_envelope_digest(&request.request, None) + .expect("request digest"), ..request }; let frame = encode_frame(&request).expect("encode request"); @@ -1616,6 +1647,51 @@ mod tests { assert!(request.validate_payload_digest().is_ok()); } + #[test] + fn exec_deadline_round_trips_and_cannot_be_extended_on_retry() { + let before = unix_time_millis().unwrap(); + let request = RequestEnvelope::new(Request::Exec { + spec: ExecSpecWire { + program: "/bin/true".to_string(), + args: Vec::new(), + shell: None, + runtime_helper: None, + env: Vec::new(), + workdir: None, + pty: false, + }, + }) + .unwrap(); + let after = unix_time_millis().unwrap(); + let window_ms = u64::try_from(EXEC_REQUEST_RETRY_WINDOW.as_millis()).unwrap(); + let deadline = request.exec_expires_at_unix_ms.unwrap(); + assert!((before + window_ms..=after + window_ms).contains(&deadline)); + let frame = encode_frame(&request).unwrap(); + let mut retry: RequestEnvelope = decode_frame(&frame).unwrap(); + assert_eq!(retry, request); + retry.validate_payload_digest().unwrap(); + retry.exec_expires_at_unix_ms = Some(deadline + 1); + assert!(matches!( + retry.validate_payload_digest(), + Err(FrameError::PayloadDigestMismatch) + )); + retry.exec_expires_at_unix_ms = None; + assert!(matches!( + retry.validate_payload_digest(), + Err(FrameError::PayloadDigestMismatch) + )); + } + + #[test] + fn non_exec_requests_have_no_deadline() { + let envelope = RequestEnvelope::new(Request::Confirm).unwrap(); + assert!(envelope.exec_expires_at_unix_ms.is_none()); + let encoded = serde_json::to_value(&envelope).unwrap(); + assert!(encoded.get("exec_expires_at_unix_ms").is_none()); + let decoded: RequestEnvelope = serde_json::from_value(encoded).unwrap(); + decoded.validate_payload_digest().unwrap(); + } + #[test] fn request_digest_is_stable_across_map_order_and_detects_mutation() { let mut first = std::collections::HashMap::new(); @@ -1641,7 +1717,10 @@ mod tests { ); for provider_env in [first, second] { let request = build(provider_env); - assert_eq!(request_payload_digest(&request).expect("digest"), expected); + assert_eq!( + request_envelope_digest(&request, None).expect("digest"), + expected + ); // Deserialization reconstructs the map with an independent hash // seed; validation must retain the sender's canonical digest. diff --git a/crates/openshell-sandbox-backend/src/runtime.rs b/crates/openshell-sandbox-backend/src/runtime.rs index 641d50cd24..b4a644ebb6 100644 --- a/crates/openshell-sandbox-backend/src/runtime.rs +++ b/crates/openshell-sandbox-backend/src/runtime.rs @@ -35,11 +35,11 @@ use tokio::net::UnixStream; use tokio_stream::wrappers::ReceiverStream; use crate::boundary_protocol::{ - AgentSpecWire, DnsQueryResultWire, ExecSpecWire, ExitStatusWire, MAX_CONTROL_FRAME_BYTES, - Request, RequestEnvelope, Response, ResponseEnvelope, STREAM_EXIT, STREAM_STDERR, STREAM_STDIN, - STREAM_STDIN_CLOSED, STREAM_STDOUT, SandboxPolicyWire, SandboxRuntimeDescriptor, - SandboxTlsClientConfig, SandboxTransport, SignalWire, decode_frame, encode_frame, - read_stream_frame, validate_resource_claims, write_stream_frame, + AgentSpecWire, DnsQueryResultWire, EXEC_REQUEST_RETRY_WINDOW, ExecSpecWire, ExitStatusWire, + MAX_CONTROL_FRAME_BYTES, Request, RequestEnvelope, Response, ResponseEnvelope, STREAM_EXIT, + STREAM_STDERR, STREAM_STDIN, STREAM_STDIN_CLOSED, STREAM_STDOUT, SandboxPolicyWire, + SandboxRuntimeDescriptor, SandboxTlsClientConfig, SandboxTransport, SignalWire, decode_frame, + encode_frame, read_stream_frame, validate_resource_claims, write_stream_frame, }; use crate::mediation::{self, DnsQueryWire, MediationFrame, MediationFrameKind}; @@ -1398,7 +1398,7 @@ impl BoundaryClient { request: Request, ) -> Result<(BoundaryDuplexStream, Response), BackendError> { let envelope = Self::prepare_request(request)?; - tokio::time::timeout(REQUEST_TIMEOUT, async { + tokio::time::timeout(EXEC_REQUEST_RETRY_WINDOW, async { loop { let generation = self.connection_generation().await; match self.open_exchange_envelope(&envelope).await { @@ -1415,7 +1415,10 @@ impl BoundaryClient { }) .await .map_err(|_| { - BackendError::Unavailable("boundary idempotent stream request timed out".to_string()) + BackendError::Unavailable( + "exec startup recovery deadline expired; execution outcome may be unknown" + .to_string(), + ) })? } diff --git a/crates/openshell-sandbox/src/boundary_server.rs b/crates/openshell-sandbox/src/boundary_server.rs index f84f98d424..4de7c63405 100644 --- a/crates/openshell-sandbox/src/boundary_server.rs +++ b/crates/openshell-sandbox/src/boundary_server.rs @@ -65,7 +65,8 @@ mod linux { ProcessKindWire, ProcessSnapshotWire, Request, RequestEnvelope, Response, ResponseEnvelope, STREAM_EXIT, STREAM_NETWORK_DECISION, STREAM_STDERR, STREAM_STDIN, STREAM_STDIN_CLOSED, STREAM_STDOUT, SandboxPolicyWire, SessionSnapshotWire, SignalWire, encode_frame, - read_frame, read_stream_frame, validate_resource_claims, write_frame, write_stream_frame, + read_frame, read_stream_frame, unix_time_millis, validate_resource_claims, write_frame, + write_stream_frame, }; const CONTROL_IO_TIMEOUT: Duration = Duration::from_secs(30); @@ -1110,20 +1111,24 @@ mod linux { .map_err(|error| format!("write boundary termination response: {error}")); } Request::Exec { spec } => { - let started = - match runtime.start_exec(&request.request_id, &request.payload_digest, spec) { - Ok(started) => started, - Err(response) => { - return write_frame( - &mut stream, - &ResponseEnvelope { - request_id: request.request_id, - response, - }, - ) - .map_err(|error| format!("write exec error response: {error}")); - } - }; + let started = match runtime.start_exec( + &request.request_id, + &request.payload_digest, + request.exec_expires_at_unix_ms, + spec, + ) { + Ok(started) => started, + Err(response) => { + return write_frame( + &mut stream, + &ResponseEnvelope { + request_id: request.request_id, + response, + }, + ) + .map_err(|error| format!("write exec error response: {error}")); + } + }; if let Err(error) = write_frame( &mut stream, &ResponseEnvelope { @@ -1328,10 +1333,9 @@ mod linux { mediation_active: tokio::sync::Mutex<()>, next_mediation_stream_id: AtomicU64, exec_handles: Mutex>, - /// Never evicted within a boundary generation. Reclaiming process I/O - /// must not make an old command executable again. At capacity, reject - /// new commands instead of silently weakening at-most-once execution. - exec_requests: Mutex>, + /// Keep exec tombstones through their admission deadline. Afterwards, + /// even a delayed first attempt is rejected without retaining its ID. + exec_requests: Mutex, replay_ledger: Mutex, network_broker: NetworkBroker, workload_launcher: @@ -1350,25 +1354,58 @@ mod linux { status: Arc>>, } - #[allow(clippy::result_large_err)] - fn reserve_exec_request( - requests: &mut std::collections::HashSet, - request_id: &str, - ) -> Result<(), Response> { - if requests.contains(request_id) { - return Err(guest_error( - BoundaryErrorKind::Denied, - "exec request has expired; it cannot be executed again", - )); + #[derive(Default)] + struct ExecRequestLedger { + requests: std::collections::HashSet, + expirations: std::collections::BinaryHeap>, + /// A backwards wall-clock adjustment must not admit a request whose + /// tombstone has already been removed. + last_unix_ms: u64, + } + + impl ExecRequestLedger { + #[allow(clippy::result_large_err)] + fn validate_deadline( + &mut self, + expires_at_unix_ms: Option, + now_unix_ms: u64, + ) -> Result { + self.last_unix_ms = self.last_unix_ms.max(now_unix_ms); + while let Some(std::cmp::Reverse((expires_at, _))) = self.expirations.peek() { + if *expires_at > self.last_unix_ms { + break; + } + if let Some(std::cmp::Reverse((_, request_id))) = self.expirations.pop() { + self.requests.remove(&request_id); + } + } + let expires_at = expires_at_unix_ms.ok_or_else(|| { + guest_error( + BoundaryErrorKind::Invalid, + "exec request requires an expiration deadline", + ) + })?; + if expires_at <= self.last_unix_ms { + return Err(guest_error( + BoundaryErrorKind::Denied, + "exec request deadline expired; execution outcome may be unknown", + )); + } + Ok(expires_at) } - if requests.len() >= MAX_REPLAY_LEDGER_ENTRIES { - return Err(guest_error( - BoundaryErrorKind::Unavailable, - "boundary generation exec request limit reached", - )); + + #[allow(clippy::result_large_err)] + fn reserve(&mut self, request_id: &str, expires_at: u64) -> Result<(), Response> { + if !self.requests.insert(request_id.to_owned()) { + return Err(guest_error( + BoundaryErrorKind::Denied, + "exec request is no longer retained; it cannot be executed again", + )); + } + self.expirations + .push(std::cmp::Reverse((expires_at, request_id.to_owned()))); + Ok(()) } - requests.insert(request_id.to_owned()); - Ok(()) } struct StartedExec { @@ -1571,7 +1608,7 @@ mod linux { mediation_active: tokio::sync::Mutex::new(()), next_mediation_stream_id: AtomicU64::new(1), exec_handles: Mutex::new(std::collections::HashMap::new()), - exec_requests: Mutex::new(std::collections::HashSet::new()), + exec_requests: Mutex::new(ExecRequestLedger::default()), replay_ledger: Mutex::new(ReplayLedger::default()), network_broker, workload_launcher, @@ -2021,6 +2058,7 @@ mod linux { &self, request_id: &str, payload_digest: &str, + expires_at_unix_ms: Option, spec: ExecSpecWire, ) -> Result { let executor = { @@ -2034,6 +2072,10 @@ mod linux { process.boundary_exec() }; let mut handles = lock(&self.exec_handles); + let mut requests = lock(&self.exec_requests); + let now_unix_ms = unix_time_millis() + .map_err(|error| guest_error(BoundaryErrorKind::Unavailable, error.to_string()))?; + let expires_at = requests.validate_deadline(expires_at_unix_ms, now_unix_ms)?; if let Some((process_id, handle)) = handles .iter() .find(|(_, handle)| handle.request_id == request_id) @@ -2072,10 +2114,8 @@ mod linux { )); } } - { - let mut requests = lock(&self.exec_requests); - reserve_exec_request(&mut requests, request_id)?; - } + requests.reserve(request_id, expires_at)?; + drop(requests); let session = self .process_runtime .block_on(executor.exec(spec.into())) @@ -3678,18 +3718,45 @@ mod linux { use rcgen::{KeyPair, PKCS_ED25519}; #[test] - fn exec_tombstones_outlive_retained_handles_and_fail_closed_at_capacity() { - let mut requests = std::collections::HashSet::new(); - reserve_exec_request(&mut requests, "first").unwrap(); - // Process/I/O retention is deliberately not consulted by this - // ledger: dropping all handles cannot make this ID executable. - assert!(reserve_exec_request(&mut requests, "first").is_err()); - for index in 1..MAX_REPLAY_LEDGER_ENTRIES { - reserve_exec_request(&mut requests, &format!("request-{index}")).unwrap(); + fn exec_tombstones_expire_without_a_lifetime_limit() { + let mut ledger = ExecRequestLedger::default(); + // More than the old 4,096 limit can be admitted in one window. + for index in 0..10_000 { + let deadline = ledger.validate_deadline(Some(30_000), 0).unwrap(); + ledger + .reserve(&format!("request-{index}"), deadline) + .unwrap(); } - assert!(reserve_exec_request(&mut requests, "overflow").is_err()); - assert!(requests.contains("first")); - assert_eq!(requests.len(), MAX_REPLAY_LEDGER_ENTRIES); + assert_eq!(ledger.requests.len(), 10_000); + assert!(ledger.reserve("request-0", 30_000).is_err()); + // Expiration clears the IDs but never lets an old envelope run again. + assert!(ledger.validate_deadline(Some(30_000), 30_000).is_err()); + assert!(ledger.requests.is_empty()); + assert!(ledger.expirations.is_empty()); + let deadline = ledger.validate_deadline(Some(60_000), 30_000).unwrap(); + ledger.reserve("next-request", deadline).unwrap(); + assert_eq!(ledger.requests.len(), 1); + } + + #[test] + fn exec_deadlines_reject_missing_expired_and_delayed_first_attempts() { + let mut ledger = ExecRequestLedger::default(); + assert!(ledger.validate_deadline(None, 10_000).is_err()); + assert!(ledger.validate_deadline(Some(10_000), 10_000).is_err()); + assert!(ledger.validate_deadline(Some(9_999), 10_000).is_err()); + // A backwards clock change cannot resurrect expired requests. + assert!(ledger.validate_deadline(Some(10_000), 0).is_err()); + } + + #[test] + fn exec_tombstones_expire_in_deadline_order() { + let mut ledger = ExecRequestLedger::default(); + ledger.reserve("later", 30_000).unwrap(); + ledger.reserve("earlier", 20_000).unwrap(); + ledger.validate_deadline(Some(30_000), 20_000).unwrap(); + assert!(!ledger.requests.contains("earlier")); + assert!(ledger.requests.contains("later")); + assert!(ledger.reserve("later", 30_000).is_err()); } #[test] @@ -5093,6 +5160,7 @@ mod linux { .start_exec( &exec_request.request_id, &exec_request.payload_digest, + exec_request.exec_expires_at_unix_ms, exec_spec, ) .expect("exec after reconnect"); @@ -5387,10 +5455,38 @@ mod linux { spec: sleep_spec.clone(), }) .expect("build retained exec request"); + for deadline in [None, Some(0)] { + assert!( + boundary + .start_exec( + &sleep_request.request_id, + &sleep_request.payload_digest, + deadline, + sleep_spec.clone(), + ) + .is_err(), + "missing or expired deadlines must not start a process" + ); + assert!(lock(&boundary.exec_handles).is_empty()); + } + // Completed exec IDs must not impose the former lifetime limit on + // a real process launch or recovery of its lost start response. + { + let mut requests = lock(&boundary.exec_requests); + for index in 0..4096 { + requests + .reserve( + &format!("completed-request-{index}"), + sleep_request.exec_expires_at_unix_ms.unwrap(), + ) + .unwrap(); + } + } let started = boundary .start_exec( &sleep_request.request_id, &sleep_request.payload_digest, + sleep_request.exec_expires_at_unix_ms, sleep_spec.clone(), ) .expect("start exec whose response is disconnected"); @@ -5400,6 +5496,7 @@ mod linux { .start_exec( &sleep_request.request_id, &sleep_request.payload_digest, + sleep_request.exec_expires_at_unix_ms, sleep_spec.clone(), ) .expect("reattach exec after response loss"); @@ -5425,6 +5522,7 @@ mod linux { .start_exec( &sleep_request.request_id, &sleep_request.payload_digest, + sleep_request.exec_expires_at_unix_ms, sleep_spec, ) .is_err(), @@ -5455,7 +5553,12 @@ mod linux { let request = RequestEnvelope::new(Request::Exec { spec: spec.clone() }) .expect("build exec status request"); let exec = boundary - .start_exec(&request.request_id, &request.payload_digest, spec) + .start_exec( + &request.request_id, + &request.payload_digest, + request.exec_expires_at_unix_ms, + spec, + ) .expect("start exec after canonical exit"); for _ in 0..2 { assert_eq!( diff --git a/e2e/rust/tests/sandbox_lifecycle.rs b/e2e/rust/tests/sandbox_lifecycle.rs index a7c306fec9..5e708e097c 100644 --- a/e2e/rust/tests/sandbox_lifecycle.rs +++ b/e2e/rust/tests/sandbox_lifecycle.rs @@ -112,6 +112,28 @@ async fn delete_sandbox(name: &str) { let _ = cmd.status().await; } +#[tokio::test] +#[serial(sandbox_lifecycle)] +async fn sandbox_exec_outlives_startup_recovery_deadline() { + let mut sandbox = SandboxGuard::create(&[]) + .await + .expect("create sandbox for exec recovery deadline"); + let output = tokio::time::timeout( + Duration::from_secs(60), + sandbox.exec(&["sh", "-c", "sleep 31; printf past-recovery-deadline"]), + ) + .await; + let next = sandbox.exec(&["printf", "fresh-exec"]).await; + sandbox.cleanup().await; + + assert_eq!( + output.expect("exec timed out").expect("exec failed"), + "past-recovery-deadline", + "expiration must not terminate an admitted command" + ); + assert_eq!(next.expect("fresh exec after expiration"), "fresh-exec"); +} + #[tokio::test] #[serial(sandbox_lifecycle)] async fn sandbox_exec_large_output_is_complete() { From 48d9ab3d0d9a343365dea1b0cd87565050ac658e Mon Sep 17 00:00:00 2001 From: Shiju Date: Fri, 2 Oct 2026 22:10:01 +0000 Subject: [PATCH 03/13] fix(cli): check final exec status and warn on partial input (#3803) * feat(cli): stream non-TTY exec input before EOF Add --stream-stdin using the existing interactive exec RPC without a PTY. Preserve separate output streams and enforce the existing 4 MiB cumulative input cap while forwarding input. Require explicit clean stdin EOF and drain the response through its final gRPC status. Cover held-open input, limits, cancellation, trailers, and default finite-input behavior with subprocess and live sandbox regressions. Signed-off-by: Shiju * test(cli): distinguish exec cancellation from stdin EOF Treat transport termination and response cancellation as separate test observations. Verify explicit stdin EOF through the shared frame writer and cover cancellation in the pinned Tonic decoder. Signed-off-by: Shiju * style(cli): use lazy optional stdin dispatch Signed-off-by: Shiju * test(cli): use imported duration in stdin EOF regression Signed-off-by: Shiju --------- Signed-off-by: Shiju --- crates/openshell-cli/src/run.rs | 6 +- .../sandbox_exec_streaming_integration.rs | 553 ++++++++++++++++++ docs/how-it-works/sandboxes/overview.mdx | 4 + 3 files changed, 561 insertions(+), 2 deletions(-) create mode 100644 crates/openshell-cli/tests/sandbox_exec_streaming_integration.rs diff --git a/crates/openshell-cli/src/run.rs b/crates/openshell-cli/src/run.rs index 028b10dfdd..7432c0175f 100644 --- a/crates/openshell-cli/src/run.rs +++ b/crates/openshell-cli/src/run.rs @@ -2506,7 +2506,7 @@ async fn sandbox_exec_streaming_grpc( if forwarded > MAX_EXEC_STDIN_BYTES { return Err(std::io::Error::new( ErrorKind::InvalidInput, - piped_stdin_limit_error().to_string(), + "streamed stdin exceeds the 4 MiB limit; the command may have processed partial input; use `sandbox upload` for larger input", )); } if stdin_tx @@ -2648,7 +2648,9 @@ async fn sandbox_exec_streaming_grpc( Some(exec_sandbox_event::Payload::Exit(exit)) => { exit_code = exit.exit_code; exit_seen = true; - break; + // Process exit does not complete the RPC. Keep draining so a + // failing final gRPC status cannot turn partial output into + // a successful execution result. } None => {} } diff --git a/crates/openshell-cli/tests/sandbox_exec_streaming_integration.rs b/crates/openshell-cli/tests/sandbox_exec_streaming_integration.rs new file mode 100644 index 0000000000..359895838a --- /dev/null +++ b/crates/openshell-cli/tests/sandbox_exec_streaming_integration.rs @@ -0,0 +1,553 @@ +// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +#![cfg(unix)] + +mod helpers; + +use std::path::Path; +use std::process::{Output, Stdio}; +use std::sync::{Arc, Mutex}; +use std::time::Duration; + +use helpers::{build_ca, build_client_cert, build_server_cert}; +use openshell_core::proto::open_shell_server::{OpenShell, OpenShellServer}; +use openshell_core::proto::{self, exec_sandbox_event, exec_sandbox_input}; +use tokio::io::{AsyncBufReadExt, AsyncReadExt, AsyncWriteExt, BufReader}; +use tokio::net::TcpListener; +use tokio::process::{Child, Command}; +use tokio::sync::mpsc; +use tokio::task::JoinHandle; +use tokio::time::timeout; +use tokio_stream::wrappers::{ReceiverStream, TcpListenerStream}; +use tonic::transport::{Certificate, Identity, Server, ServerTlsConfig}; +use tonic::{Request, Response, Status}; + +const DEADLINE: Duration = Duration::from_secs(10); +const STDIN_LIMIT: usize = 4 * 1024 * 1024; +type EventStream = ReceiverStream>; + +#[derive(Clone, Copy)] +enum Scenario { + Echo, + EarlyExit, + ErrorAfterExit, + ErrorAfterInput, +} + +#[derive(Default)] +struct Calls { + lookups: usize, + unary: usize, + starts: Vec, + input_bytes: usize, +} + +#[derive(Clone)] +struct MockGateway { + scenario: Scenario, + calls: Arc>, +} + +fn stdout(data: impl Into>) -> proto::ExecSandboxEvent { + proto::ExecSandboxEvent { + payload: Some(exec_sandbox_event::Payload::Stdout( + proto::ExecSandboxStdout { data: data.into() }, + )), + } +} + +fn exit(code: i32) -> proto::ExecSandboxEvent { + proto::ExecSandboxEvent { + payload: Some(exec_sandbox_event::Payload::Exit(proto::ExecSandboxExit { + exit_code: code, + })), + } +} + +impl MockGateway { + async fn exchange( + self, + mut input: tonic::Streaming, + output: mpsc::Sender>, + ) { + let Some(exec_sandbox_input::Payload::Start(start)) = input + .message() + .await + .expect("read start frame") + .expect("start frame") + .payload + else { + panic!("first frame must start the command"); + }; + assert!(!start.tty, "streaming pipes must not allocate a TTY"); + assert!(start.stdin.is_empty(), "stdin must follow the start frame"); + self.calls.lock().unwrap().starts.push(start); + + match self.scenario { + Scenario::EarlyExit | Scenario::ErrorAfterExit => { + let _ = output.send(Ok(stdout(b"before-exit\n".to_vec()))).await; + let _ = output.send(Ok(exit(0))).await; + if matches!(self.scenario, Scenario::ErrorAfterExit) { + send_trailer_error(&output).await; + } + return; + } + Scenario::Echo | Scenario::ErrorAfterInput => {} + } + + loop { + let message = tokio::select! { + biased; + () = output.closed() => return, + message = input.message() => message, + }; + match message { + Ok(Some(frame)) => match frame.payload { + Some(exec_sandbox_input::Payload::Stdin(bytes)) => { + self.calls.lock().unwrap().input_bytes += bytes.len(); + if matches!(self.scenario, Scenario::Echo) + && output.send(Ok(stdout(bytes))).await.is_err() + { + return; + } + } + None => return, + unexpected => panic!("unexpected input after start: {unexpected:?}"), + }, + Ok(None) => break, + Err(_) => return, + } + } + + match self.scenario { + Scenario::Echo => { + let _ = output.send(Ok(stdout(b"after-eof\n".to_vec()))).await; + let _ = output + .send(Ok(proto::ExecSandboxEvent { + payload: Some(exec_sandbox_event::Payload::Stderr( + proto::ExecSandboxStderr { + data: b"remote-stderr\n".to_vec(), + }, + )), + })) + .await; + let _ = output.send(Ok(exit(7))).await; + } + Scenario::ErrorAfterInput => { + let _ = output.send(Ok(exit(0))).await; + send_trailer_error(&output).await; + } + _ => unreachable!(), + } + } +} + +async fn send_trailer_error(output: &mpsc::Sender>) { + // Separate the Exit message from the failing final gRPC status. A client + // that stops at Exit reports success before this failure arrives. + tokio::time::sleep(Duration::from_millis(100)).await; + let _ = output + .send(Err(Status::internal("failure after exit"))) + .await; +} + +// Generate unused trait methods inside the async_trait expansion so this mock +// implements only the RPC behavior under test without hand-written boilerplate. +macro_rules! mock_gateway { + ( + unary { $( $method:ident($request:ty) -> $response:ty; )* } + client_stream { $( $client_method:ident($client_request:ty) -> $client_response:ty; )* } + server_stream { $( $server_method:ident($server_request:ty) -> $stream_type:ident($server_response:ty); )* } + bidi { $( $bidi_method:ident($bidi_request:ty) -> $bidi_type:ident($bidi_response:ty); )* } + ) => { + #[tonic::async_trait] + impl OpenShell for MockGateway { + $(async fn $method(&self, _: Request<$request>) -> Result, Status> { + Err(Status::unimplemented("unused test RPC")) + })* + $(async fn $client_method(&self, _: Request>) -> Result, Status> { + Err(Status::unimplemented("unused test RPC")) + })* + $(type $stream_type = ReceiverStream>; + async fn $server_method(&self, _: Request<$server_request>) -> Result, Status> { + Err(Status::unimplemented("unused test RPC")) + })* + $(type $bidi_type = ReceiverStream>; + async fn $bidi_method(&self, _: Request>) -> Result, Status> { + Err(Status::unimplemented("unused test RPC")) + })* + + async fn get_sandbox(&self, _: Request) -> Result, Status> { + self.calls.lock().unwrap().lookups += 1; + Ok(Response::new(proto::SandboxResponse { + sandbox: Some(proto::Sandbox { + metadata: Some(proto::datamodel::v1::ObjectMeta { + id: "test-id".to_string(), + name: "test-sandbox".to_string(), + workspace: "default".to_string(), + ..Default::default() + }), + status: Some(proto::SandboxStatus { + phase: proto::SandboxPhase::Ready.into(), + ..Default::default() + }), + ..Default::default() + }), + ..Default::default() + })) + } + + type ExecSandboxStream = EventStream; + async fn exec_sandbox(&self, _: Request) -> Result, Status> { + self.calls.lock().unwrap().unary += 1; + Err(Status::unimplemented("test requires streaming exec")) + } + + type ExecSandboxInteractiveStream = EventStream; + async fn exec_sandbox_interactive(&self, request: Request>) -> Result, Status> { + let (sender, receiver) = mpsc::channel(1); + let service = self.clone(); + tokio::spawn(async move { + service.exchange(request.into_inner(), sender).await; + }); + Ok(Response::new(ReceiverStream::new(receiver))) + } + } + }; +} + +mock_gateway! { + unary { + health(proto::HealthRequest) -> proto::HealthResponse; + get_current_user(proto::GetCurrentUserRequest) -> proto::GetCurrentUserResponse; + get_gateway_info(proto::GetGatewayInfoRequest) -> proto::GetGatewayInfoResponse; + create_sandbox(proto::CreateSandboxRequest) -> proto::SandboxResponse; + begin_rootfs_tar_staging(proto::BeginRootfsTarStagingRequest) -> proto::BeginRootfsTarStagingResponse; + list_sandboxes(proto::ListSandboxesRequest) -> proto::ListSandboxesResponse; + create_sandbox_template(proto::CreateSandboxTemplateRequest) -> proto::SandboxTemplateResponse; + get_sandbox_template(proto::GetSandboxTemplateRequest) -> proto::SandboxTemplateResponse; + list_sandbox_templates(proto::ListSandboxTemplatesRequest) -> proto::ListSandboxTemplatesResponse; + delete_sandbox_template(proto::DeleteSandboxTemplateRequest) -> proto::DeleteSandboxTemplateResponse; + list_sandbox_providers(proto::ListSandboxProvidersRequest) -> proto::ListSandboxProvidersResponse; + attach_sandbox_provider(proto::AttachSandboxProviderRequest) -> proto::AttachSandboxProviderResponse; + detach_sandbox_provider(proto::DetachSandboxProviderRequest) -> proto::DetachSandboxProviderResponse; + get_sandbox_provider_status(proto::GetSandboxProviderStatusRequest) -> proto::GetSandboxProviderStatusResponse; + delete_sandbox(proto::DeleteSandboxRequest) -> proto::DeleteSandboxResponse; + stop_sandbox(proto::StopSandboxRequest) -> proto::SandboxResponse; + start_sandbox(proto::StartSandboxRequest) -> proto::SandboxResponse; + create_ssh_session(proto::CreateSshSessionRequest) -> proto::CreateSshSessionResponse; + expose_service(proto::ExposeServiceRequest) -> proto::ServiceEndpointResponse; + get_service(proto::GetServiceRequest) -> proto::ServiceEndpointResponse; + list_services(proto::ListServicesRequest) -> proto::ListServicesResponse; + delete_service(proto::DeleteServiceRequest) -> proto::DeleteServiceResponse; + revoke_ssh_session(proto::RevokeSshSessionRequest) -> proto::RevokeSshSessionResponse; + create_provider(proto::CreateProviderRequest) -> proto::ProviderResponse; + get_provider(proto::GetProviderRequest) -> proto::ProviderResponse; + list_providers(proto::ListProvidersRequest) -> proto::ListProvidersResponse; + list_provider_profiles(proto::ListProviderProfilesRequest) -> proto::ListProviderProfilesResponse; + get_provider_profile(proto::GetProviderProfileRequest) -> proto::ProviderProfileResponse; + import_provider_profiles(proto::ImportProviderProfilesRequest) -> proto::ImportProviderProfilesResponse; + update_provider_profiles(proto::UpdateProviderProfilesRequest) -> proto::UpdateProviderProfilesResponse; + lint_provider_profiles(proto::LintProviderProfilesRequest) -> proto::LintProviderProfilesResponse; + update_provider(proto::UpdateProviderRequest) -> proto::ProviderResponse; + get_provider_refresh_status(proto::GetProviderRefreshStatusRequest) -> proto::GetProviderRefreshStatusResponse; + configure_provider_refresh(proto::ConfigureProviderRefreshRequest) -> proto::ConfigureProviderRefreshResponse; + rotate_provider_credential(proto::RotateProviderCredentialRequest) -> proto::RotateProviderCredentialResponse; + delete_provider_refresh(proto::DeleteProviderRefreshRequest) -> proto::DeleteProviderRefreshResponse; + delete_provider(proto::DeleteProviderRequest) -> proto::DeleteProviderResponse; + delete_provider_profile(proto::DeleteProviderProfileRequest) -> proto::DeleteProviderProfileResponse; + get_sandbox_config(proto::GetSandboxConfigRequest) -> proto::GetSandboxConfigResponse; + get_gateway_config(proto::GetGatewayConfigRequest) -> proto::GetGatewayConfigResponse; + update_config(proto::UpdateConfigRequest) -> proto::UpdateConfigResponse; + get_sandbox_policy_status(proto::GetSandboxPolicyStatusRequest) -> proto::GetSandboxPolicyStatusResponse; + list_sandbox_policies(proto::ListSandboxPoliciesRequest) -> proto::ListSandboxPoliciesResponse; + report_policy_status(proto::ReportPolicyStatusRequest) -> proto::ReportPolicyStatusResponse; + report_endpoint_status(proto::ReportEndpointStatusRequest) -> proto::ReportEndpointStatusResponse; + report_provider_readiness(proto::ReportProviderReadinessRequest) -> proto::ReportProviderReadinessResponse; + report_sandbox_configuration(proto::ReportSandboxConfigurationRequest) -> proto::ReportSandboxConfigurationResponse; + get_sandbox_provider_environment(proto::GetSandboxProviderEnvironmentRequest) -> proto::GetSandboxProviderEnvironmentResponse; + exchange_provider_subject_token(proto::ExchangeProviderSubjectTokenRequest) -> proto::ExchangeProviderSubjectTokenResponse; + get_sandbox_logs(proto::GetSandboxLogsRequest) -> proto::GetSandboxLogsResponse; + report_main_process_exit(proto::ReportMainProcessExitRequest) -> proto::ReportMainProcessExitResponse; + finalize_main_process_exit(proto::FinalizeMainProcessExitRequest) -> proto::FinalizeMainProcessExitResponse; + peer_report_provider_readiness(proto::ReportProviderReadinessRequest) -> proto::ReportProviderReadinessResponse; + peer_report_endpoint_status(proto::ReportEndpointStatusRequest) -> proto::ReportEndpointStatusResponse; + peer_get_sandbox_provider_status(proto::GetSandboxProviderStatusRequest) -> proto::GetSandboxProviderStatusResponse; + submit_policy_analysis(proto::SubmitPolicyAnalysisRequest) -> proto::SubmitPolicyAnalysisResponse; + get_draft_policy(proto::GetDraftPolicyRequest) -> proto::GetDraftPolicyResponse; + approve_draft_chunk(proto::ApproveDraftChunkRequest) -> proto::ApproveDraftChunkResponse; + reject_draft_chunk(proto::RejectDraftChunkRequest) -> proto::RejectDraftChunkResponse; + approve_all_draft_chunks(proto::ApproveAllDraftChunksRequest) -> proto::ApproveAllDraftChunksResponse; + edit_draft_chunk(proto::EditDraftChunkRequest) -> proto::EditDraftChunkResponse; + undo_draft_chunk(proto::UndoDraftChunkRequest) -> proto::UndoDraftChunkResponse; + clear_draft_chunks(proto::ClearDraftChunksRequest) -> proto::ClearDraftChunksResponse; + get_draft_history(proto::GetDraftHistoryRequest) -> proto::GetDraftHistoryResponse; + issue_sandbox_token(proto::IssueSandboxTokenRequest) -> proto::IssueSandboxTokenResponse; + refresh_sandbox_token(proto::RefreshSandboxTokenRequest) -> proto::RefreshSandboxTokenResponse; + create_workspace(proto::CreateWorkspaceRequest) -> proto::CreateWorkspaceResponse; + get_workspace(proto::GetWorkspaceRequest) -> proto::GetWorkspaceResponse; + list_workspaces(proto::ListWorkspacesRequest) -> proto::ListWorkspacesResponse; + delete_workspace(proto::DeleteWorkspaceRequest) -> proto::DeleteWorkspaceResponse; + add_workspace_member(proto::AddWorkspaceMemberRequest) -> proto::AddWorkspaceMemberResponse; + remove_workspace_member(proto::RemoveWorkspaceMemberRequest) -> proto::RemoveWorkspaceMemberResponse; + list_workspace_members(proto::ListWorkspaceMembersRequest) -> proto::ListWorkspaceMembersResponse; + } + client_stream { + push_sandbox_logs(proto::PushSandboxLogsRequest) -> proto::PushSandboxLogsResponse; + } + server_stream { + watch_sandbox(proto::WatchSandboxRequest) -> WatchSandboxStream(proto::SandboxStreamEvent); + } + bidi { + forward_tcp(proto::TcpForwardFrame) -> ForwardTcpStream(proto::TcpForwardFrame); + connect_supervisor(proto::SupervisorMessage) -> ConnectSupervisorStream(proto::GatewayMessage); + relay_stream(proto::RelayFrame) -> RelayStreamStream(proto::RelayFrame); + peer_relay(proto::PeerRelayFrame) -> PeerRelayStream(proto::PeerRelayFrame); + } +} + +struct TestGateway { + endpoint: String, + config: tempfile::TempDir, + calls: Arc>, + task: JoinHandle<()>, +} + +impl Drop for TestGateway { + fn drop(&mut self) { + self.task.abort(); + } +} + +impl TestGateway { + async fn start(scenario: Scenario) -> Self { + let (ca, ca_key) = build_ca(); + let (server_cert, server_key) = build_server_cert(&ca, &ca_key); + let (client_cert, client_key) = build_client_cert(&ca, &ca_key); + let config = tempfile::tempdir().unwrap(); + let certs = config.path().join("openshell/gateways/test-gateway/mtls"); + std::fs::create_dir_all(&certs).unwrap(); + std::fs::write(certs.join("ca.crt"), ca.pem()).unwrap(); + std::fs::write(certs.join("tls.crt"), client_cert).unwrap(); + std::fs::write(certs.join("tls.key"), client_key).unwrap(); + let tls = ServerTlsConfig::new() + .identity(Identity::from_pem(server_cert, server_key)) + .client_ca_root(Certificate::from_pem(ca.pem())); + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let endpoint = format!( + "https://localhost:{}", + listener.local_addr().unwrap().port() + ); + let calls = Arc::new(Mutex::new(Calls::default())); + let service = MockGateway { + scenario, + calls: Arc::clone(&calls), + }; + let task = tokio::spawn(async move { + Server::builder() + .tls_config(tls) + .unwrap() + .add_service(OpenShellServer::new(service)) + .serve_with_incoming(TcpListenerStream::new(listener)) + .await + .unwrap(); + }); + Self { + endpoint, + config, + calls, + task, + } + } + + fn command(&self) -> Command { + // Run the same subprocess regression against an official CLI when + // establishing that the unfixed version reports the wrong result. + let baseline = std::env::var_os("OPENSHELL_STREAMING_TEST_BASELINE_CLI"); + let executable = baseline + .as_deref() + .map_or_else(|| Path::new(env!("CARGO_BIN_EXE_openshell")), Path::new); + let mut command = Command::new(executable); + command.args([ + "--gateway", + "test-gateway", + "--gateway-endpoint", + &self.endpoint, + "--workspace", + "default", + "--color", + "never", + "sandbox", + "exec", + "--name", + "test-sandbox", + ]); + command + .args(["--no-tty", "--no-login-shell", "--", "test-command"]) + .env("XDG_CONFIG_HOME", self.config.path()) + .env( + "OPENSHELL_SYSTEM_GATEWAY_DIR", + self.config.path().join("system"), + ) + .env_remove("OPENSHELL_GATEWAY") + .env_remove("OPENSHELL_GATEWAY_ENDPOINT") + .env_remove("OPENSHELL_GATEWAY_INSECURE") + .env_remove("RUST_LOG") + .stdin(Stdio::piped()) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()) + .kill_on_drop(true); + command + } + + fn spawn(&self) -> Child { + self.command().spawn().unwrap() + } + + fn assert_one_stream(&self) { + let calls = self.calls.lock().unwrap(); + assert_eq!(calls.lookups, 1); + assert_eq!(calls.starts.len(), 1, "a command must not be relaunched"); + assert_eq!(calls.unary, 0, "test input must select streaming exec"); + } +} + +async fn finish(child: Child) -> Output { + timeout(DEADLINE, child.wait_with_output()) + .await + .expect("CLI did not terminate before the deadline") + .unwrap() +} + +async fn send_input(child: &mut Child, input: &[u8]) { + let mut stdin = child.stdin.take().unwrap(); + // An oversized input or failed relay may close the pipe before all bytes + // are written. The subprocess status and RPC observations decide success. + let _ = timeout(DEADLINE, stdin.write_all(input)) + .await + .expect("stdin write timed out"); +} + +#[tokio::test] +async fn default_open_pipe_exchanges_input_after_grace_and_closes_cleanly() { + let gateway = TestGateway::start(Scenario::Echo).await; + let mut child = gateway.spawn(); + let mut stdin = child.stdin.take().unwrap(); + let mut stdout = BufReader::new(child.stdout.take().unwrap()); + for request in ["before grace\n", "after grace\n"] { + stdin.write_all(request.as_bytes()).await.unwrap(); + let mut response = String::new(); + timeout(DEADLINE, stdout.read_line(&mut response)) + .await + .expect("default exec must respond before stdin EOF") + .unwrap(); + assert_eq!(response, request); + } + drop(stdin); + let mut final_stdout = String::new(); + timeout(DEADLINE, stdout.read_to_string(&mut final_stdout)) + .await + .unwrap() + .unwrap(); + let output = finish(child).await; + assert_eq!(output.status.code(), Some(7)); + assert_eq!(final_stdout, "after-eof\n"); + assert_eq!(output.stderr, b"remote-stderr\n"); + gateway.assert_one_stream(); +} + +#[tokio::test] +async fn default_remote_exit_does_not_wait_for_idle_open_stdin() { + let gateway = TestGateway::start(Scenario::EarlyExit).await; + let mut child = gateway.spawn(); + let held_open = child.stdin.take().unwrap(); + let output = finish(child).await; + assert!( + output.status.success(), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + assert_eq!(output.stdout, b"before-exit\n"); + assert!(output.stderr.is_empty()); + gateway.assert_one_stream(); + drop(held_open); +} + +#[tokio::test] +async fn default_open_pipe_checks_trailer_error_after_exit() { + let gateway = TestGateway::start(Scenario::ErrorAfterExit).await; + let mut child = gateway.spawn(); + let held_open = child.stdin.take().unwrap(); + let output = finish(child).await; + gateway.assert_one_stream(); + assert!( + !output.status.success(), + "Exit must not hide a failing trailer" + ); + assert_eq!(output.stdout, b"before-exit\n"); + assert!(String::from_utf8_lossy(&output.stderr).contains("failure after exit")); + drop(held_open); +} + +#[tokio::test] +async fn default_large_finite_input_checks_trailer_error_after_exit() { + let gateway = TestGateway::start(Scenario::ErrorAfterInput).await; + // All bytes and EOF are available before launch, so collection need not + // wait for a pipe writer. Metadata pushes this request above 1 MiB and + // selects streaming exec even when input finishes within the grace period. + let input_size = 1024 * 1024; + let input_path = gateway.config.path().join("stdin"); + std::fs::write(&input_path, vec![b'x'; input_size]).unwrap(); + let child = gateway + .command() + .stdin(Stdio::from(std::fs::File::open(input_path).unwrap())) + .spawn() + .unwrap(); + let output = finish(child).await; + gateway.assert_one_stream(); + assert_eq!(gateway.calls.lock().unwrap().input_bytes, input_size); + assert!( + !output.status.success(), + "Exit must not hide a failing trailer" + ); + assert!(String::from_utf8_lossy(&output.stderr).contains("failure after exit")); +} + +#[tokio::test] +async fn default_open_pipe_overflow_warns_after_forwarding_prefix() { + let gateway = TestGateway::start(Scenario::Echo).await; + let mut child = gateway.spawn(); + let mut stdout = BufReader::new(child.stdout.take().unwrap()); + child + .stdin + .as_mut() + .unwrap() + .write_all(b"prefix\n") + .await + .unwrap(); + let mut response = String::new(); + timeout(DEADLINE, stdout.read_line(&mut response)) + .await + .expect("prefix must reach the command before sending overflow") + .unwrap(); + assert_eq!(response, "prefix\n"); + // Drain echo output concurrently so stdout backpressure cannot prevent + // the input writer from reaching the cumulative prefix-and-remainder cap. + let drain = tokio::spawn(async move { + let mut bytes = Vec::new(); + stdout.read_to_end(&mut bytes).await.unwrap(); + }); + send_input(&mut child, &vec![b'x'; STDIN_LIMIT]).await; + let output = finish(child).await; + assert!(!output.status.success()); + let stderr = String::from_utf8_lossy(&output.stderr); + assert!( + stderr.contains("streamed stdin exceeds the 4 MiB limit"), + "{stderr}" + ); + assert!(stderr.contains("partial input"), "{stderr}"); + timeout(DEADLINE, drain).await.unwrap().unwrap(); + gateway.assert_one_stream(); + assert!(gateway.calls.lock().unwrap().input_bytes <= STDIN_LIMIT); +} diff --git a/docs/how-it-works/sandboxes/overview.mdx b/docs/how-it-works/sandboxes/overview.mdx index c39c71a3a4..e38f925327 100644 --- a/docs/how-it-works/sandboxes/overview.mdx +++ b/docs/how-it-works/sandboxes/overview.mdx @@ -334,6 +334,8 @@ as it arrives. The CLI closes remote stdin when the pipe reaches EOF and continues reading command output until the command finishes. Redirect stdin from `/dev/null` when a command needs no input. +If streamed stdin exceeds the 4 MiB limit, the CLI reports an error and warns that the command may already have processed partial input. Cancellation does not undo that work. + The command's exit code is propagated to the CLI, so `exec` works in scripts that check return codes. Large stdout and stderr streams are delivered before a successful exit. A slow reader backpressures the command. If output delivery fails, `exec` returns a @@ -341,6 +343,8 @@ failure instead of reporting success with incomplete output. If a background process keeps stdout or stderr open for more than 30 seconds after the command exits, `exec` reports an output delivery failure. +A failed final gRPC status also makes the CLI report failure, even if the remote command reported exit code zero. The CLI does not automatically retry the execution. + Run an interactive shell with a TTY: ```shell From a91d2e22e8966772b1072c6edea56f7149e1b300 Mon Sep 17 00:00:00 2001 From: "red-hat-konflux[bot]" <126015336+red-hat-konflux[bot]@users.noreply.github.com> Date: Sat, 3 Oct 2026 04:32:38 +0000 Subject: [PATCH 04/13] chore(deps): refresh rpm lockfiles (#76) Signed-off-by: red-hat-konflux <126015336+red-hat-konflux[bot]@users.noreply.github.com> Co-authored-by: red-hat-konflux[bot] <126015336+red-hat-konflux[bot]@users.noreply.github.com> --- deploy/konflux/cli/rpms.lock.yaml | 20 ++++++++++---------- deploy/konflux/e2e-odh/rpms.lock.yaml | 20 ++++++++++---------- deploy/konflux/gateway/rpms.lock.yaml | 20 ++++++++++---------- deploy/konflux/sandbox/rpms.lock.yaml | 20 ++++++++++---------- deploy/konflux/supervisor/rpms.lock.yaml | 20 ++++++++++---------- 5 files changed, 50 insertions(+), 50 deletions(-) diff --git a/deploy/konflux/cli/rpms.lock.yaml b/deploy/konflux/cli/rpms.lock.yaml index d7b4fc100f..f8adfb25d3 100644 --- a/deploy/konflux/cli/rpms.lock.yaml +++ b/deploy/konflux/cli/rpms.lock.yaml @@ -74,13 +74,13 @@ arches: name: glibc-devel evr: 2.34-275.el9_8 sourcerpm: glibc-2.34-275.el9_8.src.rpm - - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/aarch64/appstream/os/Packages/k/kernel-headers-5.14.0-687.53.1.el9_8.aarch64.rpm + - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/aarch64/appstream/os/Packages/k/kernel-headers-5.14.0-687.54.1.el9_8.aarch64.rpm repoid: ubi-9-for-aarch64-appstream-rpms - size: 2925601 - checksum: sha256:7e82c856a00206976aede9e7b3865d0a2363b1262c0f9e1949ba29c7b9c47281 + size: 2927565 + checksum: sha256:c950319c04fc9048501cd59da158858b9e4b2d173b98b3d8f56a97b56be89ca9 name: kernel-headers - evr: 5.14.0-687.53.1.el9_8 - sourcerpm: kernel-5.14.0-687.53.1.el9_8.src.rpm + evr: 5.14.0-687.54.1.el9_8 + sourcerpm: kernel-5.14.0-687.54.1.el9_8.src.rpm - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/aarch64/appstream/os/Packages/l/libasan-11.5.0-14.el9.aarch64.rpm repoid: ubi-9-for-aarch64-appstream-rpms size: 409047 @@ -1128,13 +1128,13 @@ arches: name: glibc-headers evr: 2.34-275.el9_8 sourcerpm: glibc-2.34-275.el9_8.src.rpm - - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/x86_64/appstream/os/Packages/k/kernel-headers-5.14.0-687.53.1.el9_8.x86_64.rpm + - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/x86_64/appstream/os/Packages/k/kernel-headers-5.14.0-687.54.1.el9_8.x86_64.rpm repoid: ubi-9-for-x86_64-appstream-rpms - size: 2965145 - checksum: sha256:2473df2cf5b65c762af7941fe87324e98dc0e8b42d1e5745051a0405e200b7f5 + size: 2966865 + checksum: sha256:33789c61f2d74ef1d0fd5b6a43e802d28e1b38a605f200b156050d0a7934f232 name: kernel-headers - evr: 5.14.0-687.53.1.el9_8 - sourcerpm: kernel-5.14.0-687.53.1.el9_8.src.rpm + evr: 5.14.0-687.54.1.el9_8 + sourcerpm: kernel-5.14.0-687.54.1.el9_8.src.rpm - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/x86_64/appstream/os/Packages/l/libmpc-1.2.1-4.el9.x86_64.rpm repoid: ubi-9-for-x86_64-appstream-rpms size: 66075 diff --git a/deploy/konflux/e2e-odh/rpms.lock.yaml b/deploy/konflux/e2e-odh/rpms.lock.yaml index cae27e5345..59fc71bea8 100644 --- a/deploy/konflux/e2e-odh/rpms.lock.yaml +++ b/deploy/konflux/e2e-odh/rpms.lock.yaml @@ -67,13 +67,13 @@ arches: name: glibc-devel evr: 2.34-275.el9_8 sourcerpm: glibc-2.34-275.el9_8.src.rpm - - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/aarch64/appstream/os/Packages/k/kernel-headers-5.14.0-687.53.1.el9_8.aarch64.rpm + - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/aarch64/appstream/os/Packages/k/kernel-headers-5.14.0-687.54.1.el9_8.aarch64.rpm repoid: ubi-9-appstream-rpms - size: 2925601 - checksum: sha256:7e82c856a00206976aede9e7b3865d0a2363b1262c0f9e1949ba29c7b9c47281 + size: 2927565 + checksum: sha256:c950319c04fc9048501cd59da158858b9e4b2d173b98b3d8f56a97b56be89ca9 name: kernel-headers - evr: 5.14.0-687.53.1.el9_8 - sourcerpm: kernel-5.14.0-687.53.1.el9_8.src.rpm + evr: 5.14.0-687.54.1.el9_8 + sourcerpm: kernel-5.14.0-687.54.1.el9_8.src.rpm - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/aarch64/appstream/os/Packages/l/libasan-11.5.0-14.el9.aarch64.rpm repoid: ubi-9-appstream-rpms size: 409047 @@ -1093,13 +1093,13 @@ arches: name: glibc-headers evr: 2.34-275.el9_8 sourcerpm: glibc-2.34-275.el9_8.src.rpm - - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/x86_64/appstream/os/Packages/k/kernel-headers-5.14.0-687.53.1.el9_8.x86_64.rpm + - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/x86_64/appstream/os/Packages/k/kernel-headers-5.14.0-687.54.1.el9_8.x86_64.rpm repoid: ubi-9-appstream-rpms - size: 2965145 - checksum: sha256:2473df2cf5b65c762af7941fe87324e98dc0e8b42d1e5745051a0405e200b7f5 + size: 2966865 + checksum: sha256:33789c61f2d74ef1d0fd5b6a43e802d28e1b38a605f200b156050d0a7934f232 name: kernel-headers - evr: 5.14.0-687.53.1.el9_8 - sourcerpm: kernel-5.14.0-687.53.1.el9_8.src.rpm + evr: 5.14.0-687.54.1.el9_8 + sourcerpm: kernel-5.14.0-687.54.1.el9_8.src.rpm - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/x86_64/appstream/os/Packages/l/libmpc-1.2.1-4.el9.x86_64.rpm repoid: ubi-9-appstream-rpms size: 66075 diff --git a/deploy/konflux/gateway/rpms.lock.yaml b/deploy/konflux/gateway/rpms.lock.yaml index 166160031d..04c8e58fa6 100644 --- a/deploy/konflux/gateway/rpms.lock.yaml +++ b/deploy/konflux/gateway/rpms.lock.yaml @@ -214,13 +214,13 @@ arches: name: glibc-devel evr: 2.34-275.el9_8 sourcerpm: glibc-2.34-275.el9_8.src.rpm - - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/aarch64/appstream/os/Packages/k/kernel-headers-5.14.0-687.53.1.el9_8.aarch64.rpm + - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/aarch64/appstream/os/Packages/k/kernel-headers-5.14.0-687.54.1.el9_8.aarch64.rpm repoid: ubi-9-for-aarch64-appstream-rpms - size: 2925601 - checksum: sha256:7e82c856a00206976aede9e7b3865d0a2363b1262c0f9e1949ba29c7b9c47281 + size: 2927565 + checksum: sha256:c950319c04fc9048501cd59da158858b9e4b2d173b98b3d8f56a97b56be89ca9 name: kernel-headers - evr: 5.14.0-687.53.1.el9_8 - sourcerpm: kernel-5.14.0-687.53.1.el9_8.src.rpm + evr: 5.14.0-687.54.1.el9_8 + sourcerpm: kernel-5.14.0-687.54.1.el9_8.src.rpm - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/aarch64/appstream/os/Packages/l/libasan-11.5.0-14.el9.aarch64.rpm repoid: ubi-9-for-aarch64-appstream-rpms size: 409047 @@ -1597,13 +1597,13 @@ arches: name: glibc-headers evr: 2.34-275.el9_8 sourcerpm: glibc-2.34-275.el9_8.src.rpm - - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/x86_64/appstream/os/Packages/k/kernel-headers-5.14.0-687.53.1.el9_8.x86_64.rpm + - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/x86_64/appstream/os/Packages/k/kernel-headers-5.14.0-687.54.1.el9_8.x86_64.rpm repoid: ubi-9-for-x86_64-appstream-rpms - size: 2965145 - checksum: sha256:2473df2cf5b65c762af7941fe87324e98dc0e8b42d1e5745051a0405e200b7f5 + size: 2966865 + checksum: sha256:33789c61f2d74ef1d0fd5b6a43e802d28e1b38a605f200b156050d0a7934f232 name: kernel-headers - evr: 5.14.0-687.53.1.el9_8 - sourcerpm: kernel-5.14.0-687.53.1.el9_8.src.rpm + evr: 5.14.0-687.54.1.el9_8 + sourcerpm: kernel-5.14.0-687.54.1.el9_8.src.rpm - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/x86_64/appstream/os/Packages/l/libedit-devel-3.1-39.20210216cvs.el9.x86_64.rpm repoid: ubi-9-for-x86_64-appstream-rpms size: 54318 diff --git a/deploy/konflux/sandbox/rpms.lock.yaml b/deploy/konflux/sandbox/rpms.lock.yaml index 0c1a82bb34..7f424d7551 100644 --- a/deploy/konflux/sandbox/rpms.lock.yaml +++ b/deploy/konflux/sandbox/rpms.lock.yaml @@ -67,13 +67,13 @@ arches: name: glibc-devel evr: 2.34-275.el9_8 sourcerpm: glibc-2.34-275.el9_8.src.rpm - - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/aarch64/appstream/os/Packages/k/kernel-headers-5.14.0-687.53.1.el9_8.aarch64.rpm + - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/aarch64/appstream/os/Packages/k/kernel-headers-5.14.0-687.54.1.el9_8.aarch64.rpm repoid: ubi-9-for-aarch64-appstream-rpms - size: 2925601 - checksum: sha256:7e82c856a00206976aede9e7b3865d0a2363b1262c0f9e1949ba29c7b9c47281 + size: 2927565 + checksum: sha256:c950319c04fc9048501cd59da158858b9e4b2d173b98b3d8f56a97b56be89ca9 name: kernel-headers - evr: 5.14.0-687.53.1.el9_8 - sourcerpm: kernel-5.14.0-687.53.1.el9_8.src.rpm + evr: 5.14.0-687.54.1.el9_8 + sourcerpm: kernel-5.14.0-687.54.1.el9_8.src.rpm - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/aarch64/appstream/os/Packages/l/libasan-11.5.0-14.el9.aarch64.rpm repoid: ubi-9-for-aarch64-appstream-rpms size: 409047 @@ -1009,13 +1009,13 @@ arches: name: glibc-headers evr: 2.34-275.el9_8 sourcerpm: glibc-2.34-275.el9_8.src.rpm - - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/x86_64/appstream/os/Packages/k/kernel-headers-5.14.0-687.53.1.el9_8.x86_64.rpm + - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/x86_64/appstream/os/Packages/k/kernel-headers-5.14.0-687.54.1.el9_8.x86_64.rpm repoid: ubi-9-for-x86_64-appstream-rpms - size: 2965145 - checksum: sha256:2473df2cf5b65c762af7941fe87324e98dc0e8b42d1e5745051a0405e200b7f5 + size: 2966865 + checksum: sha256:33789c61f2d74ef1d0fd5b6a43e802d28e1b38a605f200b156050d0a7934f232 name: kernel-headers - evr: 5.14.0-687.53.1.el9_8 - sourcerpm: kernel-5.14.0-687.53.1.el9_8.src.rpm + evr: 5.14.0-687.54.1.el9_8 + sourcerpm: kernel-5.14.0-687.54.1.el9_8.src.rpm - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/x86_64/appstream/os/Packages/l/libmpc-1.2.1-4.el9.x86_64.rpm repoid: ubi-9-for-x86_64-appstream-rpms size: 66075 diff --git a/deploy/konflux/supervisor/rpms.lock.yaml b/deploy/konflux/supervisor/rpms.lock.yaml index c1b2aaf53b..77c2c2d14f 100644 --- a/deploy/konflux/supervisor/rpms.lock.yaml +++ b/deploy/konflux/supervisor/rpms.lock.yaml @@ -67,13 +67,13 @@ arches: name: glibc-devel evr: 2.34-275.el9_8 sourcerpm: glibc-2.34-275.el9_8.src.rpm - - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/aarch64/appstream/os/Packages/k/kernel-headers-5.14.0-687.53.1.el9_8.aarch64.rpm + - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/aarch64/appstream/os/Packages/k/kernel-headers-5.14.0-687.54.1.el9_8.aarch64.rpm repoid: ubi-9-for-aarch64-appstream-rpms - size: 2925601 - checksum: sha256:7e82c856a00206976aede9e7b3865d0a2363b1262c0f9e1949ba29c7b9c47281 + size: 2927565 + checksum: sha256:c950319c04fc9048501cd59da158858b9e4b2d173b98b3d8f56a97b56be89ca9 name: kernel-headers - evr: 5.14.0-687.53.1.el9_8 - sourcerpm: kernel-5.14.0-687.53.1.el9_8.src.rpm + evr: 5.14.0-687.54.1.el9_8 + sourcerpm: kernel-5.14.0-687.54.1.el9_8.src.rpm - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/aarch64/appstream/os/Packages/l/libasan-11.5.0-14.el9.aarch64.rpm repoid: ubi-9-for-aarch64-appstream-rpms size: 409047 @@ -1037,13 +1037,13 @@ arches: name: glibc-headers evr: 2.34-275.el9_8 sourcerpm: glibc-2.34-275.el9_8.src.rpm - - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/x86_64/appstream/os/Packages/k/kernel-headers-5.14.0-687.53.1.el9_8.x86_64.rpm + - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/x86_64/appstream/os/Packages/k/kernel-headers-5.14.0-687.54.1.el9_8.x86_64.rpm repoid: ubi-9-for-x86_64-appstream-rpms - size: 2965145 - checksum: sha256:2473df2cf5b65c762af7941fe87324e98dc0e8b42d1e5745051a0405e200b7f5 + size: 2966865 + checksum: sha256:33789c61f2d74ef1d0fd5b6a43e802d28e1b38a605f200b156050d0a7934f232 name: kernel-headers - evr: 5.14.0-687.53.1.el9_8 - sourcerpm: kernel-5.14.0-687.53.1.el9_8.src.rpm + evr: 5.14.0-687.54.1.el9_8 + sourcerpm: kernel-5.14.0-687.54.1.el9_8.src.rpm - url: https://cdn-ubi.redhat.com/content/public/ubi/dist/ubi9/9/x86_64/appstream/os/Packages/l/libmpc-1.2.1-4.el9.x86_64.rpm repoid: ubi-9-for-x86_64-appstream-rpms size: 66075 From adcd28b620e68d62f2d12c3caea4c70662ec6330 Mon Sep 17 00:00:00 2001 From: Eric Curtin Date: Sat, 3 Oct 2026 19:59:57 +0000 Subject: [PATCH 05/13] fix(install): stop waiting when the gateway service fails (#4143) Fixes #4040 Signed-off-by: Eric Curtin --- install.sh | 24 ++++++++- tasks/scripts/test-install-sh.sh | 91 ++++++++++++++++++++++++++++++++ 2 files changed, 114 insertions(+), 1 deletion(-) diff --git a/install.sh b/install.sh index 83b7c479be..7cc8637f6e 100755 --- a/install.sh +++ b/install.sh @@ -935,10 +935,21 @@ start_user_gateway() { info "registering local gateway as ${TARGET_USER}..." register_local_gateway - wait_for_local_gateway_listener + wait_for_local_gateway_listener user_gateway_service_failed wait_for_local_gateway_status } +# Succeeds when the gateway user service has failed or is waiting to restart +# after a failure. A unit that is starting or running does not match, even if +# it failed before it was restarted. +user_gateway_service_failed() { + _unit_state="$(as_target_user systemctl --user show openshell-gateway -p ActiveState -p SubState 2>/dev/null)" || return 1 + case "$_unit_state" in + *ActiveState=failed* | *SubState=auto-restart*) return 0 ;; + esac + return 1 +} + dump_local_gateway_diagnostics() { _lines="${OPENSHELL_INSTALL_LOG_LINES:-80}" case "$_lines" in @@ -1015,10 +1026,14 @@ dump_user_service_gateway_diagnostics() { fi } +# An optional command name stops the wait early when it succeeds, so a service +# that already failed does not run out the full timeout. wait_for_local_gateway_listener() { + _failed_check="${1:-}" _timeout="${OPENSHELL_INSTALL_GATEWAY_TIMEOUT:-30}" _elapsed=0 _last_output="" + _service_failed=0 _probe_url="$(local_gateway_endpoint)/" _mtls_dir="${TARGET_HOME}/.config/openshell/gateways/openshell/mtls" @@ -1030,12 +1045,19 @@ wait_for_local_gateway_listener() { info "local gateway listener is reachable" return 0 fi + if [ -n "$_failed_check" ] && "$_failed_check"; then + _service_failed=1 + break + fi sleep 1 _elapsed=$((_elapsed + 1)) done [ -z "$_last_output" ] || printf '%s\n' "$_last_output" >&2 dump_local_gateway_diagnostics + if [ "$_service_failed" -eq 1 ]; then + error "the openshell-gateway service failed to start; fix the cause shown above, then run: systemctl --user restart openshell-gateway" + fi error "local gateway listener did not become reachable at ${_probe_url} within ${_timeout}s" } diff --git a/tasks/scripts/test-install-sh.sh b/tasks/scripts/test-install-sh.sh index 581251891c..c0a42b6700 100755 --- a/tasks/scripts/test-install-sh.sh +++ b/tasks/scripts/test-install-sh.sh @@ -539,6 +539,97 @@ if [[ -z $(find "$snap_user_tls" -maxdepth 0 -perm 700) ]]; then exit 1 fi +assert_user_gateway_service_failed() { + local name=$1 + local unit_state=$2 + local expected=$3 + local actual=0 + + ( + as_target_user() { printf '%s\n' "$unit_state"; } + user_gateway_service_failed + ) >/dev/null || actual=$? + if [ "$actual" != "$expected" ]; then + echo "FAIL: ${name}: expected status ${expected}, got ${actual}" >&2 + exit 1 + fi +} + +assert_user_gateway_service_failed "failed unit" $'ActiveState=failed\nSubState=failed' 0 +assert_user_gateway_service_failed "unit restarting after a failure" $'ActiveState=activating\nSubState=auto-restart' 0 +assert_user_gateway_service_failed "unit starting after a failure" $'ActiveState=activating\nSubState=start' 1 +assert_user_gateway_service_failed "running unit" $'ActiveState=active\nSubState=running' 1 +if ( + as_target_user() { return 1; } + user_gateway_service_failed +); then + echo "FAIL: unreachable user systemd must not count as a failed unit" >&2 + exit 1 +fi + +# Runs the listener wait against an unreachable gateway and prints the number +# of one second waits. The wait is expected to fail. +run_listener_wait() { + local unit_state=$1 + local sleeps_file="${tmpdir}/listener-sleeps" + shift + + : >"$sleeps_file" + ( + as_target_user() { + case "$1" in + systemctl) printf '%s\n' "$unit_state" ;; + *) return 7 ;; + esac + } + sleep() { printf '.' >>"$sleeps_file"; } + info() { :; } + dump_local_gateway_diagnostics() { echo "gateway diagnostics" >&2; } + TARGET_HOME="${tmpdir}/listener-home" + PLATFORM=linux + OPENSHELL_INSTALL_GATEWAY_TIMEOUT=5 wait_for_local_gateway_listener "$@" + ) >"$out" 2>"$err" && return 1 + wc -c <"$sleeps_file" | tr -d ' ' +} + +listener_mtls_dir="${tmpdir}/listener-home/.config/openshell/gateways/openshell/mtls" +mkdir -p "$listener_mtls_dir" +: >"${listener_mtls_dir}/ca.crt" +: >"${listener_mtls_dir}/tls.crt" +: >"${listener_mtls_dir}/tls.key" + +restarting_unit=$'ActiveState=activating\nSubState=auto-restart' +running_unit=$'ActiveState=active\nSubState=running' + +if [ "$(run_listener_wait "$restarting_unit" user_gateway_service_failed)" != "0" ]; then + echo "FAIL: a failed gateway service must stop the listener wait immediately" >&2 + exit 1 +fi +if [ "$(tail -n 1 "$err")" != "openshell: error: the openshell-gateway service failed to start; fix the cause shown above, then run: systemctl --user restart openshell-gateway" ]; then + echo "FAIL: a failed gateway service must end with the service error" >&2 + cat "$err" >&2 + exit 1 +fi +if ! grep -Fq "gateway diagnostics" "$err"; then + echo "FAIL: a failed gateway service must dump diagnostics" >&2 + exit 1 +fi + +if [ "$(run_listener_wait "$running_unit" user_gateway_service_failed)" != "5" ]; then + echo "FAIL: a running gateway service must not stop the listener wait" >&2 + exit 1 +fi +if ! grep -Fq "did not become reachable" "$err"; then + echo "FAIL: a running gateway service must end with the listener timeout" >&2 + cat "$err" >&2 + exit 1 +fi + +if [ "$(run_listener_wait "$restarting_unit")" != "5" ]; then + echo "FAIL: the listener wait must ignore the service state without a check" >&2 + exit 1 +fi + if [ "$(PLATFORM=darwin local_gateway_endpoint)" != "https://localhost:17670" ]; then echo "FAIL: macOS local gateway endpoint must use a TLS-compatible loopback hostname" >&2 exit 1 From fcd8fe557a8f32df65a48a25f8a56071a0b60264 Mon Sep 17 00:00:00 2001 From: udsy19 <71968569+udsy19@users.noreply.github.com> Date: Sat, 3 Oct 2026 20:01:52 +0000 Subject: [PATCH 06/13] fix(docker): pull single tag for tagless --from reference (#4124) The Docker driver passed a bare reference as CreateImageOptions.from_image with no tag. The daemon interprets a tagless fromImage as a request for every tag in the repository and pulls them all (issue #4029). Normalize a pull reference by appending ':latest' when it has neither an explicit tag nor a digest, matching 'docker pull' and the Podman driver. Parsing inspects only the final path component so a registry port (e.g. 'registry:5000/team/app') is not mistaken for a tag and a digest-pinned reference ('...@sha256:...') is left untouched. Applied at both pull sites (pull_image and pull_runtime_image). Signed-off-by: Udaya Tejas --- crates/openshell-driver-docker/src/lib.rs | 42 ++++++++++++++++++++- crates/openshell-driver-docker/src/tests.rs | 29 ++++++++++++++ 2 files changed, 69 insertions(+), 2 deletions(-) diff --git a/crates/openshell-driver-docker/src/lib.rs b/crates/openshell-driver-docker/src/lib.rs index 950b2742e8..dad13d74a4 100644 --- a/crates/openshell-driver-docker/src/lib.rs +++ b/crates/openshell-driver-docker/src/lib.rs @@ -65,6 +65,7 @@ use openshell_sandbox_backend::boundary_protocol::{ }; use opentelemetry::trace::TraceContextExt as _; use sha2::{Digest as _, Sha256}; +use std::borrow::Cow; use std::collections::{HashMap, HashSet}; use std::fmt::Write as _; use std::net::{IpAddr, Ipv4Addr, SocketAddr}; @@ -2968,7 +2969,7 @@ impl DockerComputeDriver { ); let mut stream = self.docker.create_image( Some(CreateImageOptions { - from_image: Some(image.to_string()), + from_image: Some(normalize_pull_reference(image).into_owned()), ..Default::default() }), None, @@ -6298,6 +6299,43 @@ fn container_name_for_sandbox(sandbox: &DriverSandbox) -> String { format!("{CONTAINER_NAME_PREFIX}{workspace}--{truncated_name}-{id_suffix}") } +/// Normalize an image reference for a pull request by appending `:latest` +/// when it carries neither an explicit tag nor a digest. +/// +/// The Docker daemon's `create_image` endpoint interprets a `fromImage` with +/// no tag as "every tag in the repository" and pulls them all, so a bare +/// `--from` reference such as `nicolaka/netshoot` must be resolved to a single +/// tag the way `docker pull` (and the Podman driver) do. +/// +/// Parsing mirrors [`openshell_core::driver_utils::supervisor_image_tag`]: only +/// the final path component may carry a tag, so a registry port such as the +/// `:5000` in `registry:5000/team/app` is not mistaken for one, and a +/// digest-pinned reference (`...@sha256:...`) already names an exact image and +/// is left untouched. A separate helper is needed because `supervisor_image_tag` +/// resolves a bare reference to an implied `latest` and so cannot distinguish a +/// reference that still needs a tag appended. +/// +/// Examples: +/// - `"nicolaka/netshoot"` → `"nicolaka/netshoot:latest"` +/// - `"foo:1.2"` → `"foo:1.2"` (already tagged) +/// - `"foo@sha256:abc"` → `"foo@sha256:abc"` (digest-pinned) +/// - `"registry:5000/team/app"` → `"registry:5000/team/app:latest"` +/// - `"registry:5000/team/app:v1"` → `"registry:5000/team/app:v1"` +fn normalize_pull_reference(image: &str) -> Cow<'_, str> { + // A digest-pinned reference already identifies an exact image. + if image.contains('@') { + return Cow::Borrowed(image); + } + // A `:` only denotes a tag in the final path component; earlier ones are + // registry ports (e.g. `registry:5000/team/app`). + let last_component = image.rsplit('/').next().unwrap_or(image); + if last_component.contains(':') { + Cow::Borrowed(image) + } else { + Cow::Owned(format!("{image}:latest")) + } +} + /// Docker container names may not end with `-`, `.`, or `_`. Truncation can /// leave one of those trailing, so strip them before returning. fn trim_container_name_tail(mut value: String) -> String { @@ -6329,7 +6367,7 @@ fn sanitize_docker_name(value: &str) -> String { async fn pull_runtime_image(docker: &Docker, image: &str, role: &str) -> CoreResult<()> { let mut stream = docker.create_image( Some(CreateImageOptions { - from_image: Some(image.to_string()), + from_image: Some(normalize_pull_reference(image).into_owned()), ..Default::default() }), None, diff --git a/crates/openshell-driver-docker/src/tests.rs b/crates/openshell-driver-docker/src/tests.rs index 08576f422a..435c2b7b98 100644 --- a/crates/openshell-driver-docker/src/tests.rs +++ b/crates/openshell-driver-docker/src/tests.rs @@ -3688,3 +3688,32 @@ fn admission_provisioning_failure_distinguishes_denials_from_lookup_failures() { assert_eq!(lookup.reason, "ResourceAdmissionLookupFailed"); assert_eq!(lookup.message, "inspect docker volume failed"); } + +#[test] +fn normalize_pull_reference_appends_latest_only_when_untagged() { + // A bare repository reference must resolve to a single tag so the daemon + // does not pull every tag in the repository (issue #4029). + assert_eq!( + normalize_pull_reference("nicolaka/netshoot"), + "nicolaka/netshoot:latest" + ); + // An explicit tag is preserved untouched. + assert_eq!(normalize_pull_reference("foo:1.2"), "foo:1.2"); + // A digest-pinned reference already names an exact image. + assert_eq!( + normalize_pull_reference( + "foo@sha256:0000000000000000000000000000000000000000000000000000000000000000" + ), + "foo@sha256:0000000000000000000000000000000000000000000000000000000000000000" + ); + // A registry port is not a tag, so `:latest` is still appended. + assert_eq!( + normalize_pull_reference("registry:5000/team/app"), + "registry:5000/team/app:latest" + ); + // A registry port combined with an explicit tag is left unchanged. + assert_eq!( + normalize_pull_reference("registry:5000/team/app:v1"), + "registry:5000/team/app:v1" + ); +} From d676e036a47e5bcffde8230a2a79ea43d14e47ff Mon Sep 17 00:00:00 2001 From: Shiju Date: Sat, 3 Oct 2026 20:03:39 +0000 Subject: [PATCH 07/13] fix(vm): reject conflicting workload identity selectors (#4036) * fix(vm): enforce the configured workload identity Reject conflicting policy users and groups before VM image preparation and before guest attach or process startup changes state. Validate supervisor policy updates against the protected VM workload identity. Preserve the gateway CA transport and capability-free sandbox launcher. Signed-off-by: Shiju * test(sandbox): clarify VM identity rejection fixtures Name invalid user and group fixtures distinctly and move the final workload identity into its group mismatch test. Signed-off-by: Shiju * fix(supervisor): align VM identity startup with current APIs Pass the optional rejection-log key for VM identity failures and keep generic startup-write regressions free of VM identity constraints. Repair the call sites after the branch rebase so the identity and cleanup proposals compile against the current startup helpers. Signed-off-by: Shiju * fix(vm): restore inactive sandbox workload identity Recover the persisted overlay owner before publishing stopped and terminal sandboxes. Keep resources manageable when identity metadata is invalid. Clarify fixed MicroVM ownership in policy-generation guidance. Signed-off-by: Shiju * test(vm): flush identity fixture before restart Persist the canonical identity file before readiness and report the observed exec, canonical and file-owner identities before comparing them. Signed-off-by: Shiju --------- Signed-off-by: Shiju --- crates/openshell-driver-vm/src/driver.rs | 426 ++++++++++++++++-- .../openshell-sandbox/src/boundary_server.rs | 232 ++++++++++ crates/openshell-server/src/compute/mod.rs | 59 ++- .../openshell-server/src/grpc/validation.rs | 25 + crates/openshell-supervisor/src/lib.rs | 227 ++++++++++ docs/how-it-works/policies/schema.mdx | 8 +- docs/how-it-works/sandboxes/runtimes.mdx | 4 +- e2e/rust/tests/vm_overlay.rs | 159 ++++++- skills/generate-sandbox-policy/SKILL.md | 2 + 9 files changed, 1097 insertions(+), 45 deletions(-) diff --git a/crates/openshell-driver-vm/src/driver.rs b/crates/openshell-driver-vm/src/driver.rs index 30eec2cd5e..f361a3fdb0 100644 --- a/crates/openshell-driver-vm/src/driver.rs +++ b/crates/openshell-driver-vm/src/driver.rs @@ -52,11 +52,12 @@ use openshell_core::proto::compute::v1::{ DriverSandboxTemplate as SandboxTemplate, EnsureWorkspaceRequest, EnsureWorkspaceResponse, GetCapabilitiesRequest, GetCapabilitiesResponse, GetSandboxRequest, GetSandboxResponse, GpuResourceCapabilities, ListSandboxesRequest, ListSandboxesResponse, - MemoryResourceCapabilities, ResourceCapabilities, StartSandboxRequest, StartSandboxResponse, - StopSandboxRequest, StopSandboxResponse, ValidateSandboxCreateRequest, - ValidateSandboxCreateResponse, WatchSandboxesDeletedEvent, WatchSandboxesEvent, - WatchSandboxesPlatformEvent, WatchSandboxesRequest, WatchSandboxesSandboxEvent, - compute_driver_server::ComputeDriver, watch_sandboxes_event, + MemoryResourceCapabilities, ResolvedWorkloadIdentity, ResourceCapabilities, + StartSandboxRequest, StartSandboxResponse, StopSandboxRequest, StopSandboxResponse, + ValidateSandboxCreateRequest, ValidateSandboxCreateResponse, WatchSandboxesDeletedEvent, + WatchSandboxesEvent, WatchSandboxesPlatformEvent, WatchSandboxesRequest, + WatchSandboxesSandboxEvent, WorkloadIdentityRequest, compute_driver_server::ComputeDriver, + watch_sandboxes_event, }; use openshell_core::proto_struct::{ deserialize_optional_non_empty_string_list, struct_to_json_value, @@ -1391,6 +1392,10 @@ impl VmDriver { &overlay_disk, &owner_source_disk, overlay_preparation, + sandbox + .spec + .as_ref() + .and_then(|spec| spec.workload_identity.as_ref()), ) .await .map_err(|err| Status::internal(format!("prepare guest overlay disk failed: {err}")))?; @@ -1723,6 +1728,18 @@ impl VmDriver { Some(record) if !record.deleting => { record.process = Some(process.clone()); record.gpu_bdf.clone_from(&gpu_bdf); + let identity = &runtime_descriptor.workload_identity; + record + .snapshot + .status + .get_or_insert_with(SandboxStatus::default) + .resolved_identity = Some(ResolvedWorkloadIdentity { + uid: identity.uid, + gid: identity.gid, + supplementary_gids: identity.supplementary_gids.clone(), + source: identity.source.clone(), + resource_digest: identity.resource_digest.clone(), + }); snapshot_to_publish = Some(record.snapshot.clone()); } _ => { @@ -2153,38 +2170,41 @@ impl VmDriver { continue; } - if tokio::fs::metadata(state_dir.join(SANDBOX_STOPPED_FILE)) + let inactive_condition = if tokio::fs::metadata(state_dir.join(SANDBOX_STOPPED_FILE)) .await .is_ok() { - let snapshot = sandbox_snapshot(&sandbox, stopped_condition(), false); - let mut registry = self.registry.lock().await; - registry.entry(sandbox.id.clone()).or_insert(SandboxRecord { - snapshot: snapshot.clone(), - state_dir: state_dir.clone(), - process: None, - provisioning_task: None, - gpu_bdf: None, - deleting: false, - }); - drop(registry); - self.publish_snapshot(snapshot); - info!(sandbox_id = %sandbox.id, "vm driver: restored stopped sandbox without launching compute"); - continue; - } - - if tokio::fs::try_exists(state_dir.join(MAIN_PROCESS_EXITED_FILE)) + Some(stopped_condition()) + } else if tokio::fs::try_exists(state_dir.join(MAIN_PROCESS_EXITED_FILE)) .await .unwrap_or(false) { - let snapshot = sandbox_snapshot( - &sandbox, - error_condition( - "ProcessExited", - "Canonical main process exited before VM driver restart", - ), - false, - ); + Some(error_condition( + "ProcessExited", + "Canonical main process exited before VM driver restart", + )) + } else { + None + }; + if let Some(condition) = inactive_condition { + // These sandboxes do not run the launch path that publishes + // identity. Recover it from the same validated owner marker + // before either GetSandbox or WatchSandboxes can observe them. + let identity = match read_persisted_workload_identity(&state_dir).await { + Ok(identity) => identity, + Err(error) => { + warn!(sandbox_id = %sandbox.id, %error, "vm driver: ignoring invalid persisted workload identity"); + // Keep inactive resources manageable even when their + // identity cannot be trusted. Explicit deletion still + // needs this registry entry to remove persisted state. + None + } + }; + let mut snapshot = sandbox_snapshot(&sandbox, condition, false); + snapshot + .status + .get_or_insert_with(SandboxStatus::default) + .resolved_identity = identity; let mut registry = self.registry.lock().await; registry.entry(sandbox.id.clone()).or_insert(SandboxRecord { snapshot: snapshot.clone(), @@ -2198,7 +2218,7 @@ impl VmDriver { self.publish_snapshot(snapshot); info!( sandbox_id = %sandbox.id, - "vm driver: preserved terminal sandbox without restarting canonical process" + "vm driver: restored inactive sandbox without launching compute" ); continue; } @@ -2765,6 +2785,7 @@ impl VmDriver { overlay_disk: &Path, owner_source_disk: &Path, preparation: OverlayPreparation, + requested_identity: Option<&WorkloadIdentityRequest>, ) -> Result { let span_status = openshell_otel::ErrorStatusGuard::current(); let overlay_disk = overlay_disk.to_path_buf(); @@ -2786,6 +2807,9 @@ impl VmDriver { preparation, ) .await?; + // Validate before creating or recovering an overlay. A conflicting + // request must never change the persisted owner or its files. + validate_vm_workload_identity(requested_identity, owner_state)?; let owner_state_written_before_prepare = write_owner_state && preparation == OverlayPreparation::Fresh; if owner_state_written_before_prepare { @@ -5769,6 +5793,35 @@ async fn persisted_sandbox_owner_identity( Ok(None) } +/// Reconstruct status from the owner chosen during preparation, never from a +/// new driver default or a status embedded in the persisted create request. +/// Missing or v1 markers do not contain an identity to report. Invalid owner +/// data must not be published as a resolved identity. +async fn read_persisted_workload_identity( + state_dir: &Path, +) -> Result, String> { + let contents = match tokio::fs::read_to_string(state_dir.join(SANDBOX_OWNER_STATE_FILE)).await { + Ok(contents) if contents.trim() == SANDBOX_OWNER_STATE_V1 => return Ok(None), + Ok(contents) => contents, + Err(error) if error.kind() == std::io::ErrorKind::NotFound => return Ok(None), + Err(error) => return Err(format!("read sandbox owner state: {error}")), + }; + let owner = parse_sandbox_owner_state(&contents) + .map_err(|error| format!("invalid sandbox owner state: {error}"))?; + let resource_digest = match read_persisted_image_identity(state_dir).await { + Ok(identity) => identity, + Err(error) if error.kind() == std::io::ErrorKind::NotFound => String::new(), + Err(error) => return Err(format!("read persisted VM image identity: {error}")), + }; + Ok(Some(ResolvedWorkloadIdentity { + uid: owner.uid, + gid: owner.gid, + supplementary_gids: Vec::new(), + source: "vm-config".into(), + resource_digest, + })) +} + async fn sandbox_owner_identity_from_image( image_path: &Path, ) -> Result { @@ -6788,10 +6841,35 @@ fn status_with_condition( sandbox_fd: String::new(), conditions: vec![condition], deleting, - ..Default::default() + ..snapshot.status.clone().unwrap_or_default() } } +/// VM init reconciles the named `sandbox` account to this overlay's owner. +/// Numeric selectors assert that same immutable identity; they cannot select +/// a new owner. Each empty selector independently accepts the driver default. +fn validate_vm_workload_identity( + request: Option<&WorkloadIdentityRequest>, + owner: SandboxOwnerIdentity, +) -> Result<(), String> { + let Some(request) = request else { + return Ok(()); + }; + for (field, selector, expected) in [ + ("run_as_user", request.user.as_str(), owner.uid), + ("run_as_group", request.group.as_str(), owner.gid), + ] { + if selector.is_empty() || selector == "sandbox" || selector.parse::() == Ok(expected) { + continue; + } + return Err(format!( + "VM {field} '{selector}' conflicts with the resolved workload identity {}:{}; omit the selector or request the driver-owned identity", + owner.uid, owner.gid + )); + } + Ok(()) +} + fn provisioning_condition() -> SandboxCondition { SandboxCondition { r#type: "Ready".to_string(), @@ -7458,6 +7536,172 @@ mod tests { } } + async fn assert_inactive_restore_retains_identity(marker: &str, reason: &str) { + let temp = tempfile::tempdir().unwrap(); + let mut driver = test_driver_with_extensions(LifecycleExtensionRegistry::new()); + driver.config.state_dir = temp.path().to_path_buf(); + // Current driver defaults and serialized status are not evidence of + // the identity that owns an already prepared overlay. + driver.config.sandbox_uid = Some(9000); + driver.config.sandbox_gid = Some(9001); + let sandbox = Sandbox { + id: "sb-inactive-identity".into(), + name: "inactive-identity".into(), + status: Some(SandboxStatus { + resolved_identity: Some(ResolvedWorkloadIdentity { + uid: 8000, + gid: 8001, + ..Default::default() + }), + ..Default::default() + }), + ..Default::default() + }; + let state_dir = sandboxes_root_dir(temp.path()).join(&sandbox.id); + tokio::fs::create_dir_all(&state_dir).await.unwrap(); + write_sandbox_request(&state_dir, &sandbox).await.unwrap(); + write_sandbox_owner_state( + &state_dir, + SandboxOwnerIdentity { + uid: 4242, + gid: 4343, + }, + ) + .await + .unwrap(); + write_sandbox_image_metadata(&state_dir, "unused-image", "sha256:original-image") + .await + .unwrap(); + tokio::fs::write(state_dir.join(marker), b"inactive\n") + .await + .unwrap(); + let mut events = driver.events.subscribe(); + + driver.restore_persisted_sandboxes().await; + + let restored = driver + .get_sandbox(&sandbox.id, &sandbox.name) + .await + .unwrap() + .unwrap(); + let status = restored.status.as_ref().unwrap(); + assert_eq!(status.conditions[0].reason, reason); + let identity = status + .resolved_identity + .as_ref() + .expect("restored overlay owner"); + assert_eq!((identity.uid, identity.gid), (4242, 4343)); + assert!(identity.supplementary_gids.is_empty()); + assert_eq!(identity.source, "vm-config"); + assert_eq!(identity.resource_digest, "sha256:original-image"); + let registry = driver.registry.lock().await; + let record = registry.get(&sandbox.id).unwrap(); + assert!(record.process.is_none()); + assert!(record.provisioning_task.is_none()); + drop(registry); + let event = events.try_recv().expect("restored snapshot event"); + let Some(watch_sandboxes_event::Payload::Sandbox(event)) = event.payload else { + panic!("expected sandbox snapshot event"); + }; + assert_eq!(event.sandbox.as_ref(), Some(&restored)); + assert_eq!( + tokio::fs::read_to_string(state_dir.join(SANDBOX_OWNER_STATE_FILE)) + .await + .unwrap(), + "sandbox-owner-v2:4242:4343\n" + ); + assert!(state_dir.join(marker).exists()); + } + + #[tokio::test] + async fn stopped_restore_retains_persisted_workload_identity() { + assert_inactive_restore_retains_identity(SANDBOX_STOPPED_FILE, "ComputeStopped").await; + } + + #[tokio::test] + async fn terminal_restore_retains_persisted_workload_identity() { + assert_inactive_restore_retains_identity(MAIN_PROCESS_EXITED_FILE, "ProcessExited").await; + } + + #[tokio::test] + async fn inactive_restore_keeps_invalid_metadata_manageable() { + for (marker, reason) in [ + (SANDBOX_STOPPED_FILE, "ComputeStopped"), + (MAIN_PROCESS_EXITED_FILE, "ProcessExited"), + ] { + for invalid_owner in [true, false] { + let temp = tempfile::tempdir().unwrap(); + let mut driver = test_driver_with_extensions(LifecycleExtensionRegistry::new()); + driver.config.state_dir = temp.path().to_path_buf(); + let sandbox = Sandbox { + id: "sb-invalid-owner".into(), + name: "invalid-owner".into(), + ..Default::default() + }; + let state_dir = sandboxes_root_dir(temp.path()).join(&sandbox.id); + tokio::fs::create_dir_all(&state_dir).await.unwrap(); + write_sandbox_request(&state_dir, &sandbox).await.unwrap(); + tokio::fs::write(state_dir.join(marker), b"inactive\n") + .await + .unwrap(); + let owner = if invalid_owner { + "sandbox-owner-v2:0:4343\n" + } else { + "sandbox-owner-v2:4242:4343\n" + }; + tokio::fs::write(state_dir.join(SANDBOX_OWNER_STATE_FILE), owner) + .await + .unwrap(); + if !invalid_owner { + // A directory deterministically fails read_to_string on every test host. + tokio::fs::create_dir(state_dir.join(IMAGE_IDENTITY_FILE)) + .await + .unwrap(); + } + let mut events = driver.events.subscribe(); + + driver.restore_persisted_sandboxes().await; + + let restored = driver + .get_sandbox(&sandbox.id, &sandbox.name) + .await + .unwrap() + .expect("inactive sandbox remains manageable"); + let status = restored.status.as_ref().unwrap(); + assert_eq!(status.conditions[0].reason, reason); + assert!(status.resolved_identity.is_none()); + let registry = driver.registry.lock().await; + let record = registry.get(&sandbox.id).unwrap(); + assert!(record.process.is_none()); + assert!(record.provisioning_task.is_none()); + drop(registry); + let event = events.try_recv().expect("inactive snapshot event"); + let Some(watch_sandboxes_event::Payload::Sandbox(event)) = event.payload else { + panic!("expected sandbox snapshot"); + }; + assert_eq!(event.sandbox.as_ref(), Some(&restored)); + assert!(state_dir.join(marker).exists()); + assert_eq!( + tokio::fs::read_to_string(state_dir.join(SANDBOX_OWNER_STATE_FILE)) + .await + .unwrap(), + owner + ); + assert!( + driver + .delete_sandbox(&sandbox.id, &sandbox.name) + .await + .unwrap() + .deleted + ); + assert!( + !state_dir.exists(), + "explicit delete removes retained state" + ); + } + } + } + #[tokio::test] async fn startup_does_not_restore_terminal_canonical_process() { let temp = tempfile::tempdir().unwrap(); @@ -7492,6 +7736,10 @@ mod tests { assert!(record.process.is_none()); assert!(record.provisioning_task.is_none()); let status = record.snapshot.status.as_ref().expect("terminal status"); + assert!( + status.resolved_identity.is_none(), + "an absent owner marker must not invent an identity" + ); assert!(status.conditions.iter().any(|condition| { condition.reason == "ProcessExited" && condition.status == "False" })); @@ -7591,6 +7839,7 @@ mod tests { Path::new("/unused"), Path::new("/unused"), OverlayPreparation::Fresh, + None, ) .instrument(parent) .await; @@ -8177,6 +8426,117 @@ mod tests { } } + #[tokio::test] + async fn conflicting_vm_identity_preserves_overlay_and_owner() { + let directory = tempfile::tempdir().unwrap(); + let owner = SandboxOwnerIdentity { + uid: 1000, + gid: 1001, + }; + write_sandbox_owner_state(directory.path(), owner) + .await + .unwrap(); + let overlay = directory.path().join(SANDBOX_OVERLAY_IMAGE); + std::fs::write(&overlay, b"existing overlay must not be touched").unwrap(); + let mut driver = test_driver_with_extensions(LifecycleExtensionRegistry::new()); + driver.config.sandbox_uid = Some(10000); + driver.config.sandbox_gid = Some(10001); + let request = WorkloadIdentityRequest { + user: "10000".into(), + group: "10001".into(), + }; + let error = driver + .prepare_runtime_overlay( + directory.path(), + &overlay, + Path::new("/must-not-read-image"), + OverlayPreparation::PreserveExisting, + Some(&request), + ) + .await + .unwrap_err(); + assert!(error.contains("run_as_user '10000'"), "{error}"); + assert!(error.contains("1000:1001"), "{error}"); + assert_eq!( + std::fs::read(&overlay).unwrap(), + b"existing overlay must not be touched" + ); + assert_eq!( + std::fs::read_to_string(directory.path().join(SANDBOX_OWNER_STATE_FILE)).unwrap(), + owner.marker_contents() + ); + } + + #[test] + fn vm_workload_identity_checks_independent_numeric_and_symbolic_selectors() { + let owner = SandboxOwnerIdentity { + uid: 1000, + gid: 1001, + }; + validate_vm_workload_identity(None, owner).unwrap(); + for (user, group) in [ + ("", ""), + ("1000", ""), + ("", "1001"), + ("sandbox", "1001"), + ("1000", "sandbox"), + ("sandbox", "sandbox"), + ] { + validate_vm_workload_identity( + Some(&WorkloadIdentityRequest { + user: user.into(), + group: group.into(), + }), + owner, + ) + .unwrap(); + } + for (user, group, field) in [ + ("10000", "", "run_as_user"), + ("", "10001", "run_as_group"), + ("sandbox", "10001", "run_as_group"), + ("nobody", "", "run_as_user"), + ] { + let error = validate_vm_workload_identity( + Some(&WorkloadIdentityRequest { + user: user.into(), + group: group.into(), + }), + owner, + ) + .unwrap_err(); + assert!(error.contains(field), "{error}"); + assert!(error.contains("1000:1001"), "{error}"); + } + } + + #[test] + fn vm_lifecycle_status_retains_resolved_workload_identity() { + let identity = ResolvedWorkloadIdentity { + uid: 1000, + gid: 1001, + source: "vm-config".into(), + resource_digest: "sha256:image".into(), + ..Default::default() + }; + let snapshot = Sandbox { + status: Some(SandboxStatus { + resolved_identity: Some(identity.clone()), + ..Default::default() + }), + ..Default::default() + }; + for condition in [ + provisioning_condition(), + stopped_condition(), + deleting_condition(), + error_condition("ProcessExited", "exited"), + ] { + let status = status_with_condition(&snapshot, condition, false); + assert_eq!(status.resolved_identity.as_ref(), Some(&identity)); + } + } + #[tokio::test] async fn unmarked_overlay_uses_current_image_instead_of_blind_legacy_identity() { let dir = unique_temp_dir(); diff --git a/crates/openshell-sandbox/src/boundary_server.rs b/crates/openshell-sandbox/src/boundary_server.rs index 4de7c63405..80ca6f32cd 100644 --- a/crates/openshell-sandbox/src/boundary_server.rs +++ b/crates/openshell-sandbox/src/boundary_server.rs @@ -358,6 +358,34 @@ mod linux { Ok(()) } + /// VM selectors assert the protected overlay owner; they cannot choose a + /// replacement identity. Guest init maps `sandbox` to that owner before + /// launching this boundary, so no account lookup or privilege change belongs here. + fn validate_vm_policy_identity( + config: &BoundaryConfig, + policy: &SandboxPolicyWire, + ) -> Result<(), String> { + if !config.resource_claims.contains_key("vm.generation") { + return Ok(()); + } + let identity = &config.workload_identity; + for (field, selector, expected) in [ + ("run_as_user", policy.run_as_user.as_deref(), identity.uid), + ("run_as_group", policy.run_as_group.as_deref(), identity.gid), + ] { + let Some(selector) = selector.filter(|value| !value.is_empty()) else { + continue; + }; + if selector != "sandbox" && selector.parse::() != Ok(expected) { + return Err(format!( + "VM {field} '{selector}' conflicts with the resolved workload identity {}:{}; omit the selector or request the driver-owned identity", + identity.uid, identity.gid + )); + } + } + Ok(()) + } + fn tls_paths_are_absolute( tls: &openshell_sandbox_backend::boundary_protocol::SandboxTlsServerConfig, ) -> bool { @@ -2287,6 +2315,11 @@ mod linux { } fn attach(&self, policy: SandboxPolicyWire) -> Response { + // Reject a conflicting request before establishing the boundary + // or retaining the caller's policy for later replay. + if let Err(error) = validate_vm_policy_identity(&self.config, &policy) { + return guest_error(BoundaryErrorKind::Denied, error); + } let mut state = lock(&self.state); let accepted = match &*state { RuntimeState::AwaitingAttach => { @@ -2479,6 +2512,11 @@ mod linux { provider_env: std::collections::HashMap, provider_files: std::collections::HashMap, ) -> Response { + // Check the supplied launch policy before installing materials or + // replacing its selectors with the measured driver's numeric pair. + if let Err(error) = validate_vm_policy_identity(&self.config, &policy) { + return guest_error(BoundaryErrorKind::Denied, error); + } let spec = match resolve_agent_spec(spec) { Ok(spec) => spec, Err(error) => return guest_error(BoundaryErrorKind::Process, error), @@ -4710,6 +4748,200 @@ mod linux { validate_config(&config).unwrap(); validate_running_identity(&config.workload_identity, false).unwrap(); + let mut wrong_user = config.workload_identity.clone(); + wrong_user.uid = if wrong_user.uid == 10000 { + 10001 + } else { + 10000 + }; + assert!(validate_running_identity(&wrong_user, false).is_err()); + let mut wrong_group = config.workload_identity; + wrong_group.gid = if wrong_group.gid == 10000 { + 10001 + } else { + 10000 + }; + assert!(validate_running_identity(&wrong_group, false).is_err()); + } + + fn vm_identity_test_runtime() -> (tokio::runtime::Runtime, Arc) { + let process_runtime = tokio::runtime::Builder::new_multi_thread() + .worker_threads(2) + .enable_all() + .build() + .expect("test process runtime"); + let mut boundary = { + let _entered = process_runtime.enter(); + availability_test_runtime().0 + }; + let config = &mut Arc::get_mut(&mut boundary) + .expect("test boundary has one owner") + .config; + config + .resource_claims + .insert("vm.generation".to_string(), config.generation.clone()); + (process_runtime, boundary) + } + + fn vm_identity_test_policy(user: Option<&str>, group: Option<&str>) -> SandboxPolicyWire { + SandboxPolicyWire::from(openshell_core::policy::SandboxPolicy { + version: 1, + filesystem: openshell_core::policy::FilesystemPolicy::default(), + network: openshell_core::policy::NetworkPolicy::default(), + landlock: openshell_core::policy::LandlockPolicy::default(), + process: openshell_core::policy::ProcessPolicy { + run_as_user: user.map(str::to_string), + run_as_group: group.map(str::to_string), + }, + }) + } + + #[test] + fn vm_policy_identity_checks_independent_selectors() { + let (_runtime, mut boundary) = vm_identity_test_runtime(); + let uid = boundary.config.workload_identity.uid.to_string(); + let gid = boundary.config.workload_identity.gid.to_string(); + for (user, group) in [ + (None, None), + (Some(""), Some("")), + (Some(uid.as_str()), None), + (None, Some(gid.as_str())), + (Some(uid.as_str()), Some(gid.as_str())), + (Some("sandbox"), Some(gid.as_str())), + (Some(uid.as_str()), Some("sandbox")), + (Some("sandbox"), Some("sandbox")), + ] { + validate_vm_policy_identity( + &boundary.config, + &vm_identity_test_policy(user, group), + ) + .expect("matching or omitted VM selectors"); + } + let wrong_user = if uid == "10000" { "10001" } else { "10000" }; + let wrong_group = if gid == "10000" { "10001" } else { "10000" }; + for (user, group, field) in [ + (Some(wrong_user), None, "run_as_user"), + (None, Some(wrong_group), "run_as_group"), + (Some(uid.as_str()), Some(wrong_group), "run_as_group"), + (Some(wrong_user), Some(gid.as_str()), "run_as_user"), + (Some(wrong_user), Some(wrong_group), "run_as_user"), + ] { + let error = validate_vm_policy_identity( + &boundary.config, + &vm_identity_test_policy(user, group), + ) + .expect_err("either mismatched selector must fail"); + assert!(error.contains(field), "{error}"); + assert!(error.contains(&format!("{uid}:{gid}")), "{error}"); + } + for malformed in [ + "root", + "-1", + "4294967296", + "1000:1000", + " sandbox", + "sandbox\n", + ] { + for (user, group) in [(Some(malformed), None), (None, Some(malformed))] { + assert!( + validate_vm_policy_identity( + &boundary.config, + &vm_identity_test_policy(user, group), + ) + .is_err(), + "invalid selector {malformed:?} must fail" + ); + } + } + Arc::get_mut(&mut boundary) + .expect("test boundary has one owner") + .config + .resource_claims + .remove("vm.generation"); + validate_vm_policy_identity( + &boundary.config, + &vm_identity_test_policy(Some(wrong_user), Some("image-user")), + ) + .expect("non-VM identity behavior is unchanged"); + } + + #[test] + fn vm_attach_rejects_conflicting_identity_without_binding() { + let (_runtime, boundary) = vm_identity_test_runtime(); + let identity = &boundary.config.workload_identity; + let uid = identity.uid.to_string(); + let gid = identity.gid.to_string(); + let wrong_user = if uid == "10000" { "10001" } else { "10000" }; + let wrong_group = if gid == "10000" { "10001" } else { "10000" }; + for (user, group, field, requested) in [ + (Some(wrong_user), None, "run_as_user", wrong_user), + (None, Some(wrong_group), "run_as_group", wrong_group), + ] { + let response = boundary.attach(vm_identity_test_policy(user, group)); + let Response::Error { kind, message } = response else { + panic!("conflicting identity was attached: {response:?}"); + }; + assert_eq!(kind, BoundaryErrorKind::Denied); + assert!( + message.contains(&format!("{field} '{requested}'")), + "{message}" + ); + assert!(message.contains(&format!("{uid}:{gid}")), "{message}"); + assert!(matches!( + *lock(&boundary.state), + RuntimeState::AwaitingAttach + )); + assert!(lock(&boundary.attached_policy).is_none()); + } + assert!(matches!( + boundary.attach(vm_identity_test_policy(Some("sandbox"), Some("sandbox"))), + Response::Attached { .. } + )); + } + + #[test] + fn vm_start_rejects_conflicting_identity_without_launch() { + let (_runtime, boundary) = vm_identity_test_runtime(); + let identity = &boundary.config.workload_identity; + let uid = identity.uid.to_string(); + let gid = identity.gid.to_string(); + let wrong_user = if uid == "10000" { "10001" } else { "10000" }; + let wrong_group = if gid == "10000" { "10001" } else { "10000" }; + *lock(&boundary.state) = RuntimeState::Ready(PreparedBoundary { + network_broker: boundary.network_broker.clone(), + }); + for (user, group, field, requested) in [ + (Some(wrong_user), None, "run_as_user", wrong_user), + (None, Some(wrong_group), "run_as_group", wrong_group), + ] { + let response = boundary.start_agent( + boundary.config.boundary_id.clone(), + AgentSpecWire { + program: "/bin/true".to_string(), + args: Vec::new(), + workdir: None, + timeout_secs: 5, + interactive: false, + }, + vm_identity_test_policy(user, group), + None, + None, + 0, + std::collections::HashMap::new(), + std::collections::HashMap::new(), + ); + let Response::Error { kind, message } = response else { + panic!("conflicting identity reached process start: {response:?}"); + }; + assert_eq!(kind, BoundaryErrorKind::Denied); + assert!( + message.contains(&format!("{field} '{requested}'")), + "{message}" + ); + assert!(message.contains(&format!("{uid}:{gid}")), "{message}"); + assert!(matches!(*lock(&boundary.state), RuntimeState::Ready(_))); + assert!(lock(&boundary.started_agent).is_none()); + } } #[test] diff --git a/crates/openshell-server/src/compute/mod.rs b/crates/openshell-server/src/compute/mod.rs index 264cd77bbb..088231f6e1 100644 --- a/crates/openshell-server/src/compute/mod.rs +++ b/crates/openshell-server/src/compute/mod.rs @@ -5992,15 +5992,31 @@ fn public_status_from_driver( phase: SandboxPhase, current_policy_version: u32, ) -> SandboxStatus { + let mut conditions = status + .conditions + .iter() + .map(public_condition_from_driver) + .collect::>(); + if let Some(identity) = &status.resolved_identity { + // Keep the requested policy intact. This condition reports the + // immutable runtime identity through existing CLI/API status views. + conditions.retain(|condition| condition.r#type != "WorkloadIdentity"); + conditions.push(SandboxCondition { + r#type: "WorkloadIdentity".to_string(), + status: "True".to_string(), + reason: "DriverResolved".to_string(), + message: format!( + "Resolved workload UID:GID is {}:{}", + identity.uid, identity.gid + ), + transition_time: None, + }); + } SandboxStatus { agent_pod: status.instance_id.clone(), agent_fd: status.agent_fd.clone(), sandbox_fd: status.sandbox_fd.clone(), - conditions: status - .conditions - .iter() - .map(public_condition_from_driver) - .collect(), + conditions, phase: phase as i32, current_policy_version, main_process_instance_id: String::new(), @@ -9764,6 +9780,39 @@ mod tests { } } + #[test] + fn public_status_reports_driver_identity_without_rewriting_policy_or_readiness() { + let mut driver_status = + make_driver_status(make_driver_condition("Starting", "VM is starting")); + let unresolved = public_status_from_driver(&driver_status, SandboxPhase::Provisioning, 0); + assert!( + !unresolved + .conditions + .iter() + .any(|condition| condition.r#type == "WorkloadIdentity") + ); + driver_status.resolved_identity = Some( + openshell_core::proto::compute::v1::ResolvedWorkloadIdentity { + uid: 1000, + gid: 1001, + source: "vm-config".into(), + resource_digest: "sha256:image".into(), + ..Default::default() + }, + ); + let status = public_status_from_driver(&driver_status, SandboxPhase::Provisioning, 0); + assert_eq!(status.phase, SandboxPhase::Provisioning as i32); + assert_eq!(status.current_policy_version, 0); + assert_eq!(status.conditions[0], unresolved.conditions[0]); + let identity = status + .conditions + .iter() + .find(|condition| condition.r#type == "WorkloadIdentity") + .unwrap(); + assert_eq!(identity.reason, "DriverResolved"); + assert_eq!(identity.message, "Resolved workload UID:GID is 1000:1001"); + } + fn ready_driver_sandbox(id: &str, name: &str) -> DriverSandbox { DriverSandbox { id: id.to_string(), diff --git a/crates/openshell-server/src/grpc/validation.rs b/crates/openshell-server/src/grpc/validation.rs index f72c571ba5..abdfccda10 100644 --- a/crates/openshell-server/src/grpc/validation.rs +++ b/crates/openshell-server/src/grpc/validation.rs @@ -2403,6 +2403,31 @@ mod tests { // ---- Exec validation ---- + #[test] + fn validate_static_fields_rejects_independent_process_identity_changes() { + let baseline = ProtoSandboxPolicy { + process: Some(openshell_core::proto::ProcessPolicy { + run_as_user: "sandbox".into(), + run_as_group: "1001".into(), + }), + ..Default::default() + }; + for (user, group) in [ + ("10000", "1001"), + ("sandbox", "10001"), + ("", "1001"), + ("sandbox", ""), + ] { + let mut changed = baseline.clone(); + changed.process = Some(openshell_core::proto::ProcessPolicy { + run_as_user: user.into(), + run_as_group: group.into(), + }); + let error = validate_static_fields_unchanged(&baseline, &changed).unwrap_err(); + assert!(error.message().contains("process policy cannot be changed")); + } + } + #[test] fn reject_control_chars_allows_normal_values() { assert!(reject_control_chars("hello world", "test").is_ok()); diff --git a/crates/openshell-supervisor/src/lib.rs b/crates/openshell-supervisor/src/lib.rs index 99607dbedb..5f2c95bd1f 100644 --- a/crates/openshell-supervisor/src/lib.rs +++ b/crates/openshell-supervisor/src/lib.rs @@ -701,6 +701,14 @@ pub async fn run_sandbox( ImagePolicyDiscovery::Missing }; + let vm_policy_identity = runtime_descriptor + .resource_claims + .contains_key("vm.generation") + .then_some(VmPolicyIdentity { + uid: runtime_descriptor.workload_identity.uid, + gid: runtime_descriptor.workload_identity.gid, + }); + // Load policy and initialize OPA engine let openshell_endpoint_for_proxy = openshell_endpoint.clone(); let sandbox_name_for_agg = sandbox.clone(); @@ -721,6 +729,7 @@ pub async fn run_sandbox( policy_data, &extension_credentials, LocalPolicyIdentity::Required, + vm_policy_identity, Some(image_discovery), &RemoteStartupGateway { endpoint: openshell_endpoint.clone().unwrap_or_default(), @@ -1092,6 +1101,7 @@ pub async fn run_sandbox( sandbox: poll_sandbox, opa_engine: poll_engine, loaded_policy_origin, + vm_identity: vm_policy_identity, entrypoint_pid: poll_pid, interval_secs: poll_interval_secs, ocsf_enabled: poll_ocsf_enabled, @@ -2044,6 +2054,39 @@ enum LocalPolicyIdentity { EndpointOnly, } +/// The VM driver fixes overlay ownership before this supervisor starts. Guest +/// init maps the `sandbox` account to this pair; the host must not resolve +/// guest selectors through its own account database. +#[derive(Clone, Copy)] +struct VmPolicyIdentity { + uid: u32, + gid: u32, +} + +impl VmPolicyIdentity { + fn validate(self, policy: &openshell_core::proto::SandboxPolicy) -> Result<()> { + let Some(process) = policy.process.as_ref() else { + return Ok(()); + }; + for (field, selector, expected) in [ + ("run_as_user", process.run_as_user.as_str(), self.uid), + ("run_as_group", process.run_as_group.as_str(), self.gid), + ] { + if !selector.is_empty() + && selector != "sandbox" + && selector.parse::() != Ok(expected) + { + return Err(miette::miette!( + "VM {field} '{selector}' conflicts with the resolved workload identity {}:{}; omit the selector or request the driver-owned identity", + self.uid, + self.gid + )); + } + } + Ok(()) + } +} + struct CapturedProviderEnvironment { credentials: ProviderCredentialState, expires_at_ms: Option, @@ -2100,6 +2143,7 @@ async fn load_policy( extension_credentials, local_policy_identity, None, + None, &RemoteStartupGateway { endpoint: openshell_endpoint.unwrap_or_default(), }, @@ -2119,6 +2163,7 @@ async fn load_policy_with_gateway( policy_data: Option, extension_credentials: &openshell_extension_core::ExtensionCredentialStore, local_policy_identity: LocalPolicyIdentity, + vm_identity: Option, image_discovery: Option, gateway: &impl StartupGateway, ) -> Result<( @@ -2408,6 +2453,24 @@ async fn load_policy_with_gateway( reconciliation_attempts = 0; continue; } + // Admission must reject incompatible image-discovered or repaired + // selectors before reporting this policy effective. + if let Some(identity) = vm_identity + && let Err(error) = identity.validate(&proto_policy) + { + reject_startup_configuration( + gateway, + &mut rejection_log, + id, + &instance_id, + &snapshot, + &error.to_string(), + None, + ) + .await?; + reconciliation_attempts = 0; + continue; + } let provider = grpc_retry("Startup provider environment", || gateway.provider(id)).await?; if provider.provider_env_revision != snapshot.provider_env_revision { @@ -2862,6 +2925,7 @@ async fn reload_gateway_policy_runtime( entrypoint_pid, middleware, transparent_tcp, + None, || {}, ) .await @@ -2873,8 +2937,16 @@ async fn reload_gateway_configuration_runtime( entrypoint_pid: u32, middleware: MiddlewareReloadContext<'_>, transparent_tcp: TransparentTcpReloadState, + vm_identity: Option, commit_credentials: impl FnOnce(), ) -> std::result::Result { + if let (Some(identity), Some(policy)) = (vm_identity, policy) { + // Global policy changes also reach this path. Validate before any + // policy generation, middleware or provider credentials are committed. + identity + .validate(policy) + .map_err(GatewayRuntimeReloadError::PolicyValidation)?; + } if let Some(policy) = policy && policy_contains_explicit_tcp(policy) { @@ -3540,6 +3612,8 @@ struct PolicyPollLoopContext { /// explicit local-file override from an unbound gateway revision so the /// former is never replaced by policy polling. loaded_policy_origin: LoadedPolicyOrigin, + /// Immutable VM overlay identity, also enforced for global policy updates. + vm_identity: Option, entrypoint_pid: Arc, interval_secs: u64, ocsf_enabled: Arc, @@ -4448,6 +4522,7 @@ async fn run_policy_poll_loop_with_client( connector: &ctx.middleware_connector, }, ctx.transparent_tcp, + ctx.vm_identity, || { if let Some(prepared) = prepared_provider.as_ref() { ctx.provider_credentials.install_prepared(prepared); @@ -5500,6 +5575,150 @@ network_policies: assert!(log.changed(&snapshot, "invalid policy")); } + #[tokio::test(start_paused = true)] + async fn startup_rejects_vm_identity_until_matching_policy_is_available() { + use openshell_core::proto::{ConfigurationAdmissionState, PolicySource}; + let mut policy = proto_policy_fixture(); + enrich_proto_baseline_paths(&mut policy); + policy.process = Some(openshell_core::proto::ProcessPolicy { + run_as_user: "10000".into(), + run_as_group: "1001".into(), + }); + let (reports, mut reported) = tokio::sync::mpsc::unbounded_channel(); + let gateway = TestStartupGateway { + desired: Arc::new(std::sync::Mutex::new(settings_poll_result( + Some(policy), + 1, + PolicySource::Sandbox, + ))), + reports, + reject_next_accept: Arc::new(AtomicBool::new(false)), + snapshot_error: None, + report_error: None, + pending_snapshot: false, + pending_acceptance: false, + }; + let active_gateway = gateway.clone(); + let handle = tokio::spawn(async move { + load_policy_with_gateway( + Some("sandbox-id".into()), + Some("sandbox".into()), + Some("http://unused.invalid".into()), + None, + None, + &openshell_extension_core::ExtensionCredentialStore::new(), + LocalPolicyIdentity::Required, + Some(VmPolicyIdentity { + uid: 1000, + gid: 1001, + }), + Some(ImagePolicyDiscovery::Missing), + &active_gateway, + ) + .await + }); + assert_eq!( + reported.recv().await, + Some(ConfigurationAdmissionState::Pending) + ); + assert_eq!( + reported.recv().await, + Some(ConfigurationAdmissionState::Rejected) + ); + assert!( + !handle.is_finished(), + "mismatched policy must never be returned as effective" + ); + { + let mut desired = gateway.desired.lock().unwrap(); + desired + .policy + .as_mut() + .unwrap() + .process + .as_mut() + .unwrap() + .run_as_user = "sandbox".into(); + desired.config_revision += 1; + } + assert_eq!( + reported.recv().await, + Some(ConfigurationAdmissionState::Accepted) + ); + handle + .await + .unwrap() + .expect("matching repair should permit startup"); + } + + #[test] + fn vm_startup_identity_validates_selectors_independently() { + let identity = VmPolicyIdentity { + uid: 1000, + gid: 1001, + }; + for (user, group, accepted) in [ + ("", "", true), + ("1000", "", true), + ("", "1001", true), + ("sandbox", "sandbox", true), + ("10000", "", false), + ("", "10001", false), + ] { + let policy = openshell_core::proto::SandboxPolicy { + process: Some(openshell_core::proto::ProcessPolicy { + run_as_user: user.into(), + run_as_group: group.into(), + }), + ..Default::default() + }; + assert_eq!( + identity.validate(&policy).is_ok(), + accepted, + "{user}:{group}" + ); + } + } + + #[tokio::test] + async fn vm_identity_conflict_prevents_runtime_policy_and_credential_commit() { + let mut policy = proto_policy_fixture(); + let engine = OpaEngine::from_proto(&policy).unwrap(); + let before = engine.current_generation(); + policy.process = Some(openshell_core::proto::ProcessPolicy { + run_as_user: "sandbox".into(), + run_as_group: "10000".into(), + }); + let committed = AtomicBool::new(false); + let result = reload_gateway_configuration_runtime( + &engine, + Some(&policy), + 0, + MiddlewareReloadContext { + desired_services: &[], + authentication: &MiddlewareAuthentication::default(), + registry_changed: false, + connector: &default_middleware_connector(), + }, + TransparentTcpReloadState::default(), + Some(VmPolicyIdentity { + uid: 1000, + gid: 1001, + }), + || { + committed.store(true, Ordering::SeqCst); + }, + ) + .await; + let Err(GatewayRuntimeReloadError::PolicyValidation(error)) = result else { + panic!("conflicting VM group should fail policy validation"); + }; + assert!(error.to_string().contains("run_as_group '10000'")); + assert!(error.to_string().contains("1000:1001")); + assert_eq!(engine.current_generation(), before); + assert!(!committed.load(Ordering::SeqCst)); + } + #[tokio::test(start_paused = true)] async fn startup_pending_gateway_calls_exhaust_their_budgets() { for pending_snapshot in [true, false] { @@ -5529,6 +5748,7 @@ network_policies: None, &openshell_extension_core::ExtensionCredentialStore::new(), LocalPolicyIdentity::Required, + None, Some(ImagePolicyDiscovery::Missing), &gateway, ), @@ -5605,6 +5825,7 @@ network_policies: None, &openshell_extension_core::ExtensionCredentialStore::new(), LocalPolicyIdentity::Required, + None, Some(ImagePolicyDiscovery::Missing), &gateway, ), @@ -5650,6 +5871,7 @@ network_policies: None, &openshell_extension_core::ExtensionCredentialStore::new(), LocalPolicyIdentity::Required, + None, Some(ImagePolicyDiscovery::Missing), &active_gateway, ) @@ -5983,6 +6205,7 @@ network_policies: None, &openshell_extension_core::ExtensionCredentialStore::new(), LocalPolicyIdentity::Required, + None, Some(discovery), &startup_gateway, ) @@ -6131,6 +6354,7 @@ network_policies: None, &openshell_extension_core::ExtensionCredentialStore::new(), LocalPolicyIdentity::Required, + None, Some(discovery), &gateway, ), @@ -6456,6 +6680,7 @@ network_policies: None, &openshell_extension_core::ExtensionCredentialStore::new(), LocalPolicyIdentity::Required, + None, Some(ImagePolicyDiscovery::Missing), &gateway, ), @@ -6550,6 +6775,7 @@ network_policies: None, &openshell_extension_core::ExtensionCredentialStore::new(), LocalPolicyIdentity::Required, + None, Some(discovery.clone()), &gateway, ), @@ -7539,6 +7765,7 @@ network_policies: sandbox: "sandbox-test-name".to_string(), opa_engine, loaded_policy_origin, + vm_identity: None, entrypoint_pid: Arc::new(AtomicU32::new(0)), interval_secs: 0, ocsf_enabled: Arc::new(AtomicBool::new(false)), diff --git a/docs/how-it-works/policies/schema.mdx b/docs/how-it-works/policies/schema.mdx index a190cd5a91..db85e3a19d 100644 --- a/docs/how-it-works/policies/schema.mdx +++ b/docs/how-it-works/policies/schema.mdx @@ -89,11 +89,9 @@ and skipped. | `run_as_group` | string | Driver default | `sandbox` or a numeric GID for the workload. | A numeric ID must be from `1` through `4294967294`, so OpenShell rejects root. -Each field is independent, so you can set one and let the compute driver choose -the other. Only Docker and Podman apply these fields, and only from the policy -that you pass when you create the sandbox. Without them, Docker and Podman use -the image's `USER`. Kubernetes and VM sandboxes run as the identity configured -for their driver. +Each field is independent, so you can set one and let the compute driver choose the other. Docker and Podman use these fields to select the workload identity from the policy passed at creation; omitted fields use the image's `USER`. Kubernetes uses the identity configured for its driver. + +MicroVM preserves the driver's resolved overlay owner. An explicit numeric selector must match that owner; `sandbox` names the guest account reconciled to that owner. A conflicting selector fails before the workload starts, and omitted fields accept the driver default. `openshell sandbox get ` reports the resolved UID:GID in its `WorkloadIdentity` condition. This condition reports identity selection, not workload readiness. Sandbox policy updates cannot change process selectors; global policy updates must still match the resolved identity. Ordinary exec uses the same identity as the canonical process. ```yaml showLineNumbers={false} process: diff --git a/docs/how-it-works/sandboxes/runtimes.mdx b/docs/how-it-works/sandboxes/runtimes.mdx index c1090d1450..50e980e857 100644 --- a/docs/how-it-works/sandboxes/runtimes.mdx +++ b/docs/how-it-works/sandboxes/runtimes.mdx @@ -292,7 +292,7 @@ openshell sandbox create \ ## Sandbox User Identity -Set `process.run_as_user` and `process.run_as_group` in the sandbox policy to choose the sandbox user. Any non-root UID or GID is allowed. When a field is unset, the driver supplies it: +On Docker and Podman, set `process.run_as_user` and `process.run_as_group` in the creation policy to choose the sandbox user. Selectors accept `sandbox` or numeric IDs from `1` through `4294967294`. Kubernetes and MicroVM select the identity through driver configuration. When a field is unset, the driver supplies it: | Driver | Default identity | |---|---| @@ -300,4 +300,6 @@ Set `process.run_as_user` and `process.run_as_group` in the sandbox policy to ch | Kubernetes | OpenShift SCC namespace annotations, otherwise `1000`. Override with `sandbox_uid` and `sandbox_gid`. | | MicroVM | The image's `sandbox` account, otherwise `1000`. Override with `sandbox_uid` and `sandbox_gid`. | +MicroVM persists the resolved UID:GID with the writable overlay and retains it across restarts, even if driver defaults change. Explicit policy selectors must match that identity. For example, `run_as_user: "10000"` fails when the overlay owner is `1000:1000`; OpenShell reports both values without changing file ownership. The symbolic selector `sandbox` refers to the guest account for that owner, and each omitted selector accepts its driver default. The `WorkloadIdentity` condition in `openshell sandbox get ` shows the resolved pair separately from workload readiness. Canonical and exec processes use this same pair; live policy updates cannot change it. + On Docker, the image's `WORKDIR` becomes the workspace. Images with no `WORKDIR`, `/`, or `/sandbox` use `/sandbox`. Any other `WORKDIR` must exist in the image and be writable by the sandbox user. Podman, Kubernetes, and MicroVM always use `/sandbox`. diff --git a/e2e/rust/tests/vm_overlay.rs b/e2e/rust/tests/vm_overlay.rs index 8d1797ef27..16a509376d 100644 --- a/e2e/rust/tests/vm_overlay.rs +++ b/e2e/rust/tests/vm_overlay.rs @@ -4,10 +4,12 @@ //! VM-driver-specific assertions for the sandbox root filesystem. use std::process::Stdio; +use std::time::Duration; use openshell_e2e::harness::binary::openshell_cmd; +use openshell_e2e::harness::cli::{run_cli, wait_for_sandbox_phase}; use openshell_e2e::harness::output::strip_ansi; -use openshell_e2e::harness::sandbox::SandboxGuard; +use openshell_e2e::harness::sandbox::{SandboxGuard, unique_sandbox_name}; #[tokio::test] async fn vm_overlay() { @@ -55,3 +57,158 @@ async fn vm_overlay() { sandbox.cleanup().await; } + +// VM stop terminates the guest without flushing its page cache. Flush this +// fixture before readiness so restart checks durable file content and ownership. +const IDENTITY_MAIN: &str = "set -eu; if ! test -f /sandbox/canonical-identity; then printf '%s:%s\\n' \"$(id -u)\" \"$(id -g)\" > /sandbox/canonical-identity; sync; fi; echo vm-identity-ready; exec sleep infinity"; + +fn identity_policy(user: &str, group: &str) -> tempfile::NamedTempFile { + let file = tempfile::NamedTempFile::new().expect("temporary identity policy"); + std::fs::write( + file.path(), + format!( + r#"version: 1 +filesystem_policy: + include_workdir: true + read_only: [/usr, /lib, /lib64, /proc, /etc, /dev/urandom] + read_write: [/sandbox, /tmp, /dev/null] +landlock: + compatibility: hard_requirement +process: + run_as_user: "{user}" + run_as_group: "{group}" +network_policies: {{}} +"# + ), + ) + .expect("write identity policy"); + file +} + +async fn assert_workload_identity(sandbox: &SandboxGuard) -> (String, String) { + let output = sandbox.exec(&[ + "sh", "-c", + "set -eu; actual=$(id -u):$(id -g); canonical=$(cat /sandbox/canonical-identity); owner=$(stat -c %u:%g /sandbox/canonical-identity); printf 'exec=%s canonical=%s owner=%s\\n' \"$actual\" \"$canonical\" \"$owner\"; test \"$canonical\" = \"$actual\"; test \"$owner\" = \"$actual\"; printf 'identity=%s\\n' \"$actual\"", + ]).await.expect("canonical and exec identities and file ownership agree"); + let clean = strip_ansi(&output); + let pair = clean + .lines() + .find_map(|line| line.strip_prefix("identity=")) + .expect("observed workload identity"); + let (uid, gid) = pair.split_once(':').expect("UID:GID pair"); + let (status, code) = run_cli(&["sandbox", "get", &sandbox.name, "--output", "json"]).await; + assert_eq!(code, 0, "{status}"); + let status: serde_json::Value = + serde_json::from_str(&strip_ansi(&status)).expect("sandbox status JSON"); + assert!( + status["conditions"] + .as_array() + .expect("conditions") + .iter() + .any(|condition| { + condition["type"] == "WorkloadIdentity" + && condition["message"] == format!("Resolved workload UID:GID is {pair}") + }), + "status must report observed workload identity: {status}" + ); + (uid.to_string(), gid.to_string()) +} + +#[tokio::test] +async fn vm_identity_matches_status_and_rejects_conflicts() { + // Use observed defaults so the test also works with configured driver IDs. + // The canonical process writes a file before signaling readiness; exec + // verifies its content and ownership independently of the status report. + let default_policy = identity_policy("", ""); + let mut sandbox = SandboxGuard::create_keep_with_args( + &[ + "--policy", + default_policy.path().to_str().unwrap(), + "--no-tty", + ], + &["sh", "-c", IDENTITY_MAIN], + "vm-identity-ready", + ) + .await + .expect("omitted selectors use driver identity"); + let (uid, gid) = assert_workload_identity(&sandbox).await; + let original = sandbox + .exec(&["stat", "-c", "%u:%g", "/sandbox/canonical-identity"]) + .await + .unwrap(); + let (output, code) = run_cli(&["sandbox", "stop", &sandbox.name]).await; + assert_eq!(code, 0, "{output}"); + wait_for_sandbox_phase(&sandbox.name, "Stopped", Duration::from_secs(60)) + .await + .unwrap(); + let (output, code) = run_cli(&["sandbox", "start", &sandbox.name]).await; + assert_eq!(code, 0, "{output}"); + wait_for_sandbox_phase(&sandbox.name, "Ready", Duration::from_secs(120)) + .await + .unwrap(); + assert_eq!( + assert_workload_identity(&sandbox).await, + (uid.clone(), gid.clone()) + ); + let restored = sandbox + .exec(&["stat", "-c", "%u:%g", "/sandbox/canonical-identity"]) + .await + .unwrap(); + assert_eq!( + strip_ansi(&restored), + strip_ansi(&original), + "restart must retain the owned overlay file" + ); + sandbox.cleanup().await; + + for (user, group) in [ + (uid.as_str(), ""), + ("", gid.as_str()), + ("sandbox", "sandbox"), + ] { + let policy = identity_policy(user, group); + let mut matching = SandboxGuard::create_keep_with_args( + &["--policy", policy.path().to_str().unwrap(), "--no-tty"], + &["sh", "-c", IDENTITY_MAIN], + "vm-identity-ready", + ) + .await + .expect("matching numeric or symbolic selector"); + assert_eq!( + assert_workload_identity(&matching).await, + (uid.clone(), gid.clone()) + ); + matching.cleanup().await; + } + + let wrong_uid = if uid == "10000" { "10001" } else { "10000" }; + let wrong_gid = if gid == "10000" { "10001" } else { "10000" }; + for (user, group, field) in [ + (wrong_uid, "", "run_as_user"), + ("", wrong_gid, "run_as_group"), + ] { + let policy = identity_policy(user, group); + let name = unique_sandbox_name(); + let mut cleanup = SandboxGuard::manage_existing(name.clone()); + let result = SandboxGuard::create(&[ + "--name", + &name, + "--policy", + policy.path().to_str().unwrap(), + "--no-tty", + ]) + .await; + let error = match result { + Ok(mut launched) => { + launched.cleanup().await; + panic!("conflicting VM identity reached Ready"); + } + Err(error) => error, + }; + assert!( + error.contains(field) && error.contains(&format!("{uid}:{gid}")), + "rejection must name selector and resolved identity: {error}" + ); + cleanup.cleanup().await; + } +} diff --git a/skills/generate-sandbox-policy/SKILL.md b/skills/generate-sandbox-policy/SKILL.md index ac03c8576b..d626aaf1b3 100644 --- a/skills/generate-sandbox-policy/SKILL.md +++ b/skills/generate-sandbox-policy/SKILL.md @@ -514,6 +514,8 @@ When the user explicitly requests `process.run_as_user` or (`4294967295`). Warn that a low numeric identity inherits permissions granted to the same ID on image files, mounted volumes, or devices. +For MicroVM, numeric selectors must match the resolved owner of the sandbox's writable overlay. User and group are checked independently; either mismatch prevents startup. If the owner UID:GID is unknown, omit the selectors or use `sandbox` so the driver retains that identity. Do not choose another numeric identity or suggest that the policy can change an existing overlay's owner. For example, with an owner of `1000:1000`, a request for UID `10000` must be rejected with an explanation and the omission/`sandbox` alternatives, rather than generating a policy that the VM cannot start. + If the user provides a file path, write to it. Otherwise, ask where to place it. A common convention is a project-local policy file (e.g., `sandbox-policy.yaml`) passed to `openshell sandbox create --policy ` or set via the `OPENSHELL_SANDBOX_POLICY` env var. ### Mode C: Present Only (no file write) From 4188eabaf2d41fbf94f7d793f411af21b3a4685b Mon Sep 17 00:00:00 2001 From: Shiju Date: Sat, 3 Oct 2026 20:03:40 +0000 Subject: [PATCH 08/13] fix(vm): stop image workers before cleaning staging files (#4039) Run image preparation in an owned worker process, reserve its process identity until cleanup completes, and protect staging with leases so cancellation and recovery cannot race with another preparation attempt. Signed-off-by: Shiju --- crates/openshell-driver-vm/README.md | 4 + crates/openshell-driver-vm/src/driver.rs | 878 ++++++++++++-- crates/openshell-driver-vm/src/main.rs | 9 + crates/openshell-driver-vm/src/preparation.rs | 1016 +++++++++++++++++ .../tests/image_preparation.rs | 383 +++++++ 5 files changed, 2202 insertions(+), 88 deletions(-) create mode 100644 crates/openshell-driver-vm/src/preparation.rs create mode 100644 crates/openshell-driver-vm/tests/image_preparation.rs diff --git a/crates/openshell-driver-vm/README.md b/crates/openshell-driver-vm/README.md index a0d83e1e4d..334d968dee 100644 --- a/crates/openshell-driver-vm/README.md +++ b/crates/openshell-driver-vm/README.md @@ -219,6 +219,10 @@ The requested sandbox image is never selected as the bootstrap image. Operators must configure either `bootstrap_image` or `default_image`; when both are empty, the driver fails during startup. +Image and writable-overlay preparation run in owned worker processes. Stopping or deleting a sandbox cancels its worker and formatter processes, waits for them to exit, then removes the attempt's temporary files under `/images/preparations/`. New overlays, including retries after an interrupted first start, are built there and renamed into sandbox state only after completion. A file lock inherited by the worker's children keeps cleanup from deleting files that a process still owns. If cleanup cannot establish that the processes have stopped, the operation returns an error and retains both temporary files and sandbox state. + +At startup, the driver reclaims inactive attempts in this directory. It preserves active attempts, committed image caches, and unmarked staging from older releases. Concurrent preparations serialize image-cache publication; an interrupted attempt cannot publish a partial disk over an existing cache entry. Gateway upload slots for `rootfs_tar_path` keep their existing, separate lifecycle. + Each sandbox gets its own sparse writable `/sandboxes//overlay.ext4`. Guest init mounts overlayfs as `/` with the prepared image rootfs as lowerdir when present, otherwise the bootstrap diff --git a/crates/openshell-driver-vm/src/driver.rs b/crates/openshell-driver-vm/src/driver.rs index f361a3fdb0..f737b6e79e 100644 --- a/crates/openshell-driver-vm/src/driver.rs +++ b/crates/openshell-driver-vm/src/driver.rs @@ -3,6 +3,9 @@ #![allow(unsafe_code)] +#[path = "preparation.rs"] +mod preparation; + use crate::gpu::{GpuInventory, allocate_vsock_cid}; use crate::isolation::VmBoundarySpec; @@ -88,7 +91,7 @@ use std::process::Stdio; use std::sync::Arc; use std::sync::atomic::{AtomicU64, Ordering}; use std::time::Duration; -use tokio::io::AsyncWriteExt; +use tokio::io::{AsyncBufReadExt, AsyncWriteExt}; use tokio::process::{Child, Command}; use tokio::sync::{Mutex, broadcast, mpsc}; use tokio::task::JoinHandle; @@ -214,7 +217,7 @@ struct VmDriverTlsPaths { ca: PathBuf, } -#[derive(Debug, Clone)] +#[derive(Debug, Clone, serde::Serialize, serde::Deserialize)] struct RuntimeImagePlan { root_disk: PathBuf, image_disk: Option, @@ -579,6 +582,7 @@ struct SandboxRecord { state_dir: PathBuf, process: Option>>, provisioning_task: Option>, + preparation: Option>>, gpu_bdf: Option, deleting: bool, } @@ -612,13 +616,13 @@ fn resolve_record_id( Ok(first) } -#[derive(Debug, Clone, Copy, PartialEq, Eq)] +#[derive(Debug, Clone, Copy, PartialEq, Eq, serde::Serialize, serde::Deserialize)] enum OverlayPreparation { Fresh, PreserveExisting, } -#[derive(Debug, Clone, Copy, PartialEq, Eq)] +#[derive(Debug, Clone, Copy, PartialEq, Eq, serde::Serialize, serde::Deserialize)] struct SandboxOwnerIdentity { uid: u32, gid: u32, @@ -666,6 +670,7 @@ pub struct VmDriver { launcher_bin: PathBuf, registry: Arc>>, image_cache_lock: Arc>, + preparation_root: Option, events: broadcast::Sender, gpu_inventory: Option>>, lifecycle_extensions: Arc, @@ -754,10 +759,13 @@ impl VmDriver { launcher_bin, registry: Arc::new(Mutex::new(HashMap::new())), image_cache_lock: Arc::new(Mutex::new(())), + preparation_root: None, events, gpu_inventory, lifecycle_extensions: Arc::new(lifecycle_extensions), }; + preparation::reconcile(&image_cache_root_dir(&driver.config.state_dir)) + .map_err(|error| format!("reconcile image preparation staging: {error}"))?; driver.restore_persisted_sandboxes().await; Ok(driver) } @@ -840,17 +848,21 @@ impl VmDriver { if validate_host_supervisor(&destination).is_ok() { return Ok(destination); } - let _cache_guard = self.image_cache_lock.lock().await; + let cache_guard = self.image_cache_lock.clone().lock_owned().await; if validate_host_supervisor(&destination).is_ok() { return Ok(destination); } let destination_for_extract = destination.clone(); - tokio::task::spawn_blocking(move || extract_host_supervisor(&destination_for_extract)) - .await - .map_err(|error| { - Status::internal(format!("host supervisor extraction panicked: {error}")) - })? - .map_err(Status::failed_precondition)?; + tokio::task::spawn_blocking(move || { + // This atomic shared-cache write does not target sandbox state. + // Retain its lock until the write finishes even if provisioning is + // cancelled, so a later caller cannot start a competing extractor. + let _cache_guard = cache_guard; + extract_host_supervisor(&destination_for_extract) + }) + .await + .map_err(|error| Status::internal(format!("host supervisor extraction panicked: {error}")))? + .map_err(Status::failed_precondition)?; validate_host_supervisor(&destination).map_err(Status::failed_precondition)?; Ok(destination) } @@ -1127,6 +1139,7 @@ impl VmDriver { state_dir: state_dir.clone(), process: None, provisioning_task: None, + preparation: None, gpu_bdf: None, deleting: false, }, @@ -1312,8 +1325,9 @@ impl VmDriver { })?; let bootstrap_image_ref = self.bootstrap_image_ref()?; let bootstrap_image_identity = self - .ensure_cached_bootstrap_rootfs_image(&sandbox.id, &bootstrap_image_ref) - .await?; + .prepare_images_in_worker(&sandbox.id, &bootstrap_image_ref, None, true) + .await? + .bootstrap_image_identity; let root_disk = image_cache_rootfs_image(&self.config.state_dir, &bootstrap_image_identity); let image_disk = image_cache_rootfs_image(&self.config.state_dir, &persisted_identity); @@ -1324,8 +1338,13 @@ impl VmDriver { bootstrap_image_identity, } } else { - self.prepare_runtime_images(&sandbox.id, &image_ref, rootfs_tar_path.as_deref()) - .await? + self.prepare_images_in_worker( + &sandbox.id, + &image_ref, + rootfs_tar_path.as_deref(), + false, + ) + .await? }; let image_identity = image_plan.image_identity.clone(); self.ensure_provisioning_active(&sandbox.id).await?; @@ -1387,9 +1406,8 @@ impl VmDriver { ), ); let sandbox_owner_state = self - .prepare_runtime_overlay( - &state_dir, - &overlay_disk, + .prepare_overlay_in_worker( + &sandbox.id, &owner_source_disk, overlay_preparation, sandbox @@ -1825,7 +1843,11 @@ impl VmDriver { if let Some(task) = provisioning_task { task.abort(); + // Wait for the future to drop its worker I/O before taking the + // registered attempt. Aborting alone does not stop its subprocess. + let _ = task.await; } + self.cleanup_image_preparation(&record_id).await?; if let Some(process) = process { let mut process = process.lock().await; process.deleting = true; @@ -1881,6 +1903,11 @@ impl VmDriver { let record = registry .get(&id) .ok_or_else(|| Status::not_found("sandbox not found"))?; + if record.preparation.is_some() && record.provisioning_task.is_none() { + return Err(Status::failed_precondition( + "image preparation cleanup is still in progress", + )); + } ( id, record.state_dir.clone(), @@ -2029,7 +2056,11 @@ impl VmDriver { if let Some(task) = provisioning_task { task.abort(); + // Wait for the future to drop its worker I/O before taking the + // registered attempt. Aborting alone does not stop its subprocess. + let _ = task.await; } + self.cleanup_image_preparation(&record_id).await?; if let Some(process) = process { let mut process = process.lock().await; @@ -2211,6 +2242,7 @@ impl VmDriver { state_dir: state_dir.clone(), process: None, provisioning_task: None, + preparation: None, gpu_bdf: None, deleting: false, }); @@ -2310,6 +2342,7 @@ impl VmDriver { state_dir: state_dir.clone(), process: None, provisioning_task: None, + preparation: None, gpu_bdf: None, deleting: false, }, @@ -2657,7 +2690,7 @@ impl VmDriver { remove_state: bool, ) { self.release_gpu(sandbox_id); - let snapshot = { + let (snapshot, may_remove_state) = { let mut registry = self.registry.lock().await; let Some(record) = registry.get_mut(sandbox_id) else { return; @@ -2672,10 +2705,16 @@ impl VmDriver { error_condition(reason, message), false, )); - Some(record.snapshot.clone()) + // A failed cleanup leaves the attempt registered. Preserve all + // sandbox state until its worker and descendants are confirmed + // stopped, even when provisioning itself has already failed. + ( + Some(record.snapshot.clone()), + remove_state && record.preparation.is_none(), + ) }; - if remove_state { + if may_remove_state { let _ = tokio::fs::remove_dir_all(state_dir).await; remove_sandbox_socket_dir(&self.socket_root_fd, sandbox_id); } @@ -2693,6 +2732,235 @@ impl VmDriver { } } + /// Keep every image write in a process owned by the sandbox record. The + /// registry retains the attempt when this future is cancelled so lifecycle + /// cleanup can kill and reap it before removing any staging or state. + #[tracing::instrument( + name = "vm.prepare_images", + skip(self), + fields(otel.name = "vm.prepare_images", otel.status_code = tracing::field::Empty, sandbox.id = %sandbox_id, image.ref = %image_ref) + )] + async fn prepare_images_in_worker( + &self, + sandbox_id: &str, + image_ref: &str, + rootfs_tar: Option<&Path>, + bootstrap_only: bool, + ) -> Result { + match self + .run_preparation_in_worker(sandbox_id, image_ref, rootfs_tar, bootstrap_only, None) + .await? + { + preparation::Output::Images(plan) => Ok(plan), + preparation::Output::Overlay(_) => { + Err(Status::internal("image worker returned an overlay result")) + } + } + } + + #[tracing::instrument( + name = "vm.prepare_overlay", + skip(self), + fields(otel.name = "vm.prepare_overlay", otel.status_code = tracing::field::Empty, sandbox.id = %sandbox_id) + )] + async fn prepare_overlay_in_worker( + &self, + sandbox_id: &str, + owner_source_disk: &Path, + preparation: OverlayPreparation, + requested_identity: Option<&WorkloadIdentityRequest>, + ) -> Result { + let overlay = preparation::OverlayRequest { + source_disk: owner_source_disk.to_path_buf(), + preparation, + requested_identity: requested_identity + .map(preparation::WorkloadIdentitySelectors::from), + }; + match self + .run_preparation_in_worker(sandbox_id, "", None, false, Some(overlay)) + .await? + { + preparation::Output::Overlay(owner) => Ok(owner), + preparation::Output::Images(_) => { + Err(Status::internal("overlay worker returned an image result")) + } + } + } + + /// The worker resolves the owner from the same state directory it will + /// modify, then rejects selectors that conflict with that owner before + /// preparing or repairing the writable disk. + async fn prepare_requested_overlay( + &self, + sandbox_id: &str, + overlay: preparation::OverlayRequest, + ) -> Result { + if !overlay + .source_disk + .starts_with(image_cache_root_dir(&self.config.state_dir)) + { + return Err(Status::invalid_argument( + "overlay source must be a cached image", + )); + } + let state_dir = sandbox_state_dir(&self.config.state_dir, sandbox_id)?; + let overlay_disk = sandbox_runtime_disk_paths(&state_dir).overlay_disk; + let requested_identity = overlay + .requested_identity + .map(WorkloadIdentityRequest::from); + self.prepare_runtime_overlay( + &state_dir, + &overlay_disk, + &overlay.source_disk, + overlay.preparation, + requested_identity.as_ref(), + ) + .await + .map_err(Status::internal) + } + + async fn run_preparation_in_worker( + &self, + sandbox_id: &str, + image_ref: &str, + rootfs_tar: Option<&Path>, + bootstrap_only: bool, + overlay: Option, + ) -> Result { + let span_status = openshell_otel::ErrorStatusGuard::current(); + let bootstrap_image_ref = self.bootstrap_image_ref()?; + // Keep the driver trace continuous across the worker boundary. The + // first Pulled event completes bootstrap resolution; later events + // belong to the requested workload image. + let mut bootstrap_span = overlay.is_none().then(|| { + tracing::info_span!( + "vm.resolve_bootstrap_image", + otel.name = "vm.resolve_bootstrap_image", + otel.status_code = tracing::field::Empty, + sandbox.id = %sandbox_id, + image.ref = %bootstrap_image_ref, + ) + }); + // Independent attempts can prepare images and private overlays at the + // same time. Workers serialize only shared cache publication. + let (attempt, stdout) = { + let mut registry = self.registry.lock().await; + let record = registry + .get_mut(sandbox_id) + .filter(|record| !record.deleting) + .ok_or_else(|| Status::cancelled("sandbox provisioning cancelled"))?; + if record.preparation.is_some() { + return Err(Status::failed_precondition( + "sandbox image preparation is already active", + )); + } + let mut attempt = + preparation::Attempt::create(&image_cache_root_dir(&self.config.state_dir)) + .map_err(|error| { + Status::internal(format!("create image preparation attempt: {error}")) + })?; + let stdout = attempt.spawn( + &self.launcher_bin, + preparation::Request { + config: self.config.clone(), + sandbox_id: sandbox_id.to_string(), + image_ref: image_ref.to_string(), + rootfs_tar: rootfs_tar.map(Path::to_path_buf), + bootstrap_only, + overlay, + lease_fd: -1, + parent_pid: 0, + }, + ); + let attempt = Arc::new(Mutex::new(attempt)); + record.preparation = Some(attempt.clone()); + (attempt, stdout) + }; + let _cleanup = preparation::CleanupOnDrop(attempt.clone()); + let outcome = async { + let stdout = stdout.map_err(|error| { + Status::internal(format!("start image preparation worker: {error}")) + })?; + let mut lines = tokio::io::BufReader::new(stdout).lines(); + let mut result = None; + while let Some(line) = lines.next_line().await.map_err(|error| { + Status::internal(format!("read image preparation progress: {error}")) + })? { + match serde_json::from_str::(&line).map_err(|error| { + Status::internal(format!("decode image preparation progress: {error}")) + })? { + preparation::Message::Event(bytes) => { + let event = + WatchSandboxesEvent::decode(bytes.as_slice()).map_err(|error| { + Status::internal(format!("decode image preparation event: {error}")) + })?; + if matches!(event.payload.as_ref(), Some(watch_sandboxes_event::Payload::PlatformEvent(value)) + if value.event.as_ref().is_some_and(|event| event.reason == "Pulled")) { + bootstrap_span.take(); + } + let _ = self.events.send(event); + } + preparation::Message::Complete(value) => result = Some(value), + } + } + let status = attempt.lock().await.wait().await.map_err(|error| { + Status::internal(format!("wait for image preparation: {error}")) + })?; + if !status.success() { + return Err(Status::failed_precondition(format!( + "image preparation worker exited with {status}" + ))); + } + result + .ok_or_else(|| { + Status::internal("image preparation worker exited without a result") + })? + .map_err(Status::failed_precondition) + } + .await; + if outcome.is_err() + && let Some(span) = bootstrap_span.as_ref() + { + span.record("otel.status_code", "ERROR"); + } + self.cleanup_image_preparation(sandbox_id).await?; + span_status.finish(outcome) + } + + async fn cleanup_image_preparation(&self, sandbox_id: &str) -> Result<(), Status> { + let attempt = self + .registry + .lock() + .await + .get(sandbox_id) + .and_then(|record| record.preparation.clone()); + if let Some(attempt) = attempt { + attempt.lock().await.cleanup().await?; + if let Some(record) = self.registry.lock().await.get_mut(sandbox_id) + && record + .preparation + .as_ref() + .is_some_and(|current| Arc::ptr_eq(current, &attempt)) + { + record.preparation = None; + } + } + Ok(()) + } + + fn image_staging_dir(&self, image_identity: &str) -> PathBuf { + self.preparation_root.as_ref().map_or_else( + || image_cache_staging_dir(&self.config.state_dir, image_identity), + |root| { + root.join(format!( + "{}.staging-{}", + sanitize_image_identity(image_identity), + unique_image_cache_suffix() + )) + }, + ) + } + #[tracing::instrument( name = "vm.prepare_images", skip(self), @@ -2820,28 +3088,53 @@ impl VmDriver { } let template_path = overlay_template_image(&self.config.state_dir, overlay_size_bytes); + let overlay_metadata = tokio::fs::metadata(&overlay_disk).await; let recover_preserved_overlay = preparation == OverlayPreparation::PreserveExisting - && tokio::fs::metadata(&overlay_disk) - .await - .is_ok_and(|metadata| metadata.is_file()); + && overlay_metadata.as_ref().is_ok_and(fs::Metadata::is_file); + let publish_new_overlay = preparation == OverlayPreparation::Fresh + || matches!(overlay_metadata, Err(error) if error.kind() == std::io::ErrorKind::NotFound); if !overlay_template_image_ready(&template_path, overlay_size_bytes).await? { let _cache_guard = self.image_cache_lock.lock().await; let template_path = template_path.clone(); + let staging_dir = self.image_staging_dir("overlay-template"); + let cache_root = image_cache_root_dir(&self.config.state_dir); tokio::task::spawn_blocking(move || { - ensure_sandbox_overlay_template_image(&template_path, overlay_size_bytes) + ensure_sandbox_overlay_template_image( + &cache_root, + &template_path, + overlay_size_bytes, + &staging_dir, + ) }) .await .map_err(|err| format!("overlay template preparation panicked: {err}"))??; } let overlay_to_recover = overlay_disk.clone(); + let staging_dir = self.image_staging_dir("writable-overlay"); let result = tokio::task::spawn_blocking(move || { - prepare_sandbox_overlay_image( - &template_path, - &overlay_disk, - preparation, - overlay_size_bytes, - ) + if publish_new_overlay { + // An interrupted first start can leave no persisted overlay. + // Stage that retry just like a fresh copy, so cancellation + // cannot publish a partial disk as existing VM state. Other + // metadata errors still go through the preservation checks. + fs::create_dir_all(&staging_dir).map_err(|error| error.to_string())?; + let staging_overlay = staging_dir.join(SANDBOX_OVERLAY_IMAGE); + prepare_sandbox_overlay_image( + &template_path, + &staging_overlay, + preparation, + overlay_size_bytes, + )?; + fs::rename(&staging_overlay, &overlay_disk).map_err(|error| error.to_string()) + } else { + prepare_sandbox_overlay_image( + &template_path, + &overlay_disk, + preparation, + overlay_size_bytes, + ) + } }) .await .map_err(|err| format!("overlay image preparation panicked: {err}"))?; @@ -3252,7 +3545,7 @@ impl VmDriver { }); } - let staging_dir = image_cache_staging_dir(&self.config.state_dir, &cache_identity); + let staging_dir = self.image_staging_dir(&cache_identity); let rootfs_archive = staging_dir.join(IMAGE_EXPORT_ROOTFS_ARCHIVE); self.reset_image_staging_dir(&staging_dir).await?; @@ -3365,7 +3658,7 @@ impl VmDriver { }); } - let staging_dir = image_cache_staging_dir(&self.config.state_dir, &cache_identity); + let staging_dir = self.image_staging_dir(&cache_identity); let rootfs_archive = staging_dir.join(IMAGE_EXPORT_ROOTFS_ARCHIVE); self.reset_image_staging_dir(&staging_dir).await?; @@ -3506,7 +3799,7 @@ impl VmDriver { }); } - let staging_dir = image_cache_staging_dir(&self.config.state_dir, &cache_identity); + let staging_dir = self.image_staging_dir(&cache_identity); self.reset_image_staging_dir(&staging_dir).await?; let layout_dir = staging_dir.join(GUEST_IMAGE_OCI_LAYOUT_DIR); @@ -3704,17 +3997,28 @@ impl VmDriver { return Err(Status::failed_precondition(message)); } - if tokio::fs::metadata(&image_path).await.is_ok() { - let _ = tokio::fs::remove_dir_all(staging_dir).await; - return Ok(()); - } - tokio::fs::rename(&prepared_image, &image_path) - .await - .map_err(|err| Status::internal(format!("store prepared image disk failed: {err}")))?; + self.publish_prepared_image(&prepared_image, &image_path) + .await?; let _ = tokio::fs::remove_dir_all(staging_dir).await; Ok(()) } + async fn publish_prepared_image( + &self, + staged: &Path, + destination: &Path, + ) -> Result<(), Status> { + let cache_root = image_cache_root_dir(&self.config.state_dir); + let staged = staged.to_path_buf(); + let destination = destination.to_path_buf(); + tokio::task::spawn_blocking(move || { + preparation::publish_cache_file(&cache_root, &staged, &destination, None) + }) + .await + .map_err(|error| Status::internal(format!("cache publication panicked: {error}")))? + .map_err(|error| Status::internal(format!("store cached rootfs image failed: {error}"))) + } + #[allow(clippy::similar_names)] async fn run_image_prep_vm( &self, @@ -3726,7 +4030,8 @@ impl VmDriver { let mut command = Command::new(&self.launcher_bin); command.kill_on_drop(true); command.stdin(Stdio::null()); - command.stdout(Stdio::inherit()); + // Worker stdout carries framed progress and completion messages. + command.stdout(Stdio::from(std::io::stderr())); command.stderr(Stdio::inherit()); command.arg("--internal-run-vm"); command.arg("--vm-root-disk").arg(bootstrap_root_disk); @@ -3821,7 +4126,7 @@ impl VmDriver { ) -> Result<(), Status> { let cache_dir = image_cache_dir(&self.config.state_dir, image_identity); let image_path = image_cache_rootfs_image(&self.config.state_dir, image_identity); - let staging_dir = image_cache_staging_dir(&self.config.state_dir, image_identity); + let staging_dir = self.image_staging_dir(image_identity); let exported_rootfs = staging_dir.join(IMAGE_EXPORT_ROOTFS_ARCHIVE); let prepared_rootfs = staging_dir.join("rootfs"); let prepared_image = staging_dir.join(IMAGE_CACHE_ROOTFS_IMAGE); @@ -3928,14 +4233,8 @@ impl VmDriver { return Err(Status::failed_precondition(err)); } - if tokio::fs::metadata(&image_path).await.is_ok() { - let _ = tokio::fs::remove_dir_all(&staging_dir).await; - return Ok(()); - } - - tokio::fs::rename(&prepared_image, &image_path) - .await - .map_err(|err| Status::internal(format!("store cached rootfs image failed: {err}")))?; + self.publish_prepared_image(&prepared_image, &image_path) + .await?; let _ = tokio::fs::remove_dir_all(&staging_dir).await; Ok(()) } @@ -3952,7 +4251,7 @@ impl VmDriver { ) -> Result<(), Status> { let cache_dir = image_cache_dir(&self.config.state_dir, image_identity); let image_path = image_cache_rootfs_image(&self.config.state_dir, image_identity); - let staging_dir = image_cache_staging_dir(&self.config.state_dir, image_identity); + let staging_dir = self.image_staging_dir(image_identity); let prepared_rootfs = staging_dir.join("rootfs"); let prepared_image = staging_dir.join(IMAGE_CACHE_ROOTFS_IMAGE); @@ -4079,23 +4378,8 @@ impl VmDriver { return Err(Status::failed_precondition(err)); } - if tokio::fs::metadata(&image_path).await.is_ok() { - info!( - image_identity = %image_identity, - "vm driver: another task wrote image while we were building, discarding ours" - ); - let _ = tokio::fs::remove_dir_all(&staging_dir).await; - return Ok(()); - } - - tokio::fs::rename(&prepared_image, &image_path) - .await - .map_err(|err| Status::internal(format!("store cached rootfs image failed: {err}")))?; - info!( - image_identity = %image_identity, - image_path = %image_path.display(), - "vm driver: root disk image committed to cache" - ); + self.publish_prepared_image(&prepared_image, &image_path) + .await?; let _ = tokio::fs::remove_dir_all(&staging_dir).await; Ok(()) } @@ -4305,6 +4589,89 @@ impl VmDriver { } } +/// Execute an internal image preparation request in its own process group. +/// The driver supplies a private request file and an inherited staging lease; +/// this entry point never restores sandboxes or launches a gateway listener. +#[doc(hidden)] +pub async fn run_image_preparation_worker(request_path: &Path) -> Result<(), String> { + let (request, directory) = preparation::read_request(request_path)?; + validate_sandbox_id(&request.sandbox_id).map_err(|error| error.message().to_string())?; + preparation::watch_parent(request.parent_pid)?; + tracing_subscriber::fmt() + .with_writer(std::io::stderr) + .with_env_filter(request.config.log_level.clone()) + .try_init() + .map_err(|error| format!("initialize image preparation logging: {error}"))?; + let (events, mut receiver) = broadcast::channel(WATCH_BUFFER); + let socket_root_fd = fs::File::open(&directory) + .map_err(|error| error.to_string())? + .into(); + let launcher_bin = request + .config + .launcher_bin + .clone() + .map_or_else(std::env::current_exe, Ok) + .map_err(|error| error.to_string())?; + let driver = VmDriver { + config: request.config, + socket_root: directory.clone(), + socket_root_fd: Arc::new(socket_root_fd), + launcher_bin, + registry: Arc::new(Mutex::new(HashMap::new())), + image_cache_lock: Arc::new(Mutex::new(())), + preparation_root: Some(directory), + events, + gpu_inventory: None, + lifecycle_extensions: Arc::new(LifecycleExtensionRegistry::new()), + }; + let prepare = async { + if let Some(overlay) = request.overlay { + driver + .prepare_requested_overlay(&request.sandbox_id, overlay) + .await + .map(preparation::Output::Overlay) + } else if request.bootstrap_only { + let identity = driver + .ensure_cached_bootstrap_rootfs_image(&request.sandbox_id, &request.image_ref) + .await?; + Ok(preparation::Output::Images(RuntimeImagePlan { + root_disk: image_cache_rootfs_image(&driver.config.state_dir, &identity), + image_disk: None, + image_identity: identity.clone(), + bootstrap_image_identity: identity, + })) + } else { + driver + .prepare_runtime_images( + &request.sandbox_id, + &request.image_ref, + request.rootfs_tar.as_deref(), + ) + .await + .map(preparation::Output::Images) + } + }; + tokio::pin!(prepare); + let result = loop { + tokio::select! { + result = &mut prepare => break result, + event = receiver.recv() => { + if let Ok(event) = event { + preparation::send(&preparation::Message::Event(event.encode_to_vec())).map_err(|error| error.to_string())?; + } + } + } + }; + while let Ok(event) = receiver.try_recv() { + preparation::send(&preparation::Message::Event(event.encode_to_vec())) + .map_err(|error| error.to_string())?; + } + preparation::send(&preparation::Message::Complete( + result.map_err(|error| error.message().to_string()), + )) + .map_err(|error| error.to_string()) +} + fn read_vm_console_tail(path: &Path, limit: u64) -> Option { if limit == 0 { return None; @@ -6277,8 +6644,10 @@ async fn overlay_template_image_ready(path: &Path, size_bytes: u64) -> Result Result<(), String> { if let Ok(metadata) = fs::metadata(template_path) && metadata.is_file() @@ -6300,30 +6669,23 @@ fn ensure_sandbox_overlay_template_image( ) })?; - let staging_image = parent.join(format!( - ".{}.staging-{}-{}", - template_path - .file_name() - .and_then(|name| name.to_str()) - .unwrap_or("overlay-template.ext4"), - std::process::id(), - openshell_core::time::now_ms() - )); + fs::create_dir_all(staging_dir).map_err(|error| error.to_string())?; + let staging_image = staging_dir.join("overlay-template.ext4"); let result = (|| { create_empty_sandbox_overlay_image(&staging_image, size_bytes)?; - fs::rename(&staging_image, template_path).map_err(|err| { - format!( - "move overlay template {} to {}: {err}", - staging_image.display(), - template_path.display() - ) - }) + preparation::publish_cache_file(cache_root, &staging_image, template_path, Some(size_bytes)) + .map_err(|err| { + format!( + "move overlay template {} to {}: {err}", + staging_image.display(), + template_path.display() + ) + }) })(); - if result.is_err() { - let _ = fs::remove_file(&staging_image); - } + // A competing publisher may have won; only our private staged file is disposable. + let _ = fs::remove_file(&staging_image); result } @@ -8467,6 +8829,61 @@ mod tests { ); } + #[tokio::test] + async fn overlay_worker_rejects_identity_conflicts_before_disk_changes() { + let directory = tempfile::tempdir().unwrap(); + let mut driver = test_driver_with_extensions(LifecycleExtensionRegistry::new()); + driver.config.state_dir = directory.path().to_path_buf(); + // A changed host default must not replace the persisted owner's identity. + driver.config.sandbox_uid = Some(10000); + driver.config.sandbox_gid = Some(10001); + let sandbox_id = "identity-worker"; + let state_dir = sandbox_state_dir(directory.path(), sandbox_id).unwrap(); + create_private_dir_all(&state_dir).await.unwrap(); + let owner = SandboxOwnerIdentity { + uid: 1000, + gid: 1001, + }; + write_sandbox_owner_state(&state_dir, owner).await.unwrap(); + let overlay_disk = sandbox_runtime_disk_paths(&state_dir).overlay_disk; + std::fs::write(&overlay_disk, b"existing overlay must not be touched").unwrap(); + + for (user, group, field) in [ + ("10000", "", "run_as_user"), + ("sandbox", "10001", "run_as_group"), + ] { + let requested_identity = WorkloadIdentityRequest { + user: user.into(), + group: group.into(), + }; + let request = preparation::OverlayRequest { + source_disk: image_cache_root_dir(directory.path()).join("must-not-read-image"), + preparation: OverlayPreparation::PreserveExisting, + requested_identity: Some(preparation::WorkloadIdentitySelectors::from( + &requested_identity, + )), + }; + // Exercise the serialized request and the same entry point the + // worker uses, so dropping either selector cannot weaken the check. + let wire = serde_json::to_vec(&request).unwrap(); + let decoded = serde_json::from_slice(&wire).unwrap(); + let error = driver + .prepare_requested_overlay(sandbox_id, decoded) + .await + .unwrap_err(); + assert!(error.message().contains(field), "{error}"); + assert!(error.message().contains("1000:1001"), "{error}"); + assert_eq!( + std::fs::read(&overlay_disk).unwrap(), + b"existing overlay must not be touched" + ); + assert_eq!( + std::fs::read_to_string(state_dir.join(SANDBOX_OWNER_STATE_FILE)).unwrap(), + owner.marker_contents() + ); + } + } + #[test] fn vm_workload_identity_checks_independent_numeric_and_symbolic_selectors() { let owner = SandboxOwnerIdentity { @@ -9131,6 +9548,7 @@ mod tests { state_dir: state_dir.clone(), process: None, provisioning_task: None, + preparation: None, gpu_bdf: None, deleting: false, }, @@ -9213,6 +9631,7 @@ mod tests { state_dir, process: None, provisioning_task: Some(provisioning_task), + preparation: None, gpu_bdf: None, deleting: false, }, @@ -9233,6 +9652,273 @@ mod tests { task.abort(); } + #[tokio::test] + async fn blocked_worker_does_not_delay_independent_image_and_overlay_requests() { + let root = tempfile::tempdir().unwrap(); + let mut driver = test_driver_with_extensions(LifecycleExtensionRegistry::new()); + driver.config.state_dir = root.path().to_path_buf(); + driver.config.bootstrap_image = "test-bootstrap".to_string(); + driver.launcher_bin = root.path().join("worker"); + let cached_root = root.path().join("images/warm/rootfs.ext4"); + fs::create_dir_all(cached_root.parent().unwrap()).unwrap(); + fs::write(&cached_root, b"cached bootstrap").unwrap(); + let images = + preparation::Message::Complete(Ok(preparation::Output::Images(RuntimeImagePlan { + root_disk: cached_root.clone(), + image_disk: None, + image_identity: "warm".to_string(), + bootstrap_image_identity: "warm".to_string(), + }))); + let overlay = preparation::Message::Complete(Ok(preparation::Output::Overlay( + SandboxOwnerIdentity { + uid: 1000, + gid: 1000, + }, + ))); + fs::write( + root.path().join("images.json"), + serde_json::to_vec(&images).unwrap(), + ) + .unwrap(); + fs::write( + root.path().join("overlay.json"), + serde_json::to_vec(&overlay).unwrap(), + ) + .unwrap(); + fs::write( + &driver.launcher_bin, + r#"#!/bin/sh + control=$(dirname "$0") + request=$(cat "$2") + case "$request" in + *'"sandbox_id":"slow-a"'*) + : > "$control/ready" + exec sleep 300 + ;; + *'"overlay":null'*) cat "$control/images.json" ;; + *) cat "$control/overlay.json" ;; + esac + printf '\n' + "#, + ) + .unwrap(); + fs::set_permissions(&driver.launcher_bin, fs::Permissions::from_mode(0o700)).unwrap(); + for id in ["slow-a", "warm-b"] { + let state_dir = sandboxes_root_dir(root.path()).join(id); + create_private_dir_all(&state_dir).await.unwrap(); + driver.registry.lock().await.insert( + id.to_string(), + SandboxRecord { + snapshot: Sandbox { + id: id.to_string(), + ..Default::default() + }, + state_dir, + process: None, + provisioning_task: None, + preparation: None, + gpu_bdf: None, + deleting: false, + }, + ); + } + + let slow_driver = driver.clone(); + let slow = tokio::spawn(async move { + slow_driver + .prepare_images_in_worker("slow-a", "test-bootstrap", None, true) + .await + }); + let warm_driver = driver.clone(); + let ready = root.path().join("ready"); + let mut warm = tokio::spawn(async move { + while !ready.exists() { + tokio::time::sleep(Duration::from_millis(10)).await; + } + let plan = warm_driver + .prepare_images_in_worker("warm-b", "test-bootstrap", None, true) + .await + .map_err(|error| error.to_string())?; + if plan.root_disk != cached_root { + return Err("warm image result was not preserved".to_string()); + } + let owner = warm_driver + .prepare_overlay_in_worker( + "warm-b", + &plan.root_disk, + OverlayPreparation::Fresh, + None, + ) + .await + .map_err(|error| error.to_string())?; + if owner + != (SandboxOwnerIdentity { + uid: 1000, + gid: 1000, + }) + { + return Err("warm overlay result was not preserved".to_string()); + } + Ok::<(), String>(()) + }); + let outcome = match tokio::time::timeout(Duration::from_secs(10), &mut warm).await { + Ok(Ok(result)) => result, + Ok(Err(error)) => Err(format!("warm task failed: {error}")), + Err(_) => { + warm.abort(); + let _ = warm.await; + Err("warm request waited for the unrelated slow worker".to_string()) + } + }; + let slow_still_running = !slow.is_finished(); + slow.abort(); + let _ = slow.await; + let warm_cleanup = driver.cleanup_image_preparation("warm-b").await; + let slow_cleanup = driver.cleanup_image_preparation("slow-a").await; + warm_cleanup.expect("warm worker cleanup"); + slow_cleanup.expect("slow worker cleanup"); + outcome.expect("independent image and overlay requests must complete"); + assert!( + slow_still_running, + "the slow worker must remain blocked during the proof" + ); + } + + async fn lifecycle_cancels_image_preparation(delete: bool) { + let temp = tempfile::tempdir().unwrap(); + let mut driver = test_driver_with_extensions(LifecycleExtensionRegistry::new()); + driver.config.state_dir = temp.path().to_path_buf(); + let id = "sandbox-cancel-image"; + let state_dir = sandboxes_root_dir(temp.path()).join(id); + create_private_dir_all(&state_dir).await.unwrap(); + let launcher = temp.path().join("worker"); + fs::write(&launcher, "#!/bin/sh\nroot=$(dirname \"$2\")\nsleep 300 &\nprintf ready > \"$root/ready\"\nwait\n").unwrap(); + fs::set_permissions(&launcher, fs::Permissions::from_mode(0o700)).unwrap(); + let cache = image_cache_root_dir(temp.path()); + let mut attempt = preparation::Attempt::create(&cache).unwrap(); + let directory = attempt.directory.clone(); + let _stdout = attempt + .spawn( + &launcher, + preparation::Request { + config: driver.config.clone(), + sandbox_id: id.to_string(), + image_ref: "test-image".to_string(), + rootfs_tar: None, + bootstrap_only: false, + overlay: None, + lease_fd: -1, + parent_pid: 0, + }, + ) + .unwrap(); + tokio::time::timeout(Duration::from_secs(5), async { + while !directory.join("ready").exists() { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .expect("preparation worker readiness"); + let attempt = Arc::new(Mutex::new(attempt)); + driver.registry.lock().await.insert( + id.to_string(), + SandboxRecord { + snapshot: Sandbox { + id: id.to_string(), + name: "cancel-image".to_string(), + ..Default::default() + }, + state_dir: state_dir.clone(), + process: None, + provisioning_task: Some(tokio::spawn(std::future::pending())), + preparation: Some(attempt.clone()), + gpu_bdf: None, + deleting: false, + }, + ); + if delete { + let response = driver.delete_sandbox(id, "").await.unwrap(); + assert!(response.deleted); + assert!(!state_dir.exists()); + } else { + driver.stop_sandbox(id, "").await.unwrap(); + assert!(state_dir.join(SANDBOX_STOPPED_FILE).exists()); + } + assert!( + !directory.exists(), + "lifecycle completion must reclaim preparation staging" + ); + attempt + .lock() + .await + .cleanup() + .await + .expect("worker was reaped and cleanup is idempotent"); + } + + #[tokio::test] + async fn stop_waits_for_image_preparation_cleanup() { + lifecycle_cancels_image_preparation(false).await; + } + + #[tokio::test] + async fn delete_waits_for_image_preparation_cleanup() { + lifecycle_cancels_image_preparation(true).await; + } + + #[tokio::test] + async fn incomplete_preparation_cleanup_failure_preserves_sandbox_state() { + let temp = tempfile::tempdir().unwrap(); + let mut driver = test_driver_with_extensions(LifecycleExtensionRegistry::new()); + driver.config.state_dir = temp.path().to_path_buf(); + let id = "sandbox-incomplete-cleanup"; + let state_dir = sandboxes_root_dir(temp.path()).join(id); + create_private_dir_all(&state_dir).await.unwrap(); + let overlay = state_dir.join(SANDBOX_OVERLAY_IMAGE); + fs::write(&overlay, b"owned overlay").unwrap(); + let attempt = Arc::new(Mutex::new( + preparation::Attempt::create(&image_cache_root_dir(temp.path())).unwrap(), + )); + driver.registry.lock().await.insert( + id.to_string(), + SandboxRecord { + snapshot: Sandbox { + id: id.to_string(), + ..Default::default() + }, + state_dir: state_dir.clone(), + process: None, + provisioning_task: None, + preparation: Some(attempt), + gpu_bdf: None, + deleting: false, + }, + ); + + // Failed worker cleanup retains the registered attempt. Its failure + // epilogue must not remove state that a descendant could still write. + driver + .fail_provisioning( + id, + &state_dir, + "PreparationFailed", + "image preparation descendants still own staging; files retained", + true, + ) + .await; + let retained = fs::read(&overlay); + driver.cleanup_image_preparation(id).await.unwrap(); + assert_eq!(retained.unwrap(), b"owned overlay"); + + driver + .fail_provisioning(id, &state_dir, "PreparationFailed", "worker stopped", true) + .await; + assert!( + !state_dir.exists(), + "successful worker cleanup permits removing failed sandbox state" + ); + } + fn test_launch_authentication(label: &str) -> (Vec, openshell_core::SandboxSessionId) { use openshell_core::jwt::{ SandboxLaunchAuthentication, SecretJwt, SessionVerificationKey, SupervisorAuthBundle, @@ -9356,6 +10042,7 @@ mod tests { launcher_bin: PathBuf::from("/tmp/openshell-driver-vm"), registry: Arc::new(Mutex::new(HashMap::new())), image_cache_lock: Arc::new(Mutex::new(())), + preparation_root: None, events: broadcast::channel(WATCH_BUFFER).0, gpu_inventory: None, lifecycle_extensions: Arc::new(LifecycleExtensionRegistry::new()), @@ -9395,6 +10082,7 @@ mod tests { launcher_bin: PathBuf::from("/tmp/openshell-driver-vm"), registry: Arc::new(Mutex::new(HashMap::new())), image_cache_lock: Arc::new(Mutex::new(())), + preparation_root: None, events: broadcast::channel(WATCH_BUFFER).0, gpu_inventory: None, lifecycle_extensions: Arc::new(LifecycleExtensionRegistry::new()), @@ -9429,6 +10117,7 @@ mod tests { launcher_bin: PathBuf::from("/tmp/openshell-driver-vm"), registry: Arc::new(Mutex::new(HashMap::new())), image_cache_lock: Arc::new(Mutex::new(())), + preparation_root: None, events: broadcast::channel(WATCH_BUFFER).0, gpu_inventory: None, lifecycle_extensions: Arc::new(LifecycleExtensionRegistry::new()), @@ -9457,6 +10146,7 @@ mod tests { launcher_bin: PathBuf::from("/tmp/openshell-driver-vm"), registry: Arc::new(Mutex::new(HashMap::new())), image_cache_lock: Arc::new(Mutex::new(())), + preparation_root: None, events: broadcast::channel(WATCH_BUFFER).0, gpu_inventory: None, lifecycle_extensions: Arc::new(LifecycleExtensionRegistry::new()), @@ -9486,6 +10176,7 @@ mod tests { launcher_bin: PathBuf::from("/tmp/openshell-driver-vm"), registry: Arc::new(Mutex::new(HashMap::new())), image_cache_lock: Arc::new(Mutex::new(())), + preparation_root: None, events: broadcast::channel(WATCH_BUFFER).0, gpu_inventory: None, lifecycle_extensions: Arc::new(LifecycleExtensionRegistry::new()), @@ -9510,6 +10201,7 @@ mod tests { launcher_bin: PathBuf::from("/tmp/openshell-driver-vm"), registry: Arc::new(Mutex::new(HashMap::new())), image_cache_lock: Arc::new(Mutex::new(())), + preparation_root: None, events: broadcast::channel(WATCH_BUFFER).0, gpu_inventory: None, lifecycle_extensions: Arc::new(LifecycleExtensionRegistry::new()), @@ -9531,6 +10223,7 @@ mod tests { launcher_bin: PathBuf::from("/tmp/openshell-driver-vm"), registry: Arc::new(Mutex::new(HashMap::new())), image_cache_lock: Arc::new(Mutex::new(())), + preparation_root: None, events: broadcast::channel(WATCH_BUFFER).0, gpu_inventory: None, lifecycle_extensions: Arc::new(LifecycleExtensionRegistry::new()), @@ -10177,6 +10870,7 @@ mod tests { launcher_bin: PathBuf::from("openshell-driver-vm"), registry: Arc::new(Mutex::new(HashMap::new())), image_cache_lock: Arc::new(Mutex::new(())), + preparation_root: None, events, gpu_inventory: None, lifecycle_extensions: Arc::new(LifecycleExtensionRegistry::new()), @@ -10242,6 +10936,7 @@ mod tests { launcher_bin: PathBuf::from("openshell-driver-vm"), registry: Arc::new(Mutex::new(HashMap::new())), image_cache_lock: Arc::new(Mutex::new(())), + preparation_root: None, events, gpu_inventory: None, lifecycle_extensions: Arc::new(LifecycleExtensionRegistry::new()), @@ -10262,6 +10957,7 @@ mod tests { state_dir: state_dir.clone(), process: None, provisioning_task: None, + preparation: None, gpu_bdf: None, deleting: false, }, @@ -10296,6 +10992,7 @@ mod tests { launcher_bin: PathBuf::from("openshell-driver-vm"), registry: Arc::new(Mutex::new(HashMap::new())), image_cache_lock: Arc::new(Mutex::new(())), + preparation_root: None, events, gpu_inventory: None, lifecycle_extensions: Arc::new(LifecycleExtensionRegistry::new()), @@ -10317,6 +11014,7 @@ mod tests { state_dir: state_dir.clone(), process: None, provisioning_task: None, + preparation: None, gpu_bdf: None, deleting: false, }, @@ -10671,6 +11369,7 @@ mod tests { state_dir, process: Some(process), provisioning_task: None, + preparation: None, gpu_bdf: None, deleting: false, }, @@ -10698,6 +11397,7 @@ mod tests { launcher_bin: PathBuf::from("openshell-driver-vm"), registry: Arc::new(Mutex::new(HashMap::new())), image_cache_lock: Arc::new(Mutex::new(())), + preparation_root: None, events, gpu_inventory: None, lifecycle_extensions: Arc::new(LifecycleExtensionRegistry::new()), @@ -11041,6 +11741,7 @@ mod tests { launcher_bin: PathBuf::from("openshell-driver-vm"), registry: Arc::new(Mutex::new(HashMap::new())), image_cache_lock: Arc::new(Mutex::new(())), + preparation_root: None, events, gpu_inventory: None, lifecycle_extensions: Arc::new(extensions), @@ -11649,6 +12350,7 @@ mod tests { state_dir: state_dir.clone(), process: None, provisioning_task: None, + preparation: None, gpu_bdf: None, deleting: false, }, diff --git a/crates/openshell-driver-vm/src/main.rs b/crates/openshell-driver-vm/src/main.rs index d7195b4cc3..4b4bcf3f6a 100644 --- a/crates/openshell-driver-vm/src/main.rs +++ b/crates/openshell-driver-vm/src/main.rs @@ -35,6 +35,9 @@ struct Args { #[arg(long, hide = true, default_value_t = false)] internal_run_vm: bool, + #[arg(long, hide = true)] + internal_prepare_image: Option, + #[arg(long = "vm-root-disk", hide = true, alias = "vm-rootfs")] vm_root_disk: Option, @@ -249,6 +252,12 @@ struct Args { #[tokio::main] async fn main() -> Result<()> { let args = Args::parse(); + if let Some(request) = args.internal_prepare_image { + openshell_driver_vm::driver::run_image_preparation_worker(&request) + .await + .map_err(|error| miette::miette!("{error}"))?; + return Ok(()); + } if args.internal_run_vm { // The VM launcher arms procguard after resolving its runtime so its // libkrun worker cannot outlive the launcher. diff --git a/crates/openshell-driver-vm/src/preparation.rs b/crates/openshell-driver-vm/src/preparation.rs new file mode 100644 index 0000000000..51f087e5e5 --- /dev/null +++ b/crates/openshell-driver-vm/src/preparation.rs @@ -0,0 +1,1016 @@ +// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +//! Own image-preparation processes and the files they may still be writing. +//! +//! A lease is inherited by every preparation subprocess. Cancellation first +//! kills and reaps the worker; staging is removed only after the last lease +//! descriptor closes. A restarted driver uses the same proof of inactivity. + +#![allow(unsafe_code)] + +use std::fs::{self, File, OpenOptions}; +use std::io::{self, Read, Write}; +use std::os::fd::AsRawFd; +use std::os::unix::fs::OpenOptionsExt; +use std::path::{Path, PathBuf}; +use std::process::{ExitStatus, Stdio}; +use std::sync::Arc; +use std::time::Duration; + +use nix::sys::signal::{Signal, killpg}; +use nix::unistd::Pid; +use serde::{Deserialize, Serialize}; +use tokio::process::{Child, ChildStdout, Command}; +use tokio::sync::Mutex; +use tonic::Status; + +use super::{ + OverlayPreparation, RuntimeImagePlan, SandboxOwnerIdentity, VmDriverConfig, + WorkloadIdentityRequest, +}; + +const ATTEMPTS_DIR: &str = "preparations"; +const REQUEST_FILE: &str = "request.json"; +const LEASE_MARKER: &[u8] = b"openshell-image-preparation-v1\n"; +const CLEANUP_TIMEOUT: Duration = Duration::from_secs(10); + +#[derive(Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub(super) struct Request { + pub config: VmDriverConfig, + pub sandbox_id: String, + pub image_ref: String, + pub rootfs_tar: Option, + pub bootstrap_only: bool, + pub overlay: Option, + pub lease_fd: i32, + pub parent_pid: u32, +} + +#[derive(Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub(super) struct OverlayRequest { + pub source_disk: PathBuf, + pub preparation: OverlayPreparation, + pub requested_identity: Option, +} + +/// Keep the requested selectors across the worker boundary. The worker must +/// validate them against the persisted overlay owner before changing any files; +/// the parent's current default identity cannot substitute for that owner. +#[derive(Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub(super) struct WorkloadIdentitySelectors { + user: String, + group: String, +} + +impl From<&WorkloadIdentityRequest> for WorkloadIdentitySelectors { + fn from(request: &WorkloadIdentityRequest) -> Self { + Self { + user: request.user.clone(), + group: request.group.clone(), + } + } +} + +impl From for WorkloadIdentityRequest { + fn from(selectors: WorkloadIdentitySelectors) -> Self { + Self { + user: selectors.user, + group: selectors.group, + } + } +} + +#[derive(Serialize, Deserialize)] +pub(super) enum Output { + Images(RuntimeImagePlan), + Overlay(SandboxOwnerIdentity), +} + +#[derive(Serialize, Deserialize)] +pub(super) enum Message { + Event(Vec), + Complete(Result), +} + +pub(super) struct Attempt { + pub directory: PathBuf, + lease_path: PathBuf, + lease: Option, + child: Option, +} + +/// Cleanup survives cancellation of the provisioning future itself. Lifecycle +/// calls also await the same mutex, so they cannot report completion early. +pub(super) struct CleanupOnDrop(pub Arc>); + +impl Drop for CleanupOnDrop { + fn drop(&mut self) { + let attempt = self.0.clone(); + tokio::spawn(async move { + if let Err(error) = attempt.lock().await.cleanup().await { + tracing::warn!(%error, "image preparation cleanup incomplete; staging retained"); + } + }); + } +} + +impl Attempt { + /// Create and lock the lease before publishing the directory. A concurrent + /// reconciler can never observe this attempt without its ownership lock. + pub fn create(cache_root: &Path) -> io::Result { + let root = cache_root.join(ATTEMPTS_DIR); + fs::create_dir_all(&root)?; + let name = format!("attempt-{:032x}", rand::random::()); + let directory = root.join(&name); + let lease_path = root.join(format!("{name}.lease")); + let mut lease = OpenOptions::new() + .read(true) + .write(true) + .create_new(true) + .mode(0o600) + .open(&lease_path)?; + lock(&lease, false)?; + lease.write_all(LEASE_MARKER)?; + lease.sync_data()?; + fs::create_dir(&directory)?; + Ok(Self { + directory, + lease_path, + lease: Some(lease), + child: None, + }) + } + + /// Spawn while the caller holds the sandbox registry lock, then register + /// this attempt before yielding. Stop/delete cannot miss a live worker. + pub fn spawn(&mut self, launcher: &Path, mut request: Request) -> io::Result { + let lease = self + .lease + .as_ref() + .ok_or_else(|| io::Error::other("preparation lease is closed"))?; + let fd = lease.as_raw_fd(); + request.lease_fd = fd; + request.parent_pid = std::process::id(); + let request_path = self.directory.join(REQUEST_FILE); + let mut file = OpenOptions::new() + .write(true) + .create_new(true) + .mode(0o600) + .open(&request_path)?; + serde_json::to_writer(&mut file, &request)?; + file.flush()?; + let mut command = Command::new(launcher); + command.arg("--internal-prepare-image").arg(request_path); + command + .stdin(Stdio::null()) + .stdout(Stdio::piped()) + .stderr(Stdio::inherit()); + command.process_group(0).kill_on_drop(true); + // SAFETY: the child only makes an async-signal-safe fcntl call. The + // parent keeps this descriptor open until the child has exited. + unsafe { + command.pre_exec(move || { + if libc::fcntl(fd, libc::F_SETFD, 0) == -1 { + return Err(io::Error::last_os_error()); + } + Ok(()) + }); + } + let mut child = command.spawn()?; + let stdout = child + .stdout + .take() + .ok_or_else(|| io::Error::other("preparation stdout is missing"))?; + self.child = Some(child); + Ok(stdout) + } + + pub async fn wait(&mut self) -> io::Result { + // EOF can precede observable process exit. Let normal worker teardown + // finish before signaling, preserving its real status and reserving + // its PID until any remaining descendants have been stopped. + wait_for_worker_exit(self.child.as_ref().and_then(Child::id)).await?; + self.kill_group()?; + self.child + .as_mut() + .ok_or_else(|| io::Error::other("preparation worker is missing"))? + .wait() + .await + } + + /// Do not report successful cleanup while a formatter or other descendant + /// still owns the lease. Preserve uncertain staging for the next reconcile. + pub async fn cleanup(&mut self) -> Result<(), Status> { + let signal_result = self.kill_group(); + #[cfg(target_os = "macos")] + let signal_result = match signal_result { + // Darwin can reject signaling an exiting process before waitid + // exposes its terminal status. Keep immediate cancellation first, + // then retry the same guarded signal only after observing exit. + Err(error) if error.raw_os_error() == Some(libc::EPERM) => { + wait_for_worker_exit(self.child.as_ref().and_then(Child::id)) + .await + .and_then(|()| self.kill_group()) + } + result => result, + }; + signal_result + .map_err(|error| Status::internal(format!("terminate image preparation: {error}")))?; + if let Some(child) = self.child.as_mut() { + tokio::time::timeout(CLEANUP_TIMEOUT, child.wait()) + .await + .map_err(|_| { + Status::deadline_exceeded("image preparation did not stop; staging retained") + })? + .map_err(|error| Status::internal(format!("reap image preparation: {error}")))?; + } + // Do not explicitly unlock: inherited descriptors must continue to + // protect files until the last worker or descendant closes its copy. + self.lease.take(); + let deadline = tokio::time::Instant::now() + CLEANUP_TIMEOUT; + loop { + match reclaim(&self.directory, &self.lease_path) { + Ok(true) => return Ok(()), + Ok(false) if tokio::time::Instant::now() < deadline => { + tokio::time::sleep(Duration::from_millis(20)).await; + } + Ok(false) => { + return Err(Status::deadline_exceeded( + "image preparation descendants still own staging; files retained", + )); + } + Err(error) => { + return Err(Status::internal(format!( + "reclaim image preparation staging: {error}" + ))); + } + } + } + } + + // macOS may replace the staging lease when proving an exited process group + // has no remaining owner; other platforms only read the worker state. + #[cfg_attr(not(target_os = "macos"), allow(clippy::needless_pass_by_ref_mut))] + fn kill_group(&mut self) -> io::Result<()> { + if let Some(id) = self.child.as_ref().and_then(Child::id) { + let pid = i32::try_from(id).map_err(io::Error::other)?; + match killpg(Pid::from_raw(pid), Signal::SIGKILL) { + Ok(()) | Err(nix::errno::Errno::ESRCH) => {} + #[cfg(target_os = "macos")] + Err(nix::errno::Errno::EPERM) + if self.exited_worker_has_no_descendant_owner(id)? => {} + Err(error) => return Err(io::Error::from_raw_os_error(error as i32)), + } + } + Ok(()) + } + + /// Darwin rejects signals to a group containing only an exited leader. + /// Accept that case only while its PID remains reserved and no descendant + /// owns the inherited staging lease. Live or uncertain ownership still fails. + #[cfg(target_os = "macos")] + fn exited_worker_has_no_descendant_owner(&mut self, id: u32) -> io::Result { + if !worker_has_exited(id)? { + return Ok(false); + } + + // Open the probe before dropping our copy so a concurrent reconciler + // cannot remove the lease pathname between release and open. Reacquire + // ownership on success and retain it until the caller reaps the worker. + let lease = OpenOptions::new() + .read(true) + .write(true) + .custom_flags(libc::O_NOFOLLOW) + .open(&self.lease_path)?; + self.lease.take(); + if !lock(&lease, true)? { + return Ok(false); + } + self.lease = Some(lease); + Ok(true) + } +} + +/// Observe termination without releasing the PID or process-group identity. +fn worker_has_exited(id: u32) -> io::Result { + // SAFETY: waitid writes initialized storage. WNOWAIT leaves the owned + // child unreaped, so its PID cannot be reused before descendant signaling. + let mut status: libc::siginfo_t = unsafe { std::mem::zeroed() }; + if unsafe { + libc::waitid( + libc::P_PID, + id, + &raw mut status, + libc::WEXITED | libc::WNOHANG | libc::WNOWAIT, + ) + } != 0 + { + return Err(io::Error::last_os_error()); + } + Ok(matches!( + status.si_code, + libc::CLD_EXITED | libc::CLD_KILLED | libc::CLD_DUMPED + )) +} + +/// Bound EOF-to-exit observation without blocking the runtime or reaping. +/// Dropping this future leaves the worker available to cancellation cleanup. +async fn wait_for_worker_exit(id: Option) -> io::Result<()> { + let Some(id) = id else { + return Ok(()); + }; + tokio::time::timeout(CLEANUP_TIMEOUT, async { + loop { + if worker_has_exited(id)? { + return Ok(()); + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .map_err(|_| { + io::Error::new( + io::ErrorKind::TimedOut, + "image preparation worker did not exit before cleanup deadline", + ) + })? +} + +impl Drop for Attempt { + fn drop(&mut self) { + // Runtime shutdown may prevent the asynchronous reaper from running. + // Signal only this worker's still-unreaped process group; the lease + // remains held by descendants until the kernel closes their files. + let _ = self.kill_group(); + } +} + +/// Remove only directories created by this protocol whose inherited lease has +/// no remaining owner. Legacy staging and shared committed images are outside +/// this namespace and are never inferred to be inactive from their age. +pub(super) fn reconcile(cache_root: &Path) -> io::Result<()> { + let root = cache_root.join(ATTEMPTS_DIR); + let entries = match fs::read_dir(&root) { + Ok(entries) => entries, + Err(error) if error.kind() == io::ErrorKind::NotFound => return Ok(()), + Err(error) => return Err(error), + }; + for entry in entries { + let entry = entry?; + let name = entry.file_name(); + let Some(name) = name.to_str().filter(|name| valid_attempt_name(name)) else { + continue; + }; + if !entry.file_type()?.is_dir() { + continue; + } + if let Err(error) = reclaim(&entry.path(), &root.join(format!("{name}.lease"))) { + tracing::warn!(path = %entry.path().display(), %error, "image preparation ownership uncertain; staging retained"); + } + } + Ok(()) +} + +fn valid_attempt_name(name: &str) -> bool { + name.strip_prefix("attempt-").is_some_and(|suffix| { + suffix.len() == 32 && suffix.bytes().all(|byte| byte.is_ascii_hexdigit()) + }) +} + +fn reclaim(directory: &Path, lease_path: &Path) -> io::Result { + if !directory.try_exists()? && !lease_path.try_exists()? { + return Ok(true); + } + let lease = match OpenOptions::new() + .read(true) + .write(true) + .custom_flags(libc::O_NOFOLLOW) + .open(lease_path) + { + Ok(lease) if lease.metadata()?.is_file() => lease, + Ok(_) => return Ok(false), + Err(error) if error.kind() == io::ErrorKind::NotFound => return Ok(false), + Err(error) => return Err(error), + }; + if !lock(&lease, true)? { + return Ok(false); + } + let mut marker = Vec::new(); + (&lease).take(64).read_to_end(&mut marker)?; + if marker != LEASE_MARKER { + return Ok(false); + } + match fs::remove_dir_all(directory) { + Ok(()) => {} + Err(error) if error.kind() == io::ErrorKind::NotFound => {} + Err(error) => return Err(error), + } + match fs::remove_file(lease_path) { + Ok(()) => {} + Err(error) if error.kind() == io::ErrorKind::NotFound => {} + Err(error) => return Err(error), + } + Ok(true) +} + +fn lock(file: &File, nonblocking: bool) -> io::Result { + let operation = libc::LOCK_EX | if nonblocking { libc::LOCK_NB } else { 0 }; + // SAFETY: flock receives a valid borrowed file descriptor and no pointers. + if unsafe { libc::flock(file.as_raw_fd(), operation) } == 0 { + return Ok(true); + } + let error = io::Error::last_os_error(); + if nonblocking && error.kind() == io::ErrorKind::WouldBlock { + Ok(false) + } else { + Err(error) + } +} + +pub(super) fn cache_lock(cache_root: &Path) -> io::Result { + let file = OpenOptions::new() + .read(true) + .write(true) + .create(true) + .truncate(false) + .mode(0o600) + .custom_flags(libc::O_NOFOLLOW) + .open(cache_root.join("preparation.lock"))?; + lock(&file, false)?; + Ok(file) +} + +/// Publish a complete staged image without replacing a ready cache entry. +/// Preparation stays outside this lock; only the readiness recheck and atomic +/// rename are serialized across workers. Call from a blocking task so the +/// lock remains owned until publication finishes even if its waiter is dropped. +pub(super) fn publish_cache_file( + cache_root: &Path, + staged: &Path, + destination: &Path, + expected_size: Option, +) -> io::Result<()> { + let _lease = cache_lock(cache_root)?; + match fs::metadata(destination) { + Ok(metadata) + if expected_size.is_none_or(|size| metadata.is_file() && metadata.len() == size) => + { + return Ok(()); + } + Ok(_) => {} + Err(error) if error.kind() == io::ErrorKind::NotFound => {} + Err(error) => return Err(error), + } + fs::rename(staged, destination) +} + +pub(super) fn read_request(path: &Path) -> Result<(Request, PathBuf), String> { + use std::os::unix::fs::MetadataExt; + let file = OpenOptions::new() + .read(true) + .custom_flags(libc::O_NOFOLLOW | libc::O_NONBLOCK) + .open(path) + .map_err(|error| error.to_string())?; + let metadata = file.metadata().map_err(|error| error.to_string())?; + if !metadata.is_file() || metadata.len() > 1024 * 1024 { + return Err( + "image preparation request must be a regular file no larger than 1 MiB".to_string(), + ); + } + let request: Request = serde_json::from_reader(file).map_err(|error| error.to_string())?; + let directory = path + .parent() + .ok_or("preparation request has no directory")?; + let name = directory + .file_name() + .and_then(|name| name.to_str()) + .filter(|name| valid_attempt_name(name)) + .ok_or("invalid preparation directory name")?; + let expected_root = super::image_cache_root_dir(&request.config.state_dir).join(ATTEMPTS_DIR); + if path.file_name().and_then(|name| name.to_str()) != Some(REQUEST_FILE) + || directory.parent() != Some(expected_root.as_path()) + { + return Err("preparation request is outside the driver staging directory".to_string()); + } + let lease_file = OpenOptions::new() + .read(true) + .custom_flags(libc::O_NOFOLLOW | libc::O_NONBLOCK) + .open(expected_root.join(format!("{name}.lease"))) + .map_err(|error| error.to_string())?; + let lease = lease_file.metadata().map_err(|error| error.to_string())?; + if !lease.is_file() { + return Err("image preparation lease must be a regular file".to_string()); + } + let mut marker = Vec::new(); + lease_file + .take(64) + .read_to_end(&mut marker) + .map_err(|error| error.to_string())?; + if marker != LEASE_MARKER { + return Err("image preparation lease marker is invalid".to_string()); + } + // SAFETY: fstat writes only to the supplied initialized stat storage. An + // invalid inherited descriptor is rejected rather than taken into ownership. + let mut inherited: libc::stat = unsafe { std::mem::zeroed() }; + if request.lease_fd < 3 || unsafe { libc::fstat(request.lease_fd, &raw mut inherited) } != 0 { + return Err("image preparation lease was not inherited".to_string()); + } + // Match MetadataExt::dev's representation: Darwin dev_t is signed, while + // Linux dev_t is already u64. The cast intentionally preserves that API's + // conversion, including the sign extension of a signed device identifier. + #[allow(clippy::cast_sign_loss, trivial_numeric_casts)] + let inherited_device = inherited.st_dev as u64; + if inherited.st_ino != lease.ino() || inherited_device != lease.dev() { + return Err("image preparation lease was not inherited".to_string()); + } + Ok((request, directory.to_path_buf())) +} + +/// A process-group watcher also covers host utilities on Linux, where a +/// parent-death signal on the worker alone cannot kill its children. +pub(super) fn watch_parent(parent_pid: u32) -> Result<(), String> { + let expected = i32::try_from(parent_pid).map_err(|error| error.to_string())?; + if nix::unistd::getpgrp() != nix::unistd::getpid() { + return Err("preparation worker must own its process group".to_string()); + } + std::thread::Builder::new() + .name("image-prep-parent".to_string()) + .spawn(move || { + loop { + if nix::unistd::getppid().as_raw() != expected { + let _ = killpg(nix::unistd::getpgrp(), Signal::SIGKILL); + std::process::exit(1); + } + std::thread::sleep(Duration::from_millis(50)); + } + }) + .map_err(|error| error.to_string())?; + Ok(()) +} + +pub(super) fn send(message: &Message) -> io::Result<()> { + let mut output = io::stdout().lock(); + serde_json::to_writer(&mut output, message)?; + output.write_all(b"\n")?; + output.flush() +} + +#[cfg(test)] +mod tests { + use super::*; + use std::os::unix::fs::{PermissionsExt, symlink}; + + fn request(root: &Path) -> Request { + Request { + config: VmDriverConfig { + state_dir: root.to_path_buf(), + ..VmDriverConfig::default() + }, + sandbox_id: "test-sandbox".to_string(), + image_ref: "test-image".to_string(), + rootfs_tar: None, + bootstrap_only: false, + overlay: None, + lease_fd: -1, + parent_pid: 0, + } + } + + async fn wait_for_file(path: &Path) { + tokio::time::timeout(Duration::from_secs(5), async { + while !path.exists() { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .expect("worker readiness"); + } + + async fn wait_for_worker_exit_without_reaping(attempt: &Attempt) { + let id = attempt.child.as_ref().unwrap().id().unwrap(); + tokio::time::timeout(Duration::from_secs(5), async { + loop { + let exited = { + // SAFETY: waitid writes initialized exit information and + // WNOWAIT leaves the owned child's PID reserved for cleanup. + let mut status: libc::siginfo_t = unsafe { std::mem::zeroed() }; + assert_eq!( + unsafe { + libc::waitid( + libc::P_PID, + id, + &raw mut status, + libc::WEXITED | libc::WNOHANG | libc::WNOWAIT, + ) + }, + 0 + ); + matches!( + status.si_code, + libc::CLD_EXITED | libc::CLD_KILLED | libc::CLD_DUMPED + ) + }; + if exited { + break; + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .expect("worker must exit without being reaped"); + } + + async fn worker_at_stdout_eof(temp: &tempfile::TempDir, script: &str) -> Attempt { + let launcher = temp.path().join("worker"); + fs::write(&launcher, script).unwrap(); + fs::set_permissions(&launcher, fs::Permissions::from_mode(0o700)).unwrap(); + let cache = super::super::image_cache_root_dir(temp.path()); + let mut attempt = Attempt::create(&cache).unwrap(); + let mut stdout = attempt.spawn(&launcher, request(temp.path())).unwrap(); + let mut output = Vec::new(); + tokio::time::timeout( + Duration::from_secs(5), + tokio::io::AsyncReadExt::read_to_end(&mut stdout, &mut output), + ) + .await + .expect("worker must close stdout") + .unwrap(); + attempt + } + + #[tokio::test] + async fn stdout_eof_keeps_successful_worker_exit_status() { + let temp = tempfile::tempdir().unwrap(); + // stdout can close before the kernel exposes the final exit status. + // Keep that interval deterministic without requiring scheduler timing. + let mut attempt = + worker_at_stdout_eof(&temp, "#!/bin/sh\nexec 1>&-\nsleep 0.1\nexit 0\n").await; + let status = attempt.wait().await; + attempt.cleanup().await.expect("cleanup EOF worker"); + + let status = status.expect("wait after stdout EOF"); + assert!(status.success(), "worker exit after stdout EOF: {status}"); + assert!(!attempt.directory.exists()); + } + + #[tokio::test] + async fn stdout_eof_with_live_descendant_cleans_group_after_worker_exit() { + let temp = tempfile::tempdir().unwrap(); + let mut attempt = worker_at_stdout_eof( + &temp, + "#!/bin/sh\nsleep 300 >/dev/null &\nexec 1>&-\nsleep 0.1\nexit 0\n", + ) + .await; + let status = attempt.wait().await; + attempt.cleanup().await.expect("stop inherited lease owner"); + + let status = status.expect("preserve leader status"); + assert!(status.success(), "worker exit with descendant: {status}"); + assert!(!attempt.directory.exists()); + assert!(attempt.child.as_ref().unwrap().id().is_none()); + } + + #[tokio::test] + async fn cancelling_stdout_eof_wait_leaves_worker_for_cleanup() { + let temp = tempfile::tempdir().unwrap(); + let mut attempt = worker_at_stdout_eof(&temp, "#!/bin/sh\nexec 1>&-\nsleep 300\n").await; + let result = tokio::time::timeout(Duration::from_millis(50), attempt.wait()).await; + attempt + .cleanup() + .await + .expect("cancelled wait cleans worker"); + + assert!(result.is_err(), "waiting for exit must remain cancellable"); + assert!(!attempt.directory.exists()); + assert!(attempt.child.as_ref().unwrap().id().is_none()); + } + + #[cfg(target_os = "macos")] + #[tokio::test] + async fn cleanup_after_stdout_eof_waits_for_exit_publication() { + // Repeated natural exits exercise Darwin's short interval between + // closing stdout and making terminal status observable to waitid. + for _ in 0..8 { + let temp = tempfile::tempdir().unwrap(); + let mut attempt = worker_at_stdout_eof(&temp, "#!/bin/sh\nexit 0\n").await; + let result = attempt.cleanup().await; + if result.is_err() { + // Reap this owned worker even when checking faulty behavior. + attempt.cleanup().await.expect("retry owned test cleanup"); + } + result.expect("natural EOF must not fail cancellation cleanup"); + assert!(!attempt.directory.exists()); + } + } + + #[tokio::test] + async fn completed_worker_without_descendants_keeps_successful_exit_status() { + let temp = tempfile::tempdir().unwrap(); + let cache = super::super::image_cache_root_dir(temp.path()); + let launcher = temp.path().join("worker"); + fs::write(&launcher, "#!/bin/sh\nexit 0\n").unwrap(); + fs::set_permissions(&launcher, fs::Permissions::from_mode(0o700)).unwrap(); + let mut attempt = Attempt::create(&cache).unwrap(); + let directory = attempt.directory.clone(); + let _stdout = attempt.spawn(&launcher, request(temp.path())).unwrap(); + wait_for_worker_exit_without_reaping(&attempt).await; + + // macOS rejects killpg for a group containing only a zombie. The + // completed worker must still return its original successful status. + assert!( + attempt + .wait() + .await + .expect("completed worker status") + .success() + ); + attempt.cleanup().await.expect("reclaim completed attempt"); + assert!(!directory.exists()); + assert!(attempt.child.as_ref().unwrap().id().is_none()); + } + + #[tokio::test] + async fn exited_worker_with_live_descendant_stops_writer_before_cleanup() { + let temp = tempfile::tempdir().unwrap(); + let cache = super::super::image_cache_root_dir(temp.path()); + let launcher = temp.path().join("worker"); + fs::write(&launcher, "#!/bin/sh\nsleep 300 &\nexit 0\n").unwrap(); + fs::set_permissions(&launcher, fs::Permissions::from_mode(0o700)).unwrap(); + let mut attempt = Attempt::create(&cache).unwrap(); + let directory = attempt.directory.clone(); + let _stdout = attempt.spawn(&launcher, request(temp.path())).unwrap(); + wait_for_worker_exit_without_reaping(&attempt).await; + + #[cfg(target_os = "macos")] + { + let id = attempt.child.as_ref().unwrap().id().unwrap(); + assert!( + !attempt.exited_worker_has_no_descendant_owner(id).unwrap(), + "an exited leader cannot prove its live descendant stopped" + ); + assert!(directory.exists()); + } + attempt + .cleanup() + .await + .expect("kill the live descendant before reclaiming its staging"); + assert!(!directory.exists()); + assert!(attempt.child.as_ref().unwrap().id().is_none()); + } + + #[tokio::test] + async fn cleanup_kills_and_reaps_worker_and_formatter_before_removing_staging() { + let temp = tempfile::tempdir().unwrap(); + let cache = super::super::image_cache_root_dir(temp.path()); + let launcher = temp.path().join("worker"); + // The shell and the simulated formatter both inherit the attempt lease. + // Killing only the shell leaves the lease locked and makes cleanup fail. + fs::write(&launcher, "#!/bin/sh\nroot=$(dirname \"$2\")\nsleep 300 &\nprintf '%s' \"$!\" > \"$root/ready\"\nwait\n").unwrap(); + fs::set_permissions(&launcher, fs::Permissions::from_mode(0o700)).unwrap(); + let mut attempt = Attempt::create(&cache).unwrap(); + let directory = attempt.directory.clone(); + let _stdout = attempt.spawn(&launcher, request(temp.path())).unwrap(); + wait_for_file(&directory.join("ready")).await; + #[cfg(target_os = "macos")] + { + let id = attempt.child.as_ref().unwrap().id().unwrap(); + assert!(!attempt.exited_worker_has_no_descendant_owner(id).unwrap()); + assert!(attempt.lease.is_some(), "a live worker retains its lease"); + } + reconcile(&cache).unwrap(); + assert!(directory.exists(), "a live formatter protects its staging"); + attempt + .cleanup() + .await + .expect("stop and reclaim preparation"); + assert!(!directory.exists()); + assert!(!attempt.lease_path.exists()); + assert!( + attempt + .child + .as_mut() + .unwrap() + .try_wait() + .unwrap() + .is_some() + ); + attempt.cleanup().await.expect("cleanup is idempotent"); + } + + #[tokio::test] + async fn aborting_provisioning_still_runs_cleanup() { + let temp = tempfile::tempdir().unwrap(); + let cache = super::super::image_cache_root_dir(temp.path()); + let attempt = Arc::new(Mutex::new(Attempt::create(&cache).unwrap())); + let directory = attempt.lock().await.directory.clone(); + fs::write(directory.join("partial.ext4"), b"unfinished image").unwrap(); + let (ready, started) = tokio::sync::oneshot::channel(); + let task = tokio::spawn({ + let attempt = attempt.clone(); + async move { + let _cleanup = CleanupOnDrop(attempt); + ready.send(()).unwrap(); + std::future::pending::<()>().await; + } + }); + started.await.unwrap(); + task.abort(); + assert!(task.await.unwrap_err().is_cancelled()); + tokio::time::timeout(Duration::from_secs(5), async { + while directory.exists() { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .expect("cancelled future must reclaim its staging"); + } + + #[test] + fn restart_reclaims_inactive_attempts_and_preserves_active_shared_and_unknown_data() { + let temp = tempfile::tempdir().unwrap(); + let cache = super::super::image_cache_root_dir(temp.path()); + let inactive = Attempt::create(&cache).unwrap(); + let inactive_directory = inactive.directory.clone(); + fs::write(inactive_directory.join("partial.ext4"), b"unfinished").unwrap(); + drop(inactive); + let active = Attempt::create(&cache).unwrap(); + let committed = cache.join("shared-image"); + fs::create_dir(&committed).unwrap(); + fs::write(committed.join("rootfs.ext4"), b"valid shared image").unwrap(); + let legacy = cache.join("image.staging-legacy"); + fs::create_dir(&legacy).unwrap(); + let unmarked = cache.join(ATTEMPTS_DIR).join(format!("attempt-{:032x}", 1)); + fs::create_dir(&unmarked).unwrap(); + let link = cache.join(ATTEMPTS_DIR).join(format!("attempt-{:032x}", 2)); + symlink(&committed, &link).unwrap(); + fs::write(unmarked.with_extension("lease"), b"unrelated lock").unwrap(); + + reconcile(&cache).unwrap(); + + assert!(!inactive_directory.exists()); + assert!(active.directory.exists()); + assert_eq!( + fs::read(committed.join("rootfs.ext4")).unwrap(), + b"valid shared image" + ); + assert!(legacy.exists()); + assert!(unmarked.exists()); + assert!(link.is_symlink()); + } + + #[tokio::test] + async fn separate_workers_serialize_cache_publication() { + let temp = tempfile::tempdir().unwrap(); + let cache = temp.path().to_path_buf(); + let first = cache_lock(&cache).unwrap(); + let (entered, mut receiver) = tokio::sync::mpsc::channel(1); + let second = tokio::task::spawn_blocking(move || { + let _second = cache_lock(&cache).unwrap(); + entered.blocking_send(()).unwrap(); + }); + assert!( + tokio::time::timeout(Duration::from_millis(50), receiver.recv()) + .await + .is_err() + ); + drop(first); + tokio::time::timeout(Duration::from_secs(5), receiver.recv()) + .await + .unwrap() + .unwrap(); + second.await.unwrap(); + } + + #[test] + fn cache_lock_child_process_helper() { + let Some(root) = std::env::var_os("OPENSHELL_TEST_PUBLICATION_LOCK_ROOT") else { + return; + }; + let root = PathBuf::from(root); + let name = std::env::var("OPENSHELL_TEST_PUBLICATION_LOCK_NAME").unwrap(); + fs::write(root.join(format!("{name}.waiting")), b"waiting").unwrap(); + if name == "second" { + let probe = OpenOptions::new() + .read(true) + .write(true) + .open(root.join("preparation.lock")) + .unwrap(); + assert!( + !lock(&probe, true).unwrap(), + "first process must own the kernel lock" + ); + fs::write(root.join("second.contended"), b"contended").unwrap(); + } + if name == "second" { + publish_cache_file( + &root, + &root.join("second.staged"), + &root.join("committed"), + Some(5), + ) + .unwrap(); + fs::write(root.join("second.acquired"), b"published").unwrap(); + return; + } + let _lock = cache_lock(&root).unwrap(); + fs::write(root.join(format!("{name}.acquired")), b"acquired").unwrap(); + let mut release = [0_u8; 1]; + io::stdin().read_exact(&mut release).unwrap(); + } + + struct LockHelper(std::process::Child); + + impl Drop for LockHelper { + fn drop(&mut self) { + let _ = self.0.kill(); + let _ = self.0.wait(); + } + } + + #[test] + fn publication_lock_is_exclusive_across_processes_and_released_on_death() { + let root = tempfile::tempdir().unwrap(); + let spawn = |name: &str| { + LockHelper( + std::process::Command::new(std::env::current_exe().unwrap()) + .args([ + "--exact", + "driver::preparation::tests::cache_lock_child_process_helper", + "--nocapture", + ]) + .env("OPENSHELL_TEST_PUBLICATION_LOCK_ROOT", root.path()) + .env("OPENSHELL_TEST_PUBLICATION_LOCK_NAME", name) + .stdin(Stdio::piped()) + .stdout(Stdio::null()) + .stderr(Stdio::inherit()) + .spawn() + .unwrap(), + ) + }; + let wait_for = |name: &str| { + let deadline = std::time::Instant::now() + Duration::from_secs(5); + while !root.path().join(name).exists() { + assert!( + std::time::Instant::now() < deadline, + "helper did not reach {name}" + ); + std::thread::sleep(Duration::from_millis(10)); + } + }; + fs::write(root.path().join("second.staged"), b"later").unwrap(); + let mut first = spawn("first"); + wait_for("first.acquired"); + let mut second = spawn("second"); + wait_for("second.contended"); + assert!( + !root.path().join("second.acquired").exists(), + "second process bypassed lock" + ); + // The lock holder commits before dying. The queued publisher must read + // readiness only after it owns the lock and preserve this complete image. + fs::write(root.path().join("committed"), b"first").unwrap(); + first.0.kill().unwrap(); + first.0.wait().unwrap(); + wait_for("second.acquired"); + assert!(second.0.wait().unwrap().success()); + assert_eq!(fs::read(root.path().join("committed")).unwrap(), b"first"); + assert!(root.path().join("second.staged").exists()); + assert!(root.path().join("preparation.lock").exists()); + // An invalid-sized template can be replaced; a valid rootfs cannot. + fs::write(root.path().join("second.staged"), b"new!").unwrap(); + publish_cache_file( + root.path(), + &root.path().join("second.staged"), + &root.path().join("committed"), + Some(4), + ) + .unwrap(); + assert_eq!(fs::read(root.path().join("committed")).unwrap(), b"new!"); + fs::write(root.path().join("third.staged"), b"third").unwrap(); + publish_cache_file( + root.path(), + &root.path().join("third.staged"), + &root.path().join("committed"), + None, + ) + .unwrap(); + assert_eq!(fs::read(root.path().join("committed")).unwrap(), b"new!"); + } + + #[test] + fn worker_rejects_request_without_inherited_lease() { + let temp = tempfile::tempdir().unwrap(); + let cache = super::super::image_cache_root_dir(temp.path()); + let attempt = Attempt::create(&cache).unwrap(); + let path = attempt.directory.join(REQUEST_FILE); + fs::write(&path, serde_json::to_vec(&request(temp.path())).unwrap()).unwrap(); + let error = read_request(&path) + .err() + .expect("missing inherited fd rejected"); + assert!(error.contains("lease was not inherited")); + } +} diff --git a/crates/openshell-driver-vm/tests/image_preparation.rs b/crates/openshell-driver-vm/tests/image_preparation.rs new file mode 100644 index 0000000000..ffa0992c9e --- /dev/null +++ b/crates/openshell-driver-vm/tests/image_preparation.rs @@ -0,0 +1,383 @@ +// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +#![cfg(feature = "compute-driver")] +#![allow(unsafe_code)] + +use std::fs::{self, File}; +use std::os::fd::AsRawFd; +use std::os::unix::fs::PermissionsExt; +use std::os::unix::process::CommandExt; +use std::process::{Command, Stdio}; + +use openshell_driver_vm::VmDriverConfig; + +/// Exercise the actual worker entry point and a real sparse-file copy without +/// downloading an image, running a VM, or requiring host filesystem utilities. +#[test] +fn worker_publishes_complete_overlay_and_reports_resolved_owner() { + worker_overlay_case(false); +} + +/// Retrying an interrupted first start must never turn an incomplete copy +/// into persisted overlay state. The next retry must still be able to start. +#[test] +fn interrupted_missing_overlay_retry_does_not_publish_partial_state() { + worker_overlay_case(true); +} + +fn worker_overlay_case(interrupt_retry: bool) { + let root = tempfile::tempdir().unwrap(); + let cache = root.path().join("images"); + let attempts = cache.join("preparations"); + let name = "attempt-00000000000000000000000000000001"; + let attempt = attempts.join(name); + fs::create_dir_all(&attempt).unwrap(); + let lease_path = attempts.join(format!("{name}.lease")); + fs::write(&lease_path, b"openshell-image-preparation-v1\n").unwrap(); + let lease = File::open(&lease_path).unwrap(); + let fd = lease.as_raw_fd(); + // SAFETY: this descriptor belongs to the live fixture file. + assert_eq!(unsafe { libc::flock(fd, libc::LOCK_EX) }, 0); + + let size = 1024 * 1024; + let template = cache.join("overlay-templates/sandbox-overlay-ext4-v1/1048576.ext4"); + fs::create_dir_all(template.parent().unwrap()).unwrap(); + let mut expected = vec![0_u8; size]; + expected[..13].copy_from_slice(b"template-data"); + fs::write(&template, &expected).unwrap(); + let sandbox = root.path().join("sandboxes/worker-test"); + fs::create_dir_all(&sandbox).unwrap(); + if interrupt_retry { + // A cancelled Fresh attempt can persist the owner before publishing + // its first overlay. Start retries this state with PreserveExisting. + fs::write( + sandbox.join("sandbox-owner-state"), + b"sandbox-owner-v2:1000:1000\n", + ) + .unwrap(); + } + let source = cache.join("test-image/rootfs.ext4"); + fs::create_dir_all(source.parent().unwrap()).unwrap(); + fs::write(&source, b"identity comes from driver configuration").unwrap(); + let config = VmDriverConfig { + state_dir: root.path().to_path_buf(), + sandbox_uid: Some(1000), + sandbox_gid: Some(1000), + overlay_disk_mib: 1, + ..VmDriverConfig::default() + }; + let request = attempt.join("request.json"); + fs::write( + &request, + serde_json::to_vec(&serde_json::json!({ + "config": config, + "sandbox_id": "worker-test", + "image_ref": "", + "rootfs_tar": null, + "bootstrap_only": false, + "overlay": { + "source_disk": source, + "preparation": if interrupt_retry { "PreserveExisting" } else { "Fresh" }, + }, + "lease_fd": fd, + "parent_pid": std::process::id(), + })) + .unwrap(), + ) + .unwrap(); + let mut command = Command::new(env!("CARGO_BIN_EXE_openshell-driver-vm")); + command + .arg("--internal-prepare-image") + .arg(&request) + .stdin(Stdio::null()) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()) + .process_group(0); + // SAFETY: only fcntl runs between fork and exec, and the fixture keeps its + // descriptor alive until the worker has exited. + unsafe { + command.pre_exec(move || { + if libc::fcntl(fd, libc::F_SETFD, 0) == -1 { + return Err(std::io::Error::last_os_error()); + } + Ok(()) + }); + } + if interrupt_retry { + let tools = root.path().join("tools"); + fs::create_dir(&tools).unwrap(); + let copy_ready = root.path().join("copy-ready"); + let cp = tools.join("cp"); + // Control only the external copy command. The actual worker selects + // the destination and executes its production preservation checks. + fs::write( + &cp, + "#!/bin/sh\nfor argument in \"$@\"; do destination=\"$argument\"; done\nprintf partial > \"$destination\"\nprintf '%s' \"$destination\" > \"$COPY_READY\"\nsleep 300\n", + ) + .unwrap(); + fs::set_permissions(&cp, fs::Permissions::from_mode(0o700)).unwrap(); + command + .env("PATH", format!("{}:/usr/bin:/bin", tools.display())) + .env("COPY_READY", ©_ready); + let mut interrupted = command.spawn().unwrap(); + let deadline = std::time::Instant::now() + std::time::Duration::from_secs(10); + while !copy_ready.exists() && std::time::Instant::now() < deadline { + if interrupted.try_wait().unwrap().is_some() { + break; + } + std::thread::sleep(std::time::Duration::from_millis(10)); + } + // Kill and reap even if readiness fails, so failed tests do not leave + // a blocked helper or worker behind. + if interrupted.try_wait().unwrap().is_none() { + let pid = i32::try_from(interrupted.id()).unwrap(); + let _ = nix::sys::signal::killpg( + nix::unistd::Pid::from_raw(pid), + nix::sys::signal::Signal::SIGKILL, + ); + } + interrupted.wait().unwrap(); + assert!(copy_ready.exists(), "worker must reach the controlled copy"); + assert!( + !sandbox.join("overlay.ext4").exists(), + "interrupted retry must leave no partial persisted overlay" + ); + let destination = fs::read_to_string(©_ready).unwrap(); + assert!(std::path::Path::new(&destination).starts_with(&attempt)); + // Retry using the real copy utility. The incomplete attempt image + // must not prevent publishing the complete overlay. + command + .env("PATH", "/usr/bin:/bin") + .env_remove("COPY_READY"); + } + let mut child = command.spawn().unwrap(); + let deadline = std::time::Instant::now() + std::time::Duration::from_secs(10); + while child.try_wait().unwrap().is_none() { + if std::time::Instant::now() >= deadline { + let pid = i32::try_from(child.id()).unwrap(); + let _ = nix::sys::signal::killpg( + nix::unistd::Pid::from_raw(pid), + nix::sys::signal::Signal::SIGKILL, + ); + let _ = child.wait(); + panic!("image preparation worker timed out"); + } + std::thread::sleep(std::time::Duration::from_millis(10)); + } + let output = child.wait_with_output().unwrap(); + assert!( + output.status.success(), + "{}", + String::from_utf8_lossy(&output.stderr) + ); + let messages = String::from_utf8(output.stdout).unwrap(); + let completion = messages + .lines() + .map(|line| { + serde_json::from_str::(line) + .expect("worker stdout must contain only protocol messages") + }) + .find_map(|message| message.get("Complete").cloned()) + .expect("worker completion"); + assert_eq!( + completion, + serde_json::json!({"Ok": {"Overlay": {"uid": 1000, "gid": 1000}}}) + ); + assert_eq!(fs::read(sandbox.join("overlay.ext4")).unwrap(), expected); + assert_eq!( + fs::read(&template).unwrap(), + expected, + "the shared template is immutable" + ); + assert_eq!( + fs::read_to_string(sandbox.join("sandbox-owner-state")).unwrap(), + "sandbox-owner-v2:1000:1000\n" + ); +} + +struct OwnedPreparationChild { + child: std::process::Child, + reaped: bool, + _lease: File, +} + +impl OwnedPreparationChild { + fn exited(&self) -> std::io::Result { + // SAFETY: initialized output storage; WNOWAIT retains our child's PID. + let mut status: libc::siginfo_t = unsafe { std::mem::zeroed() }; + if unsafe { + libc::waitid( + libc::P_PID, + self.child.id(), + &raw mut status, + libc::WEXITED | libc::WNOHANG | libc::WNOWAIT, + ) + } != 0 + { + return Err(std::io::Error::last_os_error()); + } + Ok(matches!( + status.si_code, + libc::CLD_EXITED | libc::CLD_KILLED | libc::CLD_DUMPED + )) + } + + fn stop_and_reap(&mut self) -> std::io::Result { + let pid = i32::try_from(self.child.id()).map_err(std::io::Error::other)?; + let _ = nix::sys::signal::killpg( + nix::unistd::Pid::from_raw(pid), + nix::sys::signal::Signal::SIGKILL, + ); + let status = self.child.wait()?; + self.reaped = true; + Ok(status) + } +} + +impl Drop for OwnedPreparationChild { + fn drop(&mut self) { + if !self.reaped { + let _ = self.stop_and_reap(); + } + } +} + +fn spawn_overlay_fixture( + root: &std::path::Path, + id: &str, + number: u128, + controlled_copy: bool, +) -> OwnedPreparationChild { + let attempts = root.join("images/preparations"); + let name = format!("attempt-{number:032x}"); + let attempt = attempts.join(&name); + fs::create_dir_all(&attempt).unwrap(); + let lease_path = attempts.join(format!("{name}.lease")); + fs::write(&lease_path, b"openshell-image-preparation-v1\n").unwrap(); + let lease = File::open(&lease_path).unwrap(); + let fd = lease.as_raw_fd(); + // SAFETY: lease owns fd and stays alive in the returned process guard. + assert_eq!(unsafe { libc::flock(fd, libc::LOCK_EX) }, 0); + fs::create_dir_all(root.join("sandboxes").join(id)).unwrap(); + let config = VmDriverConfig { + state_dir: root.to_path_buf(), + sandbox_uid: Some(1000), + sandbox_gid: Some(1000), + overlay_disk_mib: 1, + ..VmDriverConfig::default() + }; + let request = attempt.join("request.json"); + fs::write( + &request, + serde_json::to_vec(&serde_json::json!({ + "config": config, "sandbox_id": id, "image_ref": "", "rootfs_tar": null, + "bootstrap_only": false, + "overlay": { "source_disk": root.join("images/test-image/rootfs.ext4"), + "preparation": "Fresh", "requested_identity": null }, + "lease_fd": fd, "parent_pid": std::process::id(), + })) + .unwrap(), + ) + .unwrap(); + let mut command = Command::new(env!("CARGO_BIN_EXE_openshell-driver-vm")); + command + .arg("--internal-prepare-image") + .arg(request) + .stdin(Stdio::null()) + .stdout(File::create(root.join(format!("{id}.stdout"))).unwrap()) + .stderr(File::create(root.join(format!("{id}.stderr"))).unwrap()) + .process_group(0); + if controlled_copy { + command + .env( + "PATH", + format!("{}:/usr/bin:/bin", root.join("tools").display()), + ) + .env("COPY_READY", root.join("copy-ready")); + } else { + command.env("PATH", "/usr/bin:/bin"); + } + // SAFETY: only fcntl runs between fork and exec; the guard keeps fd alive. + unsafe { + command.pre_exec(move || { + if libc::fcntl(fd, libc::F_SETFD, 0) == -1 { + return Err(std::io::Error::last_os_error()); + } + Ok(()) + }); + } + OwnedPreparationChild { + child: command.spawn().unwrap(), + reaped: false, + _lease: lease, + } +} + +#[test] +fn independent_overlay_finishes_while_another_copy_is_blocked() { + let root = tempfile::tempdir().unwrap(); + let template = root + .path() + .join("images/overlay-templates/sandbox-overlay-ext4-v1/1048576.ext4"); + fs::create_dir_all(template.parent().unwrap()).unwrap(); + let mut expected = vec![0_u8; 1024 * 1024]; + expected[..13].copy_from_slice(b"template-data"); + fs::write(&template, &expected).unwrap(); + let source = root.path().join("images/test-image/rootfs.ext4"); + fs::create_dir_all(source.parent().unwrap()).unwrap(); + fs::write(&source, b"configured owner fixture").unwrap(); + let tools = root.path().join("tools"); + fs::create_dir(&tools).unwrap(); + fs::write(tools.join("cp"), + "#!/bin/sh\nfor value in \"$@\"; do destination=\"$value\"; done\nprintf partial > \"$destination\"\n: > \"$COPY_READY\"\nexec sleep 300\n" + ).unwrap(); + fs::set_permissions(tools.join("cp"), fs::Permissions::from_mode(0o700)).unwrap(); + let mut slow = spawn_overlay_fixture(root.path(), "slow-a", 1, true); + let deadline = std::time::Instant::now() + std::time::Duration::from_secs(10); + while !root.path().join("copy-ready").exists() { + assert!( + !slow.exited().unwrap(), + "slow worker exited before controlled copy" + ); + assert!( + std::time::Instant::now() < deadline, + "controlled copy readiness timed out" + ); + std::thread::sleep(std::time::Duration::from_millis(10)); + } + let mut warm = spawn_overlay_fixture(root.path(), "warm-b", 2, false); + let deadline = std::time::Instant::now() + std::time::Duration::from_secs(10); + while !warm.exited().unwrap() { + assert!( + std::time::Instant::now() < deadline, + "warm overlay waited for unrelated copy" + ); + std::thread::sleep(std::time::Duration::from_millis(10)); + } + let status = warm.stop_and_reap().unwrap(); + assert!( + status.success(), + "{}", + fs::read_to_string(root.path().join("warm-b.stderr")).unwrap() + ); + assert!( + !slow.exited().unwrap(), + "slow copy must remain blocked during the proof" + ); + assert_eq!( + fs::read(root.path().join("sandboxes/warm-b/overlay.ext4")).unwrap(), + expected + ); + assert_eq!(fs::read(&template).unwrap(), expected); + assert!(!root.path().join("sandboxes/slow-a/overlay.ext4").exists()); + let output = fs::read_to_string(root.path().join("warm-b.stdout")).unwrap(); + let completed = output.lines().any(|line| { + serde_json::from_str::(line) + .unwrap() + .get("Complete") + == Some(&serde_json::json!({"Ok": {"Overlay": {"uid": 1000, "gid": 1000}}})) + }); + assert!(completed, "actual worker did not report overlay completion"); + slow.stop_and_reap().unwrap(); +} From 1c123a454267f98783f92dfd2aa3814ba6d85c16 Mon Sep 17 00:00:00 2001 From: Shiju Date: Sat, 3 Oct 2026 20:20:02 +0000 Subject: [PATCH 09/13] feat(gateway): check VM host tools during config preflight (#4037) * feat(gateway): validate VM filesystem tools during preflight Check required local VM tools through config preflight and share executable resolution with VM image operations. Report selected paths and actionable errors without creating gateway or sandbox state. Bound probe output and execution time, and clean up probe descendants on interruption. Preserve pure static validation and skip local tool checks for remote driver endpoints and unrelated drivers. Fixes #3951 Related to #3955 Signed-off-by: Shiju * fix(gateway): stabilize filesystem preflight checks Combine identical filesystem-tool error arms and normalize rendered diagnostics in command tests so terminal wrapping preserves assertions. Describe driver TLS validation without depending on removed guest fields. Signed-off-by: Shiju * test(gateway): serialize preflight fixture paths as TOML Keep temporary paths quoted and escaped through the TOML serializer instead of relying on Rust Debug formatting. Signed-off-by: Shiju --------- Signed-off-by: Shiju --- crates/openshell-core/src/e2fsprogs.rs | 598 ++++++++++++++++++ crates/openshell-core/src/lib.rs | 2 + crates/openshell-driver-vm/src/rootfs.rs | 512 +++++++-------- crates/openshell-gateway/src/lib.rs | 9 + .../tests/config_preflight.rs | 276 ++++++++ crates/openshell-server/src/cli.rs | 76 ++- crates/openshell-server/src/lib.rs | 15 + deploy/man/openshell-gateway.8.md | 6 +- docs/how-it-works/gateways/configuration.mdx | 40 +- skills/debug-openshell-cluster/SKILL.md | 3 +- 10 files changed, 1211 insertions(+), 326 deletions(-) create mode 100644 crates/openshell-core/src/e2fsprogs.rs create mode 100644 crates/openshell-gateway/tests/config_preflight.rs diff --git a/crates/openshell-core/src/e2fsprogs.rs b/crates/openshell-core/src/e2fsprogs.rs new file mode 100644 index 0000000000..d6f4c5ddc4 --- /dev/null +++ b/crates/openshell-core/src/e2fsprogs.rs @@ -0,0 +1,598 @@ +// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +//! Host filesystem tools shared by VM image operations and gateway preflight. +//! +//! The gateway launches the VM driver with its environment unchanged. Resolve +//! tools here so both binaries select the same installation without linking the +//! driver runtime into the gateway or starting it during preflight. + +use std::ffi::OsStr; +use std::fs; +use std::os::unix::fs::PermissionsExt as _; +use std::path::{Path, PathBuf}; +use std::process::Command; +use std::time::Duration; +use tokio::sync::watch; + +const PROBE_TIMEOUT: Duration = Duration::from_secs(5); +const INSTALL_GUIDANCE: &str = "Install e2fsprogs 1.43 or newer, or repair the selected installation and the gateway service PATH; rerun config preflight with the service account and environment"; + +/// Resolve a tool using the VM driver's inherited PATH and existing package prefixes. +/// +/// Names are ordered alternatives: the formatter tries `mke2fs` before +/// `mkfs.ext4`. A present but broken installation returns an error and stops +/// lookup. The returned absolute path is also used for +/// image operations; the caller's environment is not changed. +pub fn resolve(names: &[&str]) -> Result { + resolve_in(names, &search_dirs(std::env::var_os("PATH").as_deref())) +} + +fn search_dirs(path: Option<&OsStr>) -> Vec { + let mut dirs: Vec<_> = path + .map(std::env::split_paths) + .into_iter() + .flatten() + .collect(); + // Preserve the VM driver's existing package-prefix fallbacks. Additional + // installations, including Linux sbin directories, belong on service PATH. + for root in ["/opt/homebrew/opt/e2fsprogs", "/usr/local/opt/e2fsprogs"] { + dirs.push(Path::new(root).join("sbin")); + dirs.push(Path::new(root).join("bin")); + } + dirs +} + +fn resolve_in(names: &[&str], dirs: &[PathBuf]) -> Result { + for name in names { + for directory in dirs { + let candidate = directory.join(name); + let metadata = match fs::metadata(&candidate) { + Ok(metadata) => metadata, + Err(error) if error.kind() == std::io::ErrorKind::NotFound => continue, + Err(error) => { + return Err(format!( + "inspect {}: {error}. {INSTALL_GUIDANCE}", + candidate.display() + )); + } + }; + if !metadata.is_file() || metadata.permissions().mode() & 0o111 == 0 { + return Err(format!( + "{} is not an executable file. {INSTALL_GUIDANCE}", + candidate.display() + )); + } + // Canonicalization makes relative PATH entries unambiguous in the + // report and ensures execution cannot redo PATH lookup differently. + return candidate.canonicalize().map_err(|error| { + format!( + "resolve {}: {error}. {INSTALL_GUIDANCE}", + candidate.display() + ) + }); + } + } + Err(format!( + "{} not found in the VM driver's search directories: {}. {INSTALL_GUIDANCE}", + names.join(" or "), + dirs.iter() + .map(|path| path.display().to_string()) + .collect::>() + .join(":") + )) +} + +/// Check a formatter, `debugfs`, and `e2fsck` without creating images or driver state. +/// +/// Each selected executable receives only `-V`, with null stdin, a five-second +/// deadline, and at most 8 KiB captured per stream. Errors retain the selected +/// path and the tool's diagnostics. Image health requires a separate check. +pub async fn preflight(cancellation: watch::Receiver) -> Result, String> { + preflight_in_with_cancellation( + &search_dirs(std::env::var_os("PATH").as_deref()), + cancellation, + ) + .await +} + +#[cfg(test)] +async fn preflight_in(dirs: &[PathBuf]) -> Result, String> { + let (_sender, cancellation) = watch::channel(false); + preflight_in_with_cancellation(dirs, cancellation).await +} + +async fn preflight_in_with_cancellation( + dirs: &[PathBuf], + mut cancellation: watch::Receiver, +) -> Result, String> { + let mut reports = Vec::new(); + for (names, identity) in [ + (&["mke2fs", "mkfs.ext4"][..], "mke2fs"), + (&["debugfs"][..], "debugfs"), + (&["e2fsck"][..], "e2fsck"), + ] { + let result = async { + let path = resolve_in(names, dirs)?; + let label = path.display(); + let output = run_version_probe(Command::new(&path), &mut cancellation) + .await + .map_err(|error| format!("run {label} -V: {error}. {INSTALL_GUIDANCE}"))?; + let stdout = String::from_utf8_lossy(&output.stdout); + let stderr = String::from_utf8_lossy(&output.stderr); + if !output.status.success() { + return Err(format!( + "{label} -V failed with status {}\nstdout: {stdout}\nstderr: {stderr}\n{INSTALL_GUIDANCE}", + output.status + )); + } + let version = supported_version(identity, &stdout) + .or_else(|| supported_version(identity, &stderr)) + .ok_or_else(|| format!( + "{label} is not a supported {identity} executable\nstdout: {stdout}\nstderr: {stderr}\n{INSTALL_GUIDANCE}" + ))?; + Ok(format!("VM host tool {identity}: {label} ({version})")) + }.await; + match result { + Ok(report) => reports.push(report), + Err(error) => { + // Retain earlier selected paths even when a later tool fails. + reports.push(error); + return Err(reports.join("\n")); + } + } + } + Ok(reports) +} + +// Keep the group leader unreaped until group cleanup so its PID cannot be +// reused while a descendant still holds a captured output pipe open. +struct ProbeProcess { + child: tokio::process::Child, + group: Option, +} + +impl ProbeProcess { + fn stop_group(&mut self) -> Result, String> { + let Some(group) = self.group.take() else { + return Ok(None); + }; + match nix::sys::signal::killpg(group, nix::sys::signal::Signal::SIGKILL) { + Ok(()) | Err(nix::errno::Errno::ESRCH) => Ok(None), + // Darwin can return EPERM for a group containing only our exited, + // unreaped leader. Defer judgment until after its owned reap. + #[cfg(target_os = "macos")] + Err(nix::errno::Errno::EPERM) => Ok(Some(group)), + Err(error) => Err(format!("terminate version probe process group: {error}")), + } + } +} + +impl Drop for ProbeProcess { + fn drop(&mut self) { + let _ = self.stop_group(); + // kill_on_drop schedules the direct child for reaping if the caller + // aborts its future. CLI signal cancellation instead awaits cleanup. + } +} + +async fn cancelled(cancellation: &mut watch::Receiver) { + loop { + if *cancellation.borrow_and_update() { + return; + } + if cancellation.changed().await.is_err() { + // A closed sender does not request cancellation. + std::future::pending::<()>().await; + } + } +} + +async fn observe_exit(group: nix::unistd::Pid) -> Result<(), String> { + use rustix::process::{Pid, WaitId, WaitIdOptions, waitid}; + + let pid = Pid::from_raw(group.as_raw()).ok_or("invalid version probe process ID")?; + loop { + let exited = waitid( + WaitId::Pid(pid), + WaitIdOptions::EXITED | WaitIdOptions::NOHANG | WaitIdOptions::NOWAIT, + ) + .map(|status| status.is_some()); + match exited { + Ok(false) | Err(rustix::io::Errno::INTR) => { + tokio::time::sleep(Duration::from_millis(10)).await; + } + Ok(true) => return Ok(()), + Err(error) => return Err(format!("observe version probe exit: {error}")), + } + } +} + +async fn run_version_probe( + command: Command, + cancellation: &mut watch::Receiver, +) -> Result { + use std::process::Stdio; + + if *cancellation.borrow() { + return Err("host tool checks cancelled".to_string()); + } + let child = tokio::process::Command::from(command) + .arg("-V") + .stdin(Stdio::null()) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()) + .process_group(0) + .kill_on_drop(true) + .spawn() + .map_err(|error| error.to_string())?; + let group = child + .id() + .and_then(|id| i32::try_from(id).ok()) + .map(nix::unistd::Pid::from_raw) + .ok_or_else(|| "version probe has no valid process ID".to_string())?; + let mut probe = ProbeProcess { + child, + group: Some(group), + }; + let stdout = probe + .child + .stdout + .take() + .ok_or("version probe stdout is unavailable")?; + let stderr = probe + .child + .stderr + .take() + .ok_or("version probe stderr is unavailable")?; + let output = tokio::select! { + biased; + () = cancelled(cancellation) => Err("host tool checks cancelled".to_string()), + result = tokio::time::timeout(PROBE_TIMEOUT, async { + let (stdout, stderr, ()) = tokio::try_join!( + read_probe_output(stdout, "stdout"), + read_probe_output(stderr, "stderr"), + observe_exit(group), + )?; + Ok((stdout, stderr)) + }) => result.unwrap_or_else(|_| Err("timed out after 5 seconds".to_string())), + }; + // Signal the group before reaping its leader, then retire the stored group + // ID before awaiting anything else. No later drop can signal a reused PID. + let group_cleanup = probe.stop_group(); + let status = tokio::time::timeout(Duration::from_secs(1), probe.child.wait()) + .await + .map_err(|_| "version probe cleanup exceeded one second".to_string()) + .and_then(|result| result.map_err(|error| format!("reap version probe: {error}"))); + let status = match (group_cleanup, status) { + (Ok(Some(group)), Ok(status)) => { + // Signal 0 only queries existence; it never signals a process. + // ESRCH after the owned reap proves no group members remain. If + // the ID was reused, this check can only fail conservatively. + match nix::sys::signal::killpg(group, None) { + Err(nix::errno::Errno::ESRCH) => Ok(status), + result => Err(format!( + "version probe process group remains after cleanup: {result:?}" + )), + } + } + (Ok(None), status) | (Err(_), status @ Err(_)) => status, + (Err(error), Ok(_)) | (Ok(Some(_)), Err(error)) => Err(error), + }; + match (output, status) { + (Ok((stdout, stderr)), Ok(status)) => Ok(std::process::Output { + status, + stdout, + stderr, + }), + (Err(error), Ok(_)) | (Ok(_), Err(error)) => Err(error), + (Err(error), Err(cleanup)) => Err(format!("{error}; {cleanup}")), + } +} + +async fn read_probe_output( + reader: impl tokio::io::AsyncRead + Unpin, + stream: &str, +) -> Result, String> { + use tokio::io::AsyncReadExt as _; + // Read one extra byte to distinguish complete output from overflow. Stop + // on overflow instead of draining an unbounded writer until the deadline. + const MAX_BYTES: usize = 8 * 1024; + let mut output = Vec::new(); + reader + .take((MAX_BYTES + 1) as u64) + .read_to_end(&mut output) + .await + .map_err(|error| format!("read version probe {stream}: {error}"))?; + if output.len() > MAX_BYTES { + output.truncate(MAX_BYTES); + return Err(format!( + "version probe {stream} exceeded {MAX_BYTES} bytes: {} [truncated]", + String::from_utf8_lossy(&output) + )); + } + Ok(output) +} + +fn supported_version<'a>(identity: &str, output: &'a str) -> Option<&'a str> { + output.lines().find_map(|line| { + let mut words = line.split_whitespace(); + if words.next() != Some(identity) { + return None; + } + let version = words.next()?; + let mut parts = version.split('.'); + let major = parts.next()?.parse::().ok()?; + let minor = parts.next()?.parse::().ok()?; + // Reject malformed version tokens rather than accepting an arbitrary + // suffix after an otherwise valid major/minor pair. + if let Some(patch) = parts.next() + && (patch.parse::().is_err() || parts.next().is_some()) + { + return None; + } + // VM image preparation uses mke2fs -d and ext4 filesystem features. + // Keep all host tools on the supported e2fsprogs release baseline. + ((major, minor) >= (1, 43)).then_some(version) + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn fake_e2fs_tool(directory: &Path, name: &str, body: &str) { + let path = directory.join(name); + fs::write(&path, format!("#!/bin/sh\n{body}\n")).expect("write tool"); + fs::set_permissions(path, fs::Permissions::from_mode(0o755)).expect("make executable"); + } + + fn fake_e2fs_installation(directory: &Path) { + for name in ["mke2fs", "debugfs", "e2fsck"] { + fake_e2fs_tool( + directory, + name, + &format!("test \"$1\" = -V || exit 64\necho '{name} 1.47.4' >&2"), + ); + } + } + + #[tokio::test] + async fn filesystem_preflight_accepts_private_prefix_and_formatter_alias() { + let temp = tempfile::tempdir().expect("private prefix"); + fake_e2fs_installation(temp.path()); + fs::rename(temp.path().join("mke2fs"), temp.path().join("mkfs.ext4")) + .expect("use formatter alias"); + preflight_in(&[temp.path().to_path_buf()]) + .await + .expect("all required tools available"); + let selected = resolve_in(&["debugfs"], &[temp.path().to_path_buf()]) + .expect("execution uses same resolver"); + let expected = temp + .path() + .join("debugfs") + .canonicalize() + .expect("tool path"); + assert_eq!(selected, expected); + } + + #[tokio::test] + async fn filesystem_preflight_rejects_missing_tools() { + let temp = tempfile::tempdir().expect("clean host search path"); + let error = preflight_in(&[temp.path().to_path_buf()]) + .await + .expect_err("missing formatter must reject preparation"); + assert!(error.contains("mke2fs or mkfs.ext4 not found"), "{error}"); + assert!(error.contains("gateway service PATH"), "{error}"); + } + + #[tokio::test] + async fn filesystem_preflight_requires_debugfs_and_recovery_tool() { + for missing in ["debugfs", "e2fsck"] { + let temp = tempfile::tempdir().expect("partial installation"); + fake_e2fs_installation(temp.path()); + fs::remove_file(temp.path().join(missing)).expect("remove required tool"); + let error = preflight_in(&[temp.path().to_path_buf()]) + .await + .expect_err("missing tool"); + assert!(error.contains(&format!("{missing} not found")), "{error}"); + assert!( + error.contains("VM host tool mke2fs:"), + "earlier path lost: {error}" + ); + } + } + + #[tokio::test] + async fn filesystem_preflight_retains_missing_interpreter_error() { + let temp = tempfile::tempdir().expect("broken installation"); + let path = temp.path().join("mke2fs"); + fs::write(&path, "#!/nonexistent/e2fsprogs-interpreter\n").expect("broken executable"); + fs::set_permissions(&path, fs::Permissions::from_mode(0o755)).expect("executable bit"); + fake_e2fs_tool(temp.path(), "mkfs.ext4", "echo 'mke2fs 1.47.4' >&2"); + let error = preflight_in(&[temp.path().to_path_buf()]) + .await + .expect_err("loader error"); + assert!(error.contains("mke2fs -V:"), "{error}"); + assert!(error.contains("No such file or directory"), "{error}"); + assert!(!error.contains("mke2fs or mkfs.ext4 not found"), "{error}"); + } + + #[tokio::test] + async fn filesystem_preflight_rejects_nonexecutable_before_another_installation() { + let first = tempfile::tempdir().expect("first installation"); + let second = tempfile::tempdir().expect("second installation"); + fs::write(first.path().join("mke2fs"), b"not executable").expect("write broken tool"); + fake_e2fs_installation(second.path()); + let path = std::env::join_paths([first.path(), second.path()]).expect("search path"); + let error = preflight_in(&std::env::split_paths(&path).collect::>()) + .await + .expect_err("broken installation must not fall back"); + assert!(error.contains("is not an executable file"), "{error}"); + assert!( + error.contains(&first.path().display().to_string()), + "{error}" + ); + } + + #[tokio::test] + async fn filesystem_preflight_preserves_failed_tool_output() { + let first = tempfile::tempdir().expect("first installation"); + let second = tempfile::tempdir().expect("second installation"); + fake_e2fs_tool(first.path(), "mke2fs", "echo 'loader failed' >&2\nexit 42"); + fake_e2fs_installation(second.path()); + let path = std::env::join_paths([first.path(), second.path()]).expect("search path"); + let error = preflight_in(&std::env::split_paths(&path).collect::>()) + .await + .expect_err("real execution failure must not fall back"); + assert!(error.contains("42"), "{error}"); + assert!(error.contains("loader failed"), "{error}"); + assert!( + error.contains(&first.path().display().to_string()), + "{error}" + ); + } + + #[tokio::test] + async fn filesystem_preflight_rejects_incompatible_tool() { + let temp = tempfile::tempdir().expect("old installation"); + fake_e2fs_installation(temp.path()); + fake_e2fs_tool(temp.path(), "debugfs", "echo 'debugfs 1.42.13' >&2"); + let error = preflight_in(&[temp.path().to_path_buf()]) + .await + .expect_err("old tool must reject preparation"); + assert!( + error.contains("not a supported debugfs executable"), + "{error}" + ); + assert!(error.contains("debugfs 1.42.13"), "{error}"); + } + + #[tokio::test] + async fn filesystem_preflight_stops_a_hung_version_probe() { + let temp = tempfile::tempdir().expect("hung installation"); + fake_e2fs_tool(temp.path(), "mke2fs", "exec /bin/sleep 30"); + let error = preflight_in(&[temp.path().to_path_buf()]) + .await + .expect_err("preflight must finish even if a tool hangs"); + assert!(error.contains("timed out after 5 seconds"), "{error}"); + } + + #[tokio::test] + async fn filesystem_preflight_bounds_each_output_stream() { + let temp = tempfile::tempdir().expect("noisy installation"); + for (redirect, stream) in [("", "stdout"), (">&2", "stderr")] { + fake_e2fs_tool( + temp.path(), + "mke2fs", + &format!("/bin/dd if=/dev/zero bs=16384 count=4 {redirect} 2>/dev/null\nexit 42"), + ); + let error = preflight_in(&[temp.path().to_path_buf()]) + .await + .expect_err("oversized probe output must fail early"); + assert!(error.contains(&format!("{stream} exceeded 8192 bytes"))); + assert!( + error.len() < 9000, + "diagnostic grew to {} bytes", + error.len() + ); + } + } + + #[tokio::test] + async fn filesystem_preflight_timeout_terminates_wrapper_descendants() { + let temp = tempfile::tempdir().expect("wrapper installation"); + let marker = temp.path().join("survived-timeout"); + fake_e2fs_tool( + temp.path(), + "mke2fs", + &format!( + "(/bin/sleep 6; echo survived > '{}') &\nwait", + marker.display() + ), + ); + let error = preflight_in(&[temp.path().to_path_buf()]) + .await + .expect_err("wrapper must time out"); + assert!(error.contains("timed out after 5 seconds"), "{error}"); + tokio::time::sleep(Duration::from_millis(1300)).await; + assert!( + !marker.exists(), + "wrapper descendant survived its probe timeout" + ); + } + + #[tokio::test] + async fn filesystem_preflight_retains_exited_leader_until_held_pipe_cleanup() { + let temp = tempfile::tempdir().expect("wrapper installation"); + let marker = temp.path().join("survived-exited-leader"); + fake_e2fs_tool( + temp.path(), + "mke2fs", + &format!( + "(/bin/sleep 6; echo survived > '{}') &\necho 'mke2fs 1.47.4'\nexit 0", + marker.display() + ), + ); + let error = preflight_in(&[temp.path().to_path_buf()]) + .await + .expect_err("descendant holds output pipe"); + assert!(error.contains("timed out after 5 seconds"), "{error}"); + tokio::time::sleep(Duration::from_millis(1300)).await; + assert!( + !marker.exists(), + "descendant of exited probe leader survived cleanup" + ); + } + + #[tokio::test] + async fn filesystem_preflight_cancellation_terminates_wrapper_descendants() { + let temp = tempfile::tempdir().expect("wrapper installation"); + let ready = temp.path().join("child-ready"); + let marker = temp.path().join("survived-cancellation"); + fake_e2fs_tool( + temp.path(), + "mke2fs", + &format!( + "(echo ready > '{}'; /bin/sleep 6; echo survived > '{}') &\nwait", + ready.display(), + marker.display() + ), + ); + let path = temp.path().to_path_buf(); + let probe = tokio::spawn(async move { preflight_in(&[path]).await }); + for _ in 0..400 { + if ready.exists() { + break; + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + if !ready.exists() { + probe.abort(); + let result = probe.await; + panic!("probe descendant did not start: {result:?}"); + } + probe.abort(); + assert!(probe.await.expect_err("cancelled task").is_cancelled()); + tokio::time::sleep(Duration::from_millis(6300)).await; + assert!(!marker.exists(), "wrapper descendant survived cancellation"); + } + + #[test] + fn filesystem_preflight_requires_expected_identity_and_version() { + assert_eq!( + supported_version("mke2fs", "mke2fs 1.43 (test)"), + Some("1.43") + ); + for output in [ + "other 1.47.4", + "mke2fs unknown", + "mke2fs 1.42.13", + "mke2fs 1.47.garbage", + "mke2fs 1.47.4.5", + "", + ] { + assert_eq!(supported_version("mke2fs", output), None, "{output}"); + } + } +} diff --git a/crates/openshell-core/src/lib.rs b/crates/openshell-core/src/lib.rs index 472175f511..6f7a050f1c 100644 --- a/crates/openshell-core/src/lib.rs +++ b/crates/openshell-core/src/lib.rs @@ -17,6 +17,8 @@ pub mod denial; pub mod driver_mounts; pub mod driver_utils; pub mod dynamic_string_allowlist; +#[cfg(unix)] +pub mod e2fsprogs; pub mod endpoint_path; pub mod endpoint_status; pub mod error; diff --git a/crates/openshell-driver-vm/src/rootfs.rs b/crates/openshell-driver-vm/src/rootfs.rs index e1a09bcd98..cd4ee6307e 100644 --- a/crates/openshell-driver-vm/src/rootfs.rs +++ b/crates/openshell-driver-vm/src/rootfs.rs @@ -370,36 +370,20 @@ pub fn set_rootfs_image_file_mode( /// Replay the ext4 journal and repair automatically correctable filesystem /// state before the driver mutates a preserved guest disk offline. pub fn recover_rootfs_image(image_path: &Path) -> Result<(), String> { - let mut failures = Vec::new(); - let mut unavailable = Vec::new(); - - for candidate in e2fs_tool_candidates("e2fsck") { - let label = candidate.display().to_string(); - match Command::new(&candidate) - .arg("-p") - .arg("-f") - .arg(image_path) - .output() - { - Ok(output) if matches!(output.status.code(), Some(0..=2)) => return Ok(()), - Ok(output) => failures.push(format!( - "{label} failed with status {}\nstdout: {}\nstderr: {}", - output.status, - String::from_utf8_lossy(&output.stdout), - String::from_utf8_lossy(&output.stderr) - )), - Err(error) if error.kind() == std::io::ErrorKind::NotFound => { - unavailable.push(format!("{label} not found")); - } - Err(error) => failures.push(format!("run {label}: {error}")), - } + let mut command = e2fs_command("e2fsck")?; + let path = PathBuf::from(command.get_program()); + let output = command.arg("-p").arg("-f").arg(image_path).output(); + match output { + Ok(output) if matches!(output.status.code(), Some(0..=2)) => Ok(()), + Ok(output) => Err(format!( + "{} failed with status {}\nstdout: {}\nstderr: {}", + path.display(), + output.status, + String::from_utf8_lossy(&output.stdout), + String::from_utf8_lossy(&output.stderr) + )), + Err(error) => Err(format!("run {}: {error}", path.display())), } - - Err(if failures.is_empty() { - unavailable.join("\n") - } else { - failures.join("\n") - }) } #[cfg(target_os = "macos")] @@ -668,75 +652,33 @@ fn round_up_to_mib(bytes: u64) -> u64 { bytes.div_ceil(MIB) * MIB } -enum FormatterAttempt { - Succeeded, - Failed(String), - Unavailable(String), -} - fn format_ext4_image_from_dir(source: &Path, image_path: &Path) -> Result<(), String> { - let candidates = ["mke2fs", "mkfs.ext4"] - .into_iter() - .flat_map(e2fs_tool_candidates); - run_ext4_formatter_candidates(candidates, |candidate| { - let label = candidate.display().to_string(); - let output = Command::new(candidate) - .arg("-q") - .arg("-F") - .arg("-t") - .arg("ext4") - .arg("-E") - .arg("root_owner=0:0") - .arg("-d") - .arg(source) - .arg(image_path) - .output(); - match output { - Ok(output) if output.status.success() => FormatterAttempt::Succeeded, - Ok(output) => FormatterAttempt::Failed(format!( - "{label} failed with status {}\nstdout: {}\nstderr: {}", - output.status, - String::from_utf8_lossy(&output.stdout), - String::from_utf8_lossy(&output.stderr) - )), - Err(err) if err.kind() == std::io::ErrorKind::NotFound => { - FormatterAttempt::Unavailable(format!("{label} not found")) - } - Err(err) => FormatterAttempt::Failed(format!("run {label}: {err}")), - } - }) - .map_err(|details| { - format!( - "failed to create ext4 rootfs image from {}: {details}. Install e2fsprogs (mke2fs/mkfs.ext4) and retry", - source.display() - ) - }) -} - -fn run_ext4_formatter_candidates( - candidates: impl IntoIterator, - mut run: impl FnMut(&Path) -> FormatterAttempt, -) -> Result<(), String> { - let mut failures = Vec::new(); - let mut unavailable = Vec::new(); - - for candidate in candidates { - match run(&candidate) { - FormatterAttempt::Succeeded => return Ok(()), - FormatterAttempt::Failed(error) => failures.push(error), - FormatterAttempt::Unavailable(error) => unavailable.push(error), - } - } - - if failures.is_empty() { - Err(if unavailable.is_empty() { - "no ext4 formatter candidates configured".to_string() - } else { - unavailable.join("\n") - }) - } else { - Err(failures.join("\n")) + let path = openshell_core::e2fsprogs::resolve(&["mke2fs", "mkfs.ext4"])?; + let output = Command::new(&path) + .arg("-q") + .arg("-F") + .arg("-t") + .arg("ext4") + .arg("-E") + .arg("root_owner=0:0") + .arg("-d") + .arg(source) + .arg(image_path) + .output() + .map_err(|error| format!("run {}: {error}", path.display()))?; + if output.status.success() { + return Ok(()); } + // A selected formatter failure must retain its diagnostics. Retrying with + // another installation can overwrite a partial image and hide the cause. + Err(format!( + "failed to create ext4 rootfs image from {}: {} failed with status {}\nstdout: {}\nstderr: {}", + source.display(), + path.display(), + output.status, + String::from_utf8_lossy(&output.stdout), + String::from_utf8_lossy(&output.stderr) + )) } fn ensure_rootfs_image_parent_dirs(image_path: &Path, guest_path: &str) { @@ -885,52 +827,40 @@ pub fn ext4_image_has_directory(image_path: &Path, guest_path: &str) -> Result { - // debugfs exits 0 whether or not the path exists; the answer - // is only in its output. - let stdout = String::from_utf8_lossy(&output.stdout); - let stderr = String::from_utf8_lossy(&output.stderr); - if stdout.contains("Type: directory") { - return Ok(true); - } - if stdout.contains("Type: ") || stderr.contains("File not found") { - return Ok(false); - } - return Err(format!( - "debugfs command '{command}' produced unrecognized output for {}\nstdout: {stdout}\nstderr: {stderr}", - image_path.display() - )); + let output = e2fs_command("debugfs")? + .arg("-R") + .arg(&command) + .arg(image_path) + .output(); + match output { + Ok(output) if output.status.success() => { + // debugfs exits 0 whether or not the path exists; the answer + // is only in its output. + let stdout = String::from_utf8_lossy(&output.stdout); + let stderr = String::from_utf8_lossy(&output.stderr); + if stdout.contains("Type: directory") { + return Ok(true); } - Ok(output) => { - last_error = Some(format!( - "{label} failed with status {}\nstdout: {}\nstderr: {}", - output.status, - String::from_utf8_lossy(&output.stdout), - String::from_utf8_lossy(&output.stderr) - )); + if stdout.contains("Type: ") || stderr.contains("File not found") { + return Ok(false); } - Err(error) if error.kind() == std::io::ErrorKind::NotFound => { - last_error = Some(format!("{label} not found")); - } - Err(error) => last_error = Some(format!("run {label}: {error}")), + Err(format!( + "debugfs command '{command}' produced unrecognized output for {}\nstdout: {stdout}\nstderr: {stderr}", + image_path.display() + )) } + Ok(output) => Err(format!( + "debugfs command '{command}' failed for {}: debugfs failed with status {}\nstdout: {}\nstderr: {}. Install e2fsprogs (debugfs) and retry", + image_path.display(), + output.status, + String::from_utf8_lossy(&output.stdout), + String::from_utf8_lossy(&output.stderr) + )), + Err(error) => Err(format!( + "debugfs command '{command}' failed for {}: {error}. Install e2fsprogs (debugfs) and retry", + image_path.display() + )), } - - Err(format!( - "debugfs command '{command}' failed for {}: {}. Install e2fsprogs (debugfs) and retry", - image_path.display(), - last_error.unwrap_or_else(|| "debugfs not found".to_string()) - )) } fn sandbox_guest_user_ids_from_image_path( @@ -949,45 +879,33 @@ fn sandbox_guest_user_ids_from_image_path( let quoted_path = debugfs_quote_absolute_path(guest_path) .expect("the static passwd path is a valid debugfs path"); let command = format!("cat {quoted_path}"); - let mut last_error = None; - - for candidate in e2fs_tool_candidates("debugfs") { - let label = candidate.display().to_string(); - match Command::new(&candidate) - .arg("-R") - .arg(&command) - .arg(image_path) - .output() - { - Ok(output) if output.status.success() => { - let passwd = String::from_utf8(output.stdout).map_err(|error| { - format!( - "read {guest_path} from {} as UTF-8: {error}", - image_path.display() - ) - })?; - return parse_sandbox_guest_user_ids(&passwd, &image_path.display().to_string()); - } - Ok(output) => { - last_error = Some(format!( - "{label} failed with status {}\nstdout: {}\nstderr: {}", - output.status, - String::from_utf8_lossy(&output.stdout), - String::from_utf8_lossy(&output.stderr) - )); - } - Err(error) if error.kind() == std::io::ErrorKind::NotFound => { - last_error = Some(format!("{label} not found")); - } - Err(error) => last_error = Some(format!("run {label}: {error}")), + let output = e2fs_command("debugfs")? + .arg("-R") + .arg(&command) + .arg(image_path) + .output(); + match output { + Ok(output) if output.status.success() => { + let passwd = String::from_utf8(output.stdout).map_err(|error| { + format!( + "read {guest_path} from {} as UTF-8: {error}", + image_path.display() + ) + })?; + parse_sandbox_guest_user_ids(&passwd, &image_path.display().to_string()) } + Ok(output) => Err(format!( + "debugfs command '{command}' failed for {}: debugfs failed with status {}\nstdout: {}\nstderr: {}. Install e2fsprogs (debugfs) and retry", + image_path.display(), + output.status, + String::from_utf8_lossy(&output.stdout), + String::from_utf8_lossy(&output.stderr) + )), + Err(error) => Err(format!( + "debugfs command '{command}' failed for {}: {error}. Install e2fsprogs (debugfs) and retry", + image_path.display() + )), } - - Err(format!( - "debugfs command '{command}' failed for {}: {}. Install e2fsprogs (debugfs) and retry", - image_path.display(), - last_error.unwrap_or_else(|| "debugfs not found".to_string()) - )) } fn sandbox_guest_user_ids(rootfs: &Path) -> Result, String> { @@ -1037,83 +955,57 @@ fn run_debugfs_batch(image_path: &Path, commands: &[String]) -> Result<(), Strin } fn run_debugfs_batch_file(image_path: &Path, command_path: &Path) -> Result<(), String> { - let mut last_error = None; - for candidate in e2fs_tool_candidates("debugfs") { - let label = candidate.display().to_string(); - let output = Command::new(&candidate) - .arg("-w") - .arg("-f") - .arg(command_path) - .arg(image_path) - .output(); - match output { - Ok(output) if output.status.success() => return Ok(()), - Ok(output) => { - last_error = Some(format!( - "{label} failed with status {}\nstdout: {}\nstderr: {}", - output.status, - String::from_utf8_lossy(&output.stdout), - String::from_utf8_lossy(&output.stderr) - )); - } - Err(err) if err.kind() == std::io::ErrorKind::NotFound => { - last_error = Some(format!("{label} not found")); - } - Err(err) => { - last_error = Some(format!("run {label}: {err}")); - } - } + let output = e2fs_command("debugfs")? + .arg("-w") + .arg("-f") + .arg(command_path) + .arg(image_path) + .output(); + match output { + Ok(output) if output.status.success() => Ok(()), + Ok(output) => Err(format!( + "debugfs batch {} failed for {}: debugfs failed with status {}\nstdout: {}\nstderr: {}. Install e2fsprogs (debugfs) and retry", + command_path.display(), + image_path.display(), + output.status, + String::from_utf8_lossy(&output.stdout), + String::from_utf8_lossy(&output.stderr) + )), + Err(error) => Err(format!( + "debugfs batch {} failed for {}: {error}. Install e2fsprogs (debugfs) and retry", + command_path.display(), + image_path.display() + )), } - Err(format!( - "debugfs batch {} failed for {}: {}. Install e2fsprogs (debugfs) and retry", - command_path.display(), - image_path.display(), - last_error.unwrap_or_else(|| "debugfs not found".to_string()) - )) } fn run_debugfs(image_path: &Path, command: &str) -> Result<(), String> { - let mut last_error = None; - for candidate in e2fs_tool_candidates("debugfs") { - let label = candidate.display().to_string(); - let output = Command::new(&candidate) - .arg("-w") - .arg("-R") - .arg(command) - .arg(image_path) - .output(); - match output { - Ok(output) if output.status.success() => return Ok(()), - Ok(output) => { - last_error = Some(format!( - "{label} failed with status {}\nstdout: {}\nstderr: {}", - output.status, - String::from_utf8_lossy(&output.stdout), - String::from_utf8_lossy(&output.stderr) - )); - } - Err(err) if err.kind() == std::io::ErrorKind::NotFound => { - last_error = Some(format!("{label} not found")); - } - Err(err) => { - last_error = Some(format!("run {label}: {err}")); - } - } + let output = e2fs_command("debugfs")? + .arg("-w") + .arg("-R") + .arg(command) + .arg(image_path) + .output(); + match output { + Ok(output) if output.status.success() => Ok(()), + Ok(output) => Err(format!( + "debugfs command '{command}' failed for {}: debugfs failed with status {}\nstdout: {}\nstderr: {}. Install e2fsprogs (debugfs) and retry", + image_path.display(), + output.status, + String::from_utf8_lossy(&output.stdout), + String::from_utf8_lossy(&output.stderr) + )), + Err(error) => Err(format!( + "debugfs command '{command}' failed for {}: {error}. Install e2fsprogs (debugfs) and retry", + image_path.display() + )), } - Err(format!( - "debugfs command '{command}' failed for {}: {}. Install e2fsprogs (debugfs) and retry", - image_path.display(), - last_error.unwrap_or_else(|| "debugfs not found".to_string()) - )) } -fn e2fs_tool_candidates(tool: &str) -> Vec { - let mut candidates = vec![PathBuf::from(tool)]; - for root in ["/opt/homebrew/opt/e2fsprogs", "/usr/local/opt/e2fsprogs"] { - candidates.push(Path::new(root).join("sbin").join(tool)); - candidates.push(Path::new(root).join("bin").join(tool)); - } - candidates +fn e2fs_command(tool: &str) -> Result { + // Preflight and image operations must execute the same selected file with + // the same inherited environment, including when PATH is restricted. + Ok(Command::new(openshell_core::e2fsprogs::resolve(&[tool])?)) } fn temporary_injection_path(image_path: &Path) -> PathBuf { @@ -1558,10 +1450,7 @@ mod tests { #[test] fn recover_rootfs_image_accepts_clean_ext4_image() { - if !e2fs_tool_candidates("e2fsck") - .iter() - .any(|candidate| Command::new(candidate).arg("-V").output().is_ok()) - { + if e2fs_command("e2fsck").is_err() { return; } @@ -1580,10 +1469,7 @@ mod tests { #[test] fn ext4_image_has_directory_distinguishes_directories_files_and_missing_paths() { - if !e2fs_tool_candidates("debugfs") - .iter() - .any(|candidate| Command::new(candidate).arg("-V").output().is_ok()) - { + if e2fs_command("debugfs").is_err() { return; } @@ -1668,58 +1554,90 @@ mod tests { assert_eq!(debugfs_quote_argument("/tmp/bad\npath"), None); } - #[test] - fn formatter_candidates_preserve_executed_failure_over_missing_fallback() { - let candidates = vec![PathBuf::from("mke2fs"), PathBuf::from("missing")]; - - let err = run_ext4_formatter_candidates(candidates, |candidate| { - if candidate == Path::new("mke2fs") { - FormatterAttempt::Failed( - "mke2fs failed with status 1\nstdout: formatter output\nstderr: no space left" - .to_string(), - ) - } else { - FormatterAttempt::Unavailable("missing not found".to_string()) - } - }) - .expect_err("formatter should fail"); - - assert!(err.contains("mke2fs failed with status 1")); - assert!(err.contains("no space left")); - assert!(!err.contains("missing not found")); + fn run_tool_fixture_test(test: &str, directory: &Path) { + // Isolate PATH in a child test process. Other tests may prepare images + // concurrently and must never observe this deliberately broken tool. + let output = Command::new(std::env::current_exe().expect("test executable")) + .args(["--exact", test, "--nocapture"]) + .env("PATH", directory) + .env("OPENSHELL_ROOTFS_TOOL_TEST_ROOT", directory) + .output() + .expect("isolated tool test"); + assert!( + output.status.success(), + "stdout: {}\nstderr: {}", + String::from_utf8_lossy(&output.stdout), + String::from_utf8_lossy(&output.stderr) + ); } #[test] - fn formatter_candidates_report_all_missing_tools() { - let candidates = vec![PathBuf::from("mke2fs"), PathBuf::from("mkfs.ext4")]; - - let err = run_ext4_formatter_candidates(candidates, |candidate| { - FormatterAttempt::Unavailable(format!("{} not found", candidate.display())) - }) - .expect_err("formatter should be unavailable"); + fn formatter_preserves_selected_tool_failure_without_retry() { + use std::os::unix::fs::PermissionsExt as _; - assert!(err.contains("mke2fs not found")); - assert!(err.contains("mkfs.ext4 not found")); + if let Some(directory) = std::env::var_os("OPENSHELL_ROOTFS_TOOL_TEST_ROOT") { + let directory = PathBuf::from(directory); + let image = directory.join("rootfs.ext4"); + File::create(&image) + .expect("image") + .set_len(16 * 1024 * 1024) + .expect("image size"); + let source = directory.join("source"); + fs::create_dir(&source).expect("source directory"); + let error = format_ext4_image_from_dir(&source, &image) + .expect_err("the selected formatter failure must not select another installation"); + assert!(error.contains("selected-formatter-failed"), "{error}"); + assert!(error.contains("42"), "{error}"); + return; + } + let directory = tempfile::tempdir().expect("tool installation"); + for (name, body) in [ + ("mke2fs", "echo selected-formatter-failed >&2\nexit 42"), + ("mkfs.ext4", "exit 0"), + ] { + let path = directory.path().join(name); + fs::write(&path, format!("#!/bin/sh\n{body}\n")).expect("tool fixture"); + fs::set_permissions(path, fs::Permissions::from_mode(0o755)).expect("executable"); + } + run_tool_fixture_test( + "rootfs::tests::formatter_preserves_selected_tool_failure_without_retry", + directory.path(), + ); } #[test] - fn formatter_candidates_accept_successful_fallback() { - let candidates = vec![PathBuf::from("first"), PathBuf::from("second")]; - let mut attempted = Vec::new(); - - run_ext4_formatter_candidates(candidates, |candidate| { - attempted.push(candidate.to_path_buf()); - if candidate == Path::new("second") { - FormatterAttempt::Succeeded - } else { - FormatterAttempt::Failed("first failed".to_string()) - } - }) - .expect("fallback should succeed"); + fn recovery_preserves_selected_path_and_execution_errors() { + use std::os::unix::fs::PermissionsExt as _; - assert_eq!( - attempted, - vec![PathBuf::from("first"), PathBuf::from("second")] + if let Some(directory) = std::env::var_os("OPENSHELL_ROOTFS_TOOL_TEST_ROOT") { + let directory = PathBuf::from(directory); + let path = directory.join("e2fsck"); + for (script, expected) in [ + ( + "#!/nonexistent/e2fsprogs-interpreter\n", + "No such file or directory", + ), + ( + "#!/bin/sh\necho recovery-failed >&2\nexit 4\n", + "recovery-failed", + ), + ] { + fs::write(&path, script).expect("recovery tool"); + fs::set_permissions(&path, fs::Permissions::from_mode(0o755)).expect("executable"); + let error = recover_rootfs_image(&directory.join("overlay.ext4")) + .expect_err("recovery failure"); + assert!( + error.contains(path.canonicalize().unwrap().to_str().unwrap()), + "{error}" + ); + assert!(error.contains(expected), "{error}"); + } + return; + } + let directory = tempfile::tempdir().expect("tool installation"); + run_tool_fixture_test( + "rootfs::tests::recovery_preserves_selected_path_and_execution_errors", + directory.path(), ); } diff --git a/crates/openshell-gateway/src/lib.rs b/crates/openshell-gateway/src/lib.rs index dc0ded3f0b..2da768f54c 100644 --- a/crates/openshell-gateway/src/lib.rs +++ b/crates/openshell-gateway/src/lib.rs @@ -381,6 +381,15 @@ impl openshell_server::ComputeDriverFactory for VmFactory { true } + async fn preflight_host_tools( + &self, + cancellation: tokio::sync::watch::Receiver, + ) -> openshell_core::Result> { + openshell_core::e2fsprogs::preflight(cancellation) + .await + .map_err(openshell_core::Error::config) + } + fn validate_config( &self, context: openshell_server::ComputeDriverConfigContext<'_>, diff --git a/crates/openshell-gateway/tests/config_preflight.rs b/crates/openshell-gateway/tests/config_preflight.rs new file mode 100644 index 0000000000..afdf956b72 --- /dev/null +++ b/crates/openshell-gateway/tests/config_preflight.rs @@ -0,0 +1,276 @@ +// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +#![cfg(all(unix, feature = "compute-driver-vm"))] + +use std::fs; +use std::os::unix::fs::PermissionsExt as _; +use std::path::{Path, PathBuf}; +use std::process::Output; +use std::time::Duration; + +struct Fixture { + root: tempfile::TempDir, + tools: PathBuf, + config: PathBuf, +} + +impl Fixture { + fn new() -> Self { + let root = tempfile::tempdir().expect("fixture root"); + let tools = root.path().join("tools"); + fs::create_dir(&tools).expect("tools directory"); + let config = root.path().join("gateway.toml"); + let fixture = Self { + root, + tools, + config, + }; + fixture.write_config(Some("vm")); + for name in ["mke2fs", "debugfs", "e2fsck"] { + fixture.tool(name, &format!( + "test \"$1\" = -V || exit 64\ntest \"$#\" = 1 || exit 65\nread ignored && exit 66\ntest \"$PREFLIGHT_TEST_ENV\" = inherited || exit 67\necho '{name} 1.47.4' >&2" + )); + } + fixture + } + + fn write_config(&self, driver: Option<&str>) { + let selector = + driver.map_or_else(String::new, |name| format!("compute_driver = {name:?}\n")); + // TOML serialization preserves quotes and escapes in the fixture path. + let state_dir = toml::Value::try_from(self.root.path().join("vm-state")) + .expect("serialize VM state directory"); + fs::write(&self.config, format!( + "[openshell]\nversion = 2\n[openshell.gateway]\ndisable_tls = true\n{selector}[openshell.drivers.vm]\nbootstrap_image = 'unreachable.invalid/vm:must-not-pull'\nstate_dir = {state_dir}\n" + )).expect("gateway configuration"); + } + + fn tool(&self, name: &str, body: &str) { + let path = self.tools.join(name); + fs::write(&path, format!("#!/bin/sh\n{body}\n")).expect("tool fixture"); + fs::set_permissions(path, fs::Permissions::from_mode(0o755)).expect("executable fixture"); + } + + fn command(&self, replay: &[&str]) -> tokio::process::Command { + let mut command = tokio::process::Command::new(env!("CARGO_BIN_EXE_openshell-gateway")); + command + .env_clear() + .env("HOME", self.root.path()) + .env("PATH", &self.tools) + .env("XDG_CONFIG_HOME", self.root.path().join("config-home")) + .env("XDG_STATE_HOME", self.root.path().join("state-home")) + .env("PREFLIGHT_TEST_ENV", "inherited") + .args(["config", "preflight"]) + .kill_on_drop(true); + if replay.is_empty() { + command.arg("--path").arg(&self.config); + } else { + command + .arg("--") + .arg("--config") + .arg(&self.config) + .args(replay); + } + command + } + + async fn run(&self, replay: &[&str]) -> Output { + let original_config = fs::read(&self.config).expect("original config"); + let mut command = self.command(replay); + let output = tokio::time::timeout(Duration::from_secs(15), command.output()) + .await + .expect("preflight must finish") + .expect("run gateway preflight"); + assert_eq!( + fs::read(&self.config).expect("unchanged config"), + original_config + ); + for name in ["vm-state", "state-home", "config-home", ".local"] { + assert!( + !self.root.path().join(name).exists(), + "preflight created {name}" + ); + } + output + } +} + +fn combined(output: &Output) -> String { + format!( + "{}{}", + String::from_utf8_lossy(&output.stdout), + String::from_utf8_lossy(&output.stderr) + ) +} + +fn normalized_diagnostic(output: &Output) -> String { + // Terminal line wrapping can split an error sentence after a long path. + combined(output) + .split_whitespace() + .filter(|word| *word != "│") + .collect::>() + .join(" ") +} + +#[tokio::test] +async fn local_vm_reports_tools_without_starting_driver_or_creating_state() { + let fixture = Fixture::new(); + let output = fixture.run(&[]).await; + assert!(output.status.success(), "{}", combined(&output)); + let report = String::from_utf8_lossy(&output.stdout); + for name in ["mke2fs", "debugfs", "e2fsck"] { + let path = fixture + .tools + .join(name) + .canonicalize() + .expect("selected path"); + assert!( + report.contains(path.to_str().expect("UTF-8 path")), + "{report}" + ); + } + // PATH contains only the filesystem fixtures, never a VM driver binary. + assert!(!fixture.tools.join("openshell-driver-vm").exists()); +} + +#[tokio::test] +async fn config_file_retains_selected_tool_error_and_corrective_guidance() { + let fixture = Fixture::new(); + fixture.tool("debugfs", "echo 'fixture loader failure' >&2\nexit 42"); + let output = fixture.run(&[]).await; + assert!(!output.status.success()); + let report = combined(&output); + for expected in [ + "debugfs", + "fixture loader failure", + "42", + "gateway service PATH", + "mke2fs", + ] { + assert!(report.contains(expected), "missing {expected}: {report}"); + } + assert!( + report.contains(fixture.tools.to_str().expect("tool path")), + "{report}" + ); +} + +#[tokio::test] +async fn local_vm_rejects_nonexecutable_and_unsupported_tools() { + let fixture = Fixture::new(); + fs::set_permissions( + fixture.tools.join("mke2fs"), + fs::Permissions::from_mode(0o644), + ) + .unwrap(); + let output = fixture.run(&[]).await; + assert!(!output.status.success()); + let report = normalized_diagnostic(&output); + assert!(report.contains("not an executable file"), "{report}"); + + fixture.tool("mke2fs", "echo 'mke2fs 1.42.13' >&2"); + let output = fixture.run(&[]).await; + assert!(!output.status.success()); + let report = normalized_diagnostic(&output); + assert!(report.contains("not a supported mke2fs"), "{report}"); +} + +#[tokio::test] +async fn local_vm_rejects_hanging_tool_with_deadline() { + let fixture = Fixture::new(); + fixture.tool("mke2fs", "exec /bin/sleep 30"); + let start = std::time::Instant::now(); + let output = fixture.run(&[]).await; + assert!(!output.status.success()); + assert!( + combined(&output).contains("timed out after 5 seconds"), + "{}", + combined(&output) + ); + assert!(start.elapsed() < Duration::from_secs(12)); +} + +#[tokio::test] +async fn signals_stop_probe_wrapper_and_descendants_before_cli_exit() { + use nix::sys::signal::{Signal, kill}; + use nix::unistd::Pid; + use std::process::Stdio; + + for signal in [Signal::SIGINT, Signal::SIGTERM] { + let fixture = Fixture::new(); + let ready = fixture.root.path().join("probe-ready"); + let survived = fixture.root.path().join("survived-signal"); + fixture.tool( + "mke2fs", + &format!( + "(echo ready > '{}'; /bin/sleep 2; echo survived > '{}') &\nwait", + ready.display(), + survived.display() + ), + ); + let child = fixture + .command(&[]) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()) + .spawn() + .expect("preflight process"); + let pid = + Pid::from_raw(i32::try_from(child.id().expect("gateway PID")).expect("valid PID")); + tokio::time::timeout(Duration::from_secs(5), async { + while !ready.exists() { + tokio::time::sleep(Duration::from_millis(10)).await; + } + }) + .await + .expect("probe descendant ready"); + kill(pid, signal).expect("signal gateway"); + let output = tokio::time::timeout(Duration::from_secs(5), child.wait_with_output()) + .await + .expect("signal cleanup deadline") + .expect("preflight exit"); + assert!(!output.status.success()); + tokio::time::sleep(Duration::from_millis(2200)).await; + assert!(!survived.exists(), "probe descendant survived {signal}"); + assert!( + combined(&output).contains(&format!("interrupted by {signal}")), + "{}", + combined(&output) + ); + assert!(!fixture.root.path().join("vm-state").exists()); + } +} + +#[tokio::test] +async fn remote_vm_reports_unperformed_checks_without_running_local_tools() { + let fixture = Fixture::new(); + fixture.tool("mke2fs", "echo 'local tool must not run' >&2\nexit 42"); + let socket = fixture.root.path().join("absent-remote.sock"); + let output = fixture + .run(&["--compute-driver-socket", socket.to_str().unwrap()]) + .await; + assert!(output.status.success(), "{}", combined(&output)); + assert!(combined(&output).contains("host tool checks not performed for a remote endpoint")); + assert!(!combined(&output).contains("local tool must not run")); + assert!(!Path::new(&socket).exists()); +} + +#[tokio::test] +async fn vm_table_without_selection_does_not_probe_tools() { + let fixture = Fixture::new(); + fixture.write_config(None); + fixture.tool("mke2fs", "exit 42"); + let output = fixture.run(&[]).await; + assert!(output.status.success(), "{}", combined(&output)); + assert!(!combined(&output).contains("VM host tool")); +} + +#[cfg(feature = "compute-driver-docker")] +#[tokio::test] +async fn unrelated_driver_does_not_require_vm_tools() { + let fixture = Fixture::new(); + fixture.tool("mke2fs", "exit 42"); + let output = fixture.run(&["--compute-driver", "docker"]).await; + assert!(output.status.success(), "{}", combined(&output)); + assert!(!combined(&output).contains("VM host tool")); +} diff --git a/crates/openshell-server/src/cli.rs b/crates/openshell-server/src/cli.rs index 87b67f5e8a..93d98ed105 100644 --- a/crates/openshell-server/src/cli.rs +++ b/crates/openshell-server/src/cli.rs @@ -286,7 +286,12 @@ pub async fn run_cli_with_compute_drivers(compute_drivers: ComputeDriverRegistry Some(Commands::GenerateCerts(args)) => certgen::run(args).await, Some(Commands::Config(args)) => match args.command { ConfigCommand::Preflight(args) => { - run_config_preflight_with_drivers(args, cli.run, &matches, &compute_drivers) + let driver = + run_config_preflight_with_drivers(args, cli.run, &matches, &compute_drivers)?; + for report in preflight_host_tools(driver).await? { + println!("{report}"); + } + Ok(()) } }, None => Box::pin(run_from_args(cli.run, matches, compute_drivers)).await, @@ -751,7 +756,7 @@ fn run_config_preflight( Some(detect_preflight_test_driver), PreflightTestFactory, )?)?; - run_config_preflight_with_drivers(args, run, matches, ®istry) + run_config_preflight_with_drivers(args, run, matches, ®istry).map(|_| ()) } fn run_config_preflight_with_drivers( @@ -759,7 +764,7 @@ fn run_config_preflight_with_drivers( run: RunArgs, matches: &ArgMatches, compute_drivers: &ComputeDriverRegistry, -) -> Result<()> { +) -> Result> { if args.gateway_args.is_empty() { return run_effective_config_preflight(args.path, run, matches, compute_drivers); } @@ -774,7 +779,7 @@ fn run_config_preflight_with_drivers( clap::error::ErrorKind::DisplayHelp | clap::error::ErrorKind::DisplayVersion ) => { - return Ok(()); + return Ok(None); } Err(error) => return Err(miette::miette!("{error}")), }; @@ -783,7 +788,7 @@ fn run_config_preflight_with_drivers( if replay.command.is_some() { // A valid non-daemon action does not consume gateway startup // configuration. Let the immediately following invocation perform it. - return Ok(()); + return Ok(None); } run_effective_config_preflight(None, replay.run, &replay_matches, compute_drivers) } @@ -793,7 +798,7 @@ fn run_effective_config_preflight( mut run: RunArgs, matches: &ArgMatches, compute_drivers: &ComputeDriverRegistry, -) -> Result<()> { +) -> Result> { let path = if path_override.is_some() { path_override } else { @@ -838,8 +843,9 @@ fn run_effective_config_preflight( gateway_tls_enabled: !run.disable_tls, endpoint_overrides: &endpoint_overrides, }; + let mut selected_driver = None; if let Some(selection) = selection.as_ref() { - crate::validate_compute_driver_config( + selected_driver = Some(crate::validate_compute_driver_config( compute_drivers, selection.name(), run.name.trim(), @@ -847,7 +853,7 @@ fn run_effective_config_preflight( &run.log_level, driver_startup, true, - )?; + )?); } else if file.is_some() { // Runtime auto-detection may connect local API sockets or launch a // bounded discovery command. Preflight must not perform those @@ -869,11 +875,11 @@ fn run_effective_config_preflight( )?; } } - Ok(()) + Ok(selected_driver) })(); match (validation, path.as_ref()) { - (Ok(()), _) => Ok(()), + (Ok(driver), _) => Ok(driver), (Err(_), Some(path)) => Err(miette::miette!( "{}", config_file::ConfigPreflightError::invalid_current(path) @@ -882,6 +888,56 @@ fn run_effective_config_preflight( } } +/// Run executable probes outside the pure configuration-validation context. +async fn preflight_host_tools( + driver: Option, +) -> Result> { + match driver { + Some(crate::ConfiguredComputeDriver::Registered(registration)) => { + let (cancellation_tx, cancellation_rx) = tokio::sync::watch::channel(false); + #[cfg(unix)] + { + use tokio::signal::unix::{SignalKind, signal}; + + // Register before polling the hook: a probe owns a separate + // process group, so default CLI termination cannot clean it up. + let mut interrupt = signal(SignalKind::interrupt()) + .map_err(|error| miette::miette!("register preflight SIGINT: {error}"))?; + let mut terminate = signal(SignalKind::terminate()) + .map_err(|error| miette::miette!("register preflight SIGTERM: {error}"))?; + let check = registration.factory.preflight_host_tools(cancellation_rx); + tokio::pin!(check); + let reason = tokio::select! { + biased; + _ = interrupt.recv() => "SIGINT", + _ = terminate.recv() => "SIGTERM", + result = &mut check => return result.map_err(|error| miette::miette!("{error}")), + }; + cancellation_tx.send_replace(true); + // The hook owns its children. Await its cancellation cleanup + // before the short-lived CLI shuts down the Tokio runtime. + let _ = check.await; + Err(miette::miette!( + "host tool preflight interrupted by {reason}" + )) + } + #[cfg(not(unix))] + { + let _cancellation_tx = cancellation_tx; + registration + .factory + .preflight_host_tools(cancellation_rx) + .await + .map_err(|error| miette::miette!("{error}")) + } + } + Some(crate::ConfiguredComputeDriver::Remote { name }) => Ok(vec![format!( + "compute driver '{name}': host tool checks not performed for a remote endpoint; run preflight on the driver host with its service account and environment" + )]), + None => Ok(Vec::new()), + } +} + fn validate_preflight_semantics( args: &RunArgs, matches: &ArgMatches, diff --git a/crates/openshell-server/src/lib.rs b/crates/openshell-server/src/lib.rs index 72efa01fbf..353ca4396b 100644 --- a/crates/openshell-server/src/lib.rs +++ b/crates/openshell-server/src/lib.rs @@ -1268,6 +1268,21 @@ pub trait ComputeDriverFactory: Send + Sync { false } + /// Check locally installed host tools after configuration validation. + /// + /// Only the explicit `config preflight` command calls this hook. Probes + /// must bound time and output, clean up on cancellation, and avoid driver + /// startup, transport connections, images, and runtime state. Return + /// operator-readable results including the selected executable paths. + /// The process inherits the gateway's account and environment. When + /// `cancellation` becomes true, finish process cleanup before returning. + async fn preflight_host_tools( + &self, + _cancellation: watch::Receiver, + ) -> Result> { + Ok(Vec::new()) + } + async fn build(&self, context: ComputeDriverBuildContext<'_>) -> Result; } diff --git a/deploy/man/openshell-gateway.8.md b/deploy/man/openshell-gateway.8.md index 9be010095a..4bb00717ed 100644 --- a/deploy/man/openshell-gateway.8.md +++ b/deploy/man/openshell-gateway.8.md @@ -120,7 +120,7 @@ Validate a gateway configuration before starting the daemon: With no path, preflight validates a nonempty OPENSHELL_GATEWAY_CONFIG. If that variable is unset, it optionally validates an auto-discovered XDG config. The -absence of either config succeeds. An explicit missing path, legacy schema-v1 +absence of either config still validates the effective daemon arguments. An explicit missing path, legacy schema-v1 file, invalid TOML, symlink, or nonregular file fails with a nonzero status. Preflight merges file and environment values and applies read-only startup checks for selector and socket normalization, registered compute-driver configuration, @@ -134,6 +134,10 @@ Arguments after **--** replace **--path** mode and are parsed as the exact gatew daemon invocation. Package wrappers use this form so command-line overrides are validated before the same arguments reach startup. +An explicitly selected local **vm** driver also checks **mke2fs** or **mkfs.ext4**, **debugfs**, and **e2fsck**. Install e2fsprogs 1.43 or newer with the operating system's package manager, then run preflight with the gateway service's account, working directory, configuration, and environment. The command reports selected executable paths and versions. A restricted service **PATH** can select different tools from an interactive shell; include the installation's bin and sbin directories in that service's environment. + +Each executable receives only **-V**, with a five-second deadline and an 8 KiB output limit per stream. Missing, non-executable, unsupported, or failing tools return nonzero status with installation or repair guidance. Preflight creates no images or runtime state and does not start the VM driver. Other drivers do not require these tools. A remote driver endpoint reports that host tool checks were not performed; check a local VM configuration on that host in the driver service's environment. + The Debian and Ubuntu systemd user unit runs preflight before certificate generation, while retaining its EnvironmentFile and bare ExecStart behavior. The Snap wrapper replays its effective daemon arguments through preflight. It first diff --git a/docs/how-it-works/gateways/configuration.mdx b/docs/how-it-works/gateways/configuration.mdx index 4cfd902daa..52ebbed7fa 100644 --- a/docs/how-it-works/gateways/configuration.mdx +++ b/docs/how-it-works/gateways/configuration.mdx @@ -1240,22 +1240,7 @@ changing it: openshell-gateway config preflight --path ~/.config/openshell/gateway.toml ``` -Without `--path`, the command validates a nonempty `OPENSHELL_GATEWAY_CONFIG`. -Otherwise, it validates an existing XDG gateway config when one is discovered. -When neither source selects a config, preflight succeeds. An explicit missing path, -a legacy schema-v1 file, invalid TOML, a symlink, or any nonregular file fails. -Preflight merges the selected file with the current `OPENSHELL_*` environment and -applies the daemon's read-only startup checks. These checks include selector and -socket normalization, registered-driver selection and configuration, rate-limit -pairs, TLS and mTLS relationships, interceptor registrations, and supervisor -middleware registrations. When a selected file omits `compute_driver`, preflight -validates each configured table for an auto-detectable driver without running the -runtime detection probes, which can connect local sockets or launch discovery -commands. It validates complete guest TLS path sets without requiring -package-generated certificates to exist before certificate generation. It does -not construct a compute driver or connect to a transport. A failed -preflight always preserves the file; it never migrates, replaces, or rewrites -configuration. +Without `--path`, the command validates a nonempty `OPENSHELL_GATEWAY_CONFIG`. Otherwise, it validates an existing XDG gateway config when one is discovered. When neither source selects a config, preflight validates the effective daemon arguments. An explicit missing path, a legacy schema-v1 file, invalid TOML, a symlink, or any nonregular file fails. Preflight merges the selected file with the current `OPENSHELL_*` environment and applies the daemon's read-only startup checks. These checks include selector and socket normalization, registered-driver selection and configuration, rate-limit pairs, TLS and mTLS relationships, interceptor registrations, and supervisor middleware registrations. When a selected file omits `compute_driver`, preflight validates each configured table for an auto-detectable driver without running the runtime detection probes, which can connect local sockets or launch discovery commands. It validates configured driver TLS requirements, including the gateway CA, without requiring package-generated certificate files to exist before certificate generation. It does not construct a compute driver or connect to a transport. A failed preflight always preserves the file; it never migrates, replaces, or rewrites configuration. To validate the exact daemon arguments that a wrapper will pass, place them after `--` instead of using `--path`: @@ -1282,3 +1267,26 @@ $EDITOR ~/.config/openshell/gateway.toml openshell-gateway config preflight --path ~/.config/openshell/gateway.toml systemctl --user restart openshell-gateway ``` + +### Check local VM host tools + +For an explicitly selected local `vm` driver, preflight also checks a formatter (`mke2fs` or `mkfs.ext4`), `debugfs`, and `e2fsck`. Install e2fsprogs 1.43 or newer with your operating system's package manager. OpenShell does not install or manage these host dependencies. + +For example, on macOS: + +```shell +brew install e2fsprogs +openshell-gateway config preflight -- --config ~/.config/openshell/gateway.toml --compute-driver vm +``` + +Run preflight with the account, configuration, working directory, and environment intended for the gateway. A successful check in an interactive shell does not check a service with a different `PATH`. For a service whose environment uses a private installation, reproduce its configured path explicitly: + +```shell +env PATH=/opt/company/e2fsprogs/sbin:/opt/company/e2fsprogs/bin:/usr/bin:/bin /usr/local/bin/openshell-gateway config preflight -- --config ~/.config/openshell/gateway.toml --compute-driver vm +``` + +Replace the executable and configuration paths with your installation's paths, and run this command as the service account. Preflight and VM image operations search inherited `PATH` entries, then the existing e2fsprogs `sbin` and `bin` directories under `/opt/homebrew/opt/e2fsprogs` and `/usr/local/opt/e2fsprogs`. They prefer `mke2fs` and try `mkfs.ext4` only when `mke2fs` is absent. A selected file that cannot run is an error; fix that installation or the service's `PATH` before rerunning the command. + +The command reports each selected executable path and version. Each tool receives only `-V`, with a five-second deadline and an 8 KiB output limit per stream. Missing, non-executable, outdated, or failing tools return a nonzero exit status with corrective guidance and the selected tool's error. The checks create no gateway state or images and do not start a driver or VM. Version checks confirm the installation; they do not test image creation or filesystem recovery. + +Other drivers do not require these VM tools. A VM configuration table alone does not select the VM driver. When a remote driver socket is configured, preflight reports that host tool checks were not performed; run the check on that host in the driver service's environment with a local VM configuration. diff --git a/skills/debug-openshell-cluster/SKILL.md b/skills/debug-openshell-cluster/SKILL.md index ab774e3a88..d7d1f3f1be 100644 --- a/skills/debug-openshell-cluster/SKILL.md +++ b/skills/debug-openshell-cluster/SKILL.md @@ -910,8 +910,7 @@ Use the VM driver logs and host diagnostics available in the user's environment. - The VM driver process is running and reachable by the gateway. - The runtime rootfs exists and matches the expected architecture. -- `mke2fs` or `mkfs.ext4` and `debugfs` from e2fsprogs are installed; explicit - `sandbox_uid`/`sandbox_gid` does not remove this prerequisite. +- `mke2fs` or `mkfs.ext4`, `debugfs`, and `e2fsck` from e2fsprogs are installed. Run `openshell-gateway config preflight` with the intended local VM configuration, service account, and environment to check the selected paths and versions before startup. A remote endpoint reports that host checks were not performed. Explicit `sandbox_uid`/`sandbox_gid` does not remove this prerequisite. - A persisted overlay identity error is resolved from its owner marker, overlay upper layer, prepared rootfs, explicit config, or current image. Do not assign `10001:10001` unless the persisted state reports that legacy identity. From e7d14edd884fe1feaa806dd01089da34ba2fdcd5 Mon Sep 17 00:00:00 2001 From: Eric Curtin Date: Sat, 3 Oct 2026 20:30:24 +0000 Subject: [PATCH 10/13] fix(relay): close outbound relay stream when target closes first (#3772) * fix(relay): close outbound relay stream when target closes first Closes #3724 out_tx was cloned into the target-reading task, so the function's own copy kept the outbound relay stream open until the client side also ended. A target that closed a keep-alive connection never reached the client as EOF, so reused connections hung forever. Move out_tx into the target-reading task instead of cloning it, so dropping it on target EOF ends the outbound stream right away. Signed-off-by: Eric Curtin * fix(relay): keep client uploads after target half-close Signed-off-by: Eric Curtin --------- Signed-off-by: Eric Curtin --- crates/openshell-cli/src/run.rs | 124 ++++++++++-- .../src/supervisor_session.rs | 179 +++++++++++++++++- 2 files changed, 283 insertions(+), 20 deletions(-) diff --git a/crates/openshell-cli/src/run.rs b/crates/openshell-cli/src/run.rs index 7432c0175f..289bfa8bb5 100644 --- a/crates/openshell-cli/src/run.rs +++ b/crates/openshell-cli/src/run.rs @@ -2304,7 +2304,6 @@ async fn forward_one_tcp_connection( service_id: String, authorization_token: String, ) -> std::result::Result<(), ForwardTcpConnectionError> { - use tokio::io::{AsyncReadExt, AsyncWriteExt}; use tokio_stream::wrappers::ReceiverStream; let (tx, rx) = tokio::sync::mpsc::channel::(16); @@ -2325,7 +2324,7 @@ async fn forward_one_tcp_connection( .await .map_err(|_| ForwardTcpConnectionError::transient("failed to initialize forward stream"))?; - let mut response = match client.forward_tcp(ReceiverStream::new(rx)).await { + let response = match client.forward_tcp(ReceiverStream::new(rx)).await { Ok(response) => response.into_inner(), Err(status) => { let err = ForwardTcpConnectionError::from_status(status); @@ -2334,7 +2333,27 @@ async fn forward_one_tcp_connection( } }; - let (mut local_read, mut local_write) = socket.into_split(); + let (local_read, local_write) = socket.into_split(); + relay_local_socket(local_read, local_write, tx, response).await +} + +/// Relay bytes between a local socket and a forward stream. +/// +/// When the target closes first, half-close the local socket but keep sending +/// client data until the client closes, so a target that only half-closes +/// still receives the rest of the request. +async fn relay_local_socket( + mut local_read: R, + mut local_write: W, + tx: tokio::sync::mpsc::Sender, + mut response: S, +) -> std::result::Result<(), ForwardTcpConnectionError> +where + R: tokio::io::AsyncRead + Unpin + Send + 'static, + W: tokio::io::AsyncWrite + Unpin, + S: tokio_stream::Stream> + Unpin, +{ + use tokio::io::{AsyncReadExt, AsyncWriteExt}; let to_gateway = tokio::spawn(async move { let mut buf = vec![0u8; 64 * 1024]; @@ -2359,8 +2378,9 @@ async fn forward_one_tcp_connection( }); while let Some(frame) = response - .message() + .next() .await + .transpose() .map_err(ForwardTcpConnectionError::from_status)? { let Some(openshell_core::proto::tcp_forward_frame::Payload::Data(data)) = frame.payload @@ -2377,7 +2397,7 @@ async fn forward_one_tcp_connection( } let _ = local_write.shutdown().await; - to_gateway.abort(); + let _ = to_gateway.await; Ok(()) } @@ -6693,15 +6713,16 @@ fn format_endpoint(endpoint: &openshell_core::proto::NetworkEndpoint) -> String #[cfg(test)] mod tests { use super::{ - PolicyGetView, ProvisioningStep, build_sandbox_resource_limits, format_endpoint, - format_log_line, git_sync_files, has_main_process_result, parse_cli_setting_value, - parse_credential_expiry_cli_value, parse_driver_config_json, + ForwardTcpConnectionError, PolicyGetView, ProvisioningStep, build_sandbox_resource_limits, + format_endpoint, format_log_line, git_sync_files, has_main_process_result, + parse_cli_setting_value, parse_credential_expiry_cli_value, parse_driver_config_json, parse_secret_material_env_pairs, policy_revision_list_json, policy_revision_to_json, proto_execution_timeout, provisioning_timeout_message, ready_false_condition_message, - resolve_from, rootfs_tar_sources_supported_for_gateway, sandbox_should_persist, - sandbox_upload_plan, service_endpoint_to_json, service_expose_status_error, - service_url_for_gateway, workspace_member_to_json, + relay_local_socket, resolve_from, rootfs_tar_sources_supported_for_gateway, + sandbox_should_persist, sandbox_upload_plan, service_endpoint_to_json, + service_expose_status_error, service_url_for_gateway, workspace_member_to_json, }; + use openshell_core::proto::TcpForwardFrame; #[test] fn draft_approval_error_explains_refreshed_evaluation() { @@ -8525,6 +8546,87 @@ mod tests { assert!(format_log_line(&log).ends_with(message)); } + fn forward_data(bytes: &[u8]) -> TcpForwardFrame { + TcpForwardFrame { + payload: Some(openshell_core::proto::tcp_forward_frame::Payload::Data( + bytes.to_vec(), + )), + } + } + + struct Relay { + client: tokio::io::DuplexStream, + to_gateway: tokio::sync::mpsc::Receiver, + response: tokio::sync::mpsc::Sender>, + task: tokio::task::JoinHandle>, + } + + fn start_relay() -> Relay { + let (client, local) = tokio::io::duplex(4096); + let (local_read, local_write) = tokio::io::split(local); + let (tx, to_gateway) = tokio::sync::mpsc::channel(16); + let (response, resp_rx) = tokio::sync::mpsc::channel(16); + let task = tokio::spawn(relay_local_socket( + local_read, + local_write, + tx, + tokio_stream::wrappers::ReceiverStream::new(resp_rx), + )); + Relay { + client, + to_gateway, + response, + task, + } + } + + #[tokio::test] + async fn relay_keeps_client_upload_after_target_half_close() { + use tokio::io::{AsyncReadExt, AsyncWriteExt}; + + let mut relay = start_relay(); + relay + .response + .send(Ok(forward_data(b"ready"))) + .await + .unwrap(); + drop(relay.response); + + let mut greeting = [0u8; 5]; + relay.client.read_exact(&mut greeting).await.unwrap(); + assert_eq!(&greeting, b"ready"); + + relay.client.write_all(b"upload").await.unwrap(); + relay.client.shutdown().await.unwrap(); + + let mut uploaded = Vec::new(); + while let Some(frame) = relay.to_gateway.recv().await { + if let Some(openshell_core::proto::tcp_forward_frame::Payload::Data(data)) = + frame.payload + { + uploaded.extend(data); + } + } + assert_eq!(uploaded, b"upload"); + relay.task.await.unwrap().unwrap(); + } + + #[tokio::test] + async fn relay_half_closes_local_socket_when_target_closes() { + use tokio::io::AsyncReadExt; + + let mut relay = start_relay(); + relay.response.send(Ok(forward_data(b"bye"))).await.unwrap(); + drop(relay.response); + + let mut received = Vec::new(); + relay.client.read_to_end(&mut received).await.unwrap(); + assert_eq!(received, b"bye"); + + drop(relay.client); + relay.task.await.unwrap().unwrap(); + } + use std::io::Write as _; use std::time::{Duration, Instant}; diff --git a/crates/openshell-supervisor-process/src/supervisor_session.rs b/crates/openshell-supervisor-process/src/supervisor_session.rs index 5be01017eb..607ddee204 100644 --- a/crates/openshell-supervisor-process/src/supervisor_session.rs +++ b/crates/openshell-supervisor-process/src/supervisor_session.rs @@ -716,21 +716,54 @@ async fn handle_relay_open( /// Forward the relay's data frames without interpreting the target protocol. async fn bridge_relay( target: Box, - mut inbound: impl tokio_stream::Stream> + Unpin, + inbound: impl tokio_stream::Stream> + Unpin, out_tx: mpsc::Sender, channel_id: String, terminating: Arc, ) -> Result<(), Box> { // Connect to the local SSH daemon on its Unix socket. - let (mut target_r, mut target_w) = tokio::io::split(target); + let (target_r, target_w) = tokio::io::split(target); debug!( channel_id = %channel_id, "relay bridge: connected to local target" ); + bridge_relay_bytes( + &channel_id, + target_r, + target_w, + out_tx, + inbound, + &terminating, + ) + .await +} + +/// Bridge bytes between a local target socket and an inbound `RelayFrame` +/// stream, sending target bytes out through `out_tx`. +/// +/// `out_tx` is moved into the target-reading task rather than cloned. A +/// clone would let the sender-side task's copy be dropped on target EOF +/// while this function's own copy stayed alive until `inbound` also ended, +/// which keeps the outbound gRPC stream open indefinitely after the target +/// closes. Moving it in means the outbound stream (and therefore the +/// client's view of the connection) closes as soon as the target does, +/// regardless of whether the client side has sent anything else. +async fn bridge_relay_bytes( + channel_id: &str, + mut target_r: impl AsyncRead + Unpin + Send + 'static, + mut target_w: impl AsyncWrite + Unpin, + out_tx: mpsc::Sender, + mut inbound: S, + terminating: &AtomicBool, +) -> Result<(), Box> +where + S: tokio_stream::Stream> + Unpin, +{ // Target → gRPC (out_tx): read local target, forward as `RelayFrame::data`. - let out_tx_writer = out_tx.clone(); + // `out_tx` is owned by this task, so dropping it on target EOF ends the + // outbound stream immediately, without waiting on the inbound side. let target_to_grpc = tokio::spawn(async move { let mut buf = vec![0u8; RELAY_CHUNK_SIZE]; loop { @@ -742,7 +775,7 @@ async fn bridge_relay( buf[..n].to_vec(), )), }; - if out_tx_writer.send(chunk).await.is_err() { + if out_tx.send(chunk).await.is_err() { break; } } @@ -769,7 +802,7 @@ async fn bridge_relay( } } Err(e) => { - if expected_transport_close_during_shutdown(&e, &terminating) { + if expected_transport_close_during_shutdown(&e, terminating) { debug!( channel_id = %channel_id, error = %e, @@ -785,10 +818,6 @@ async fn bridge_relay( // Half-close the target socket's write side so the service sees EOF. let _ = target_w.shutdown().await; - - // Dropping out_tx closes the outbound gRPC stream, letting the gateway - // observe EOF on its side too. - drop(out_tx); let _ = target_to_grpc.await; if let Some(e) = inbound_err { @@ -1233,4 +1262,136 @@ mod ocsf_event_tests { assert!(err.to_string().contains("peer PID mismatch")); accept_task.await.unwrap(); } + + /// Regression test for #3724: when the target closes after sending + /// data, the outbound relay stream must close too, even though the + /// inbound (client) side is still open. Before the fix, `out_tx` was + /// cloned into the target-reading task, so the function's own copy kept + /// the outbound stream alive until `inbound` also ended. + #[tokio::test] + async fn bridge_closes_outbound_when_target_closes_first() { + let (target, mut remote) = tokio::io::duplex(4096); + let (target_r, target_w) = tokio::io::split(target); + + let (out_tx, mut out_rx) = mpsc::channel::(16); + + // Inbound stream the client never closes during this test. + let (inbound_tx, inbound_rx) = mpsc::channel::>(16); + let inbound = tokio_stream::wrappers::ReceiverStream::new(inbound_rx); + + let terminating = AtomicBool::new(false); + let bridge = tokio::spawn(async move { + bridge_relay_bytes("chan-1", target_r, target_w, out_tx, inbound, &terminating).await + }); + + remote.write_all(b"hello").await.unwrap(); + remote.shutdown().await.unwrap(); + + let frame = out_rx.recv().await.expect("data frame expected"); + assert_eq!( + frame.payload, + Some(openshell_core::proto::relay_frame::Payload::Data( + b"hello".to_vec() + )) + ); + + // The outbound stream must end here, without the inbound side (still + // held open by `inbound_tx`) ending first. + assert!( + out_rx.recv().await.is_none(), + "outbound stream should close once the target closes" + ); + + drop(inbound_tx); + bridge + .await + .unwrap() + .expect("bridge should finish cleanly when target closes first"); + } + + /// A target that half-closes its output must still receive client data. + #[tokio::test] + async fn bridge_forwards_client_data_after_target_half_close() { + let (target, mut remote) = tokio::io::duplex(4096); + let (target_r, target_w) = tokio::io::split(target); + + let (out_tx, mut out_rx) = mpsc::channel::(16); + let (inbound_tx, inbound_rx) = mpsc::channel::>(16); + let inbound = tokio_stream::wrappers::ReceiverStream::new(inbound_rx); + + let terminating = AtomicBool::new(false); + let bridge = tokio::spawn(async move { + bridge_relay_bytes("chan-3", target_r, target_w, out_tx, inbound, &terminating).await + }); + + remote.write_all(b"ready").await.unwrap(); + remote.shutdown().await.unwrap(); + assert!(out_rx.recv().await.is_some(), "greeting frame expected"); + assert!(out_rx.recv().await.is_none(), "outbound should close"); + + inbound_tx + .send(Ok(RelayFrame { + payload: Some(openshell_core::proto::relay_frame::Payload::Data( + b"upload".to_vec(), + )), + })) + .await + .unwrap(); + let mut buf = [0u8; 6]; + remote.read_exact(&mut buf).await.unwrap(); + assert_eq!(&buf, b"upload"); + + drop(inbound_tx); + bridge.await.unwrap().expect("bridge should finish cleanly"); + } + + /// A well-behaved round trip: bytes flow both directions and the bridge + /// ends cleanly when the client closes its side. + #[tokio::test] + async fn bridge_round_trips_bytes_until_client_closes() { + let (target, mut remote) = tokio::io::duplex(4096); + let (target_r, target_w) = tokio::io::split(target); + + let (out_tx, mut out_rx) = mpsc::channel::(16); + let (inbound_tx, inbound_rx) = mpsc::channel::>(16); + let inbound = tokio_stream::wrappers::ReceiverStream::new(inbound_rx); + + let terminating = AtomicBool::new(false); + let bridge = tokio::spawn(async move { + bridge_relay_bytes("chan-2", target_r, target_w, out_tx, inbound, &terminating).await + }); + + // Client -> target. + inbound_tx + .send(Ok(RelayFrame { + payload: Some(openshell_core::proto::relay_frame::Payload::Data( + b"ping".to_vec(), + )), + })) + .await + .unwrap(); + let mut buf = [0u8; 4]; + remote.read_exact(&mut buf).await.unwrap(); + assert_eq!(&buf, b"ping"); + + // Target -> client. + remote.write_all(b"pong").await.unwrap(); + let frame = out_rx.recv().await.expect("data frame expected"); + assert_eq!( + frame.payload, + Some(openshell_core::proto::relay_frame::Payload::Data( + b"pong".to_vec() + )) + ); + + // Client closes its side first; the bridge should still complete + // once the target also closes. + drop(inbound_tx); + remote.shutdown().await.unwrap(); + + bridge + .await + .unwrap() + .expect("bridge should finish cleanly on an ordinary round trip"); + } } From a2429fcdcdf3b6e80f185317d3910fd8f84055f6 Mon Sep 17 00:00:00 2001 From: Thota Shashank Date: Sat, 3 Oct 2026 20:32:54 +0000 Subject: [PATCH 11/13] fix(tui): preserve quoted post-create command arguments (#4137) Signed-off-by: Thota shashank --- .agents/skills/tui-development/SKILL.md | 2 + Cargo.lock | 1 + Cargo.toml | 1 + crates/openshell-tui/Cargo.toml | 1 + crates/openshell-tui/src/app.rs | 91 ++++++++++++++++++++---- crates/openshell-tui/src/lib.rs | 62 +++++++++++++--- docs/how-it-works/sandboxes/overview.mdx | 2 + 7 files changed, 137 insertions(+), 23 deletions(-) diff --git a/.agents/skills/tui-development/SKILL.md b/.agents/skills/tui-development/SKILL.md index b0d546be0b..ee946adf22 100644 --- a/.agents/skills/tui-development/SKILL.md +++ b/.agents/skills/tui-development/SKILL.md @@ -330,6 +330,8 @@ TUI actions should parallel `openshell` CLI commands so users have familiar ment When adding new TUI features, check what the CLI offers and maintain consistency. +The create form parses the optional Command field as shell words before starting creation. Preserve the parsed argument vector through post-create execution and shell-escape each argument at the SSH boundary. Quoting groups arguments; expansions and operators remain literal unless the user explicitly invokes a shell such as `sh -c`. Invalid quoting must leave the form open with an error and must not queue sandbox creation. + ### Scrollable views follow k9s conventions Any scrollable content (logs, future long lists) should follow the k9s autoscroll pattern: diff --git a/Cargo.lock b/Cargo.lock index 180617e40d..a36b8cd4a5 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -5165,6 +5165,7 @@ dependencies = [ "owo-colors", "ratatui", "serde", + "shell-words", "terminal-colorsaurus", "tokio", "tonic", diff --git a/Cargo.toml b/Cargo.toml index 754a72c3f6..98e323f2ea 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -128,6 +128,7 @@ tokio-stream = "0.1" protoc-bin-vendored = "3.2.0" url = "2" indexmap = "2" +shell-words = "1.1.1" # Database sqlx = { version = "0.9", default-features = false, features = ["runtime-tokio", "tls-rustls-aws-lc-rs", "postgres", "sqlite", "migrate", "macros"] } diff --git a/crates/openshell-tui/Cargo.toml b/crates/openshell-tui/Cargo.toml index bc79d114b2..68d6e11c93 100644 --- a/crates/openshell-tui/Cargo.toml +++ b/crates/openshell-tui/Cargo.toml @@ -28,6 +28,7 @@ owo-colors = { workspace = true } serde = { workspace = true } url = { workspace = true } indexmap = { workspace = true } +shell-words = { workspace = true } [lints] workspace = true diff --git a/crates/openshell-tui/src/app.rs b/crates/openshell-tui/src/app.rs index 2cbd7b2946..f4cf7bc05e 100644 --- a/crates/openshell-tui/src/app.rs +++ b/crates/openshell-tui/src/app.rs @@ -241,9 +241,8 @@ impl GatewayEntry { // --------------------------------------------------------------------------- /// Data extracted from the create sandbox form: -/// `(name, image, command, selected_provider_names, forward_specs)`. +/// `(name, image, selected_provider_names, forward_specs)`. pub type CreateFormData = ( - String, String, String, Vec, @@ -723,8 +722,8 @@ pub struct App { pub pending_create_sandbox: bool, /// Forward specs to apply after sandbox creation completes. pub pending_forward_ports: Vec, - /// Command to exec via SSH after sandbox creation completes. - pub pending_exec_command: String, + /// Parsed arguments to exec via SSH after sandbox creation completes. + pub pending_exec_command: Vec, /// Animation ticker handle — aborted when animation stops. pub anim_handle: Option>, @@ -1078,7 +1077,7 @@ impl App { create_form: None, pending_create_sandbox: false, pending_forward_ports: Vec::new(), - pending_exec_command: String::new(), + pending_exec_command: Vec::new(), anim_handle: None, sandbox_log_lines: Vec::new(), sandbox_log_scroll: 0, @@ -2396,6 +2395,13 @@ impl App { } CreateFormField::Submit => { if key.code == KeyCode::Enter { + match shell_words::split(&form.command) { + Ok(command) => self.pending_exec_command = command, + Err(error) => { + form.status = Some(format!("Invalid command: {error}")); + return; + } + } form.anim_start = Some(Instant::now()); form.status = None; form.phase = CreatePhase::Creating; @@ -2408,7 +2414,7 @@ impl App { } /// Build the form data needed for the gRPC `CreateSandbox` request. - /// Returns `(name, image, command, selected_provider_names, forward_ports)`. + /// Returns `(name, image, selected_provider_names, forward_ports)`. pub fn create_form_data(&self) -> Option { let form = self.create_form.as_ref()?; let providers: Vec = form @@ -2428,13 +2434,7 @@ impl App { openshell_core::forward::ForwardSpec::parse(s).ok() }) .collect(); - Some(( - form.name.clone(), - form.image.clone(), - form.command.clone(), - providers, - ports, - )) + Some((form.name.clone(), form.image.clone(), providers, ports)) } // ------------------------------------------------------------------ @@ -3653,6 +3653,71 @@ mod tests { ) } + #[tokio::test] + async fn create_command_preserves_quoted_and_escaped_arguments() { + let cases: &[(&str, &[&str])] = &[ + ( + r#"/bin/sh -c "echo GOOD; read x""#, + &["/bin/sh", "-c", "echo GOOD; read x"], + ), + ( + r#"echo 'hello world' "" a\ b "it's""#, + &["echo", "hello world", "", "a b", "it's"], + ), + (r#"echo pre"fix value"post"#, &["echo", "prefix valuepost"]), + ( + "echo $HOME $(id) ; | > *.txt", + &["echo", "$HOME", "$(id)", ";", "|", ">", "*.txt"], + ), + ("echo hello", &["echo", "hello"]), + ("", &[]), + (" \t", &[]), + ]; + for (command, expected) in cases { + let mut app = test_app(); + app.create_form = Some(CreateSandboxForm { + command: (*command).into(), + focused_field: CreateFormField::Submit, + ..Default::default() + }); + app.handle_create_form_key(KeyEvent::new(KeyCode::Enter, KeyModifiers::NONE)); + assert!(app.pending_create_sandbox, "{command}"); + assert_eq!(app.pending_exec_command, *expected, "{command}"); + let form = app.create_form.as_ref().unwrap(); + assert_eq!(form.phase, CreatePhase::Creating); + assert!(form.status.is_none()); + } + } + + #[tokio::test] + async fn malformed_create_command_stays_in_form_until_corrected() { + for command in ["echo \"unfinished", "echo 'unfinished", "echo \"trailing\\"] { + let mut app = test_app(); + app.create_form = Some(CreateSandboxForm { + command: command.into(), + focused_field: CreateFormField::Submit, + ..Default::default() + }); + app.handle_create_form_key(KeyEvent::new(KeyCode::Enter, KeyModifiers::NONE)); + assert!(!app.pending_create_sandbox, "{command}"); + assert!(app.pending_exec_command.is_empty()); + let form = app.create_form.as_mut().unwrap(); + assert_eq!(form.phase, CreatePhase::Form); + assert!(form.anim_start.is_none()); + assert!( + form.status + .as_ref() + .unwrap() + .starts_with("Invalid command:") + ); + form.command = "echo corrected".into(); + app.handle_create_form_key(KeyEvent::new(KeyCode::Enter, KeyModifiers::NONE)); + assert!(app.pending_create_sandbox); + assert_eq!(app.pending_exec_command, ["echo", "corrected"]); + assert!(app.create_form.as_ref().unwrap().status.is_none()); + } + } + fn provider_profile( id: &str, credentials: Vec, diff --git a/crates/openshell-tui/src/lib.rs b/crates/openshell-tui/src/lib.rs index f235c4e003..2265329bcf 100644 --- a/crates/openshell-tui/src/lib.rs +++ b/crates/openshell-tui/src/lib.rs @@ -1083,7 +1083,7 @@ async fn handle_exec_command( terminal: &mut Terminal>, events: &mut EventHandler, sandbox_name: &str, - command: &str, + command: &[String], workspace: &str, ) -> Result<()> { let session = { @@ -1135,13 +1135,9 @@ async fn handle_exec_command( ); // Step 3: Build SSH command — same flags as handle_shell_connect but with - // the user's command appended. Each word is escaped individually so the + // the parsed command appended. Each argument is escaped individually so the // remote shell parses it correctly. - let command_str = command - .split_whitespace() - .map(shell_escape) - .collect::>() - .join(" "); + let command_str = build_exec_command(command); let mut ssh = std::process::Command::new("ssh"); ssh.arg("-o") .arg(format!("ProxyCommand={proxy_command}")) @@ -1217,6 +1213,14 @@ use openshell_core::forward::{ validate_ssh_session_response, }; +fn build_exec_command(command: &[String]) -> String { + command + .iter() + .map(|arg| shell_escape(arg)) + .collect::>() + .join(" ") +} + /// Convert a `SandboxPolicy` proto into styled ratatui lines for the policy viewer. fn render_policy_lines( policy: &openshell_core::proto::SandboxPolicy, @@ -1400,12 +1404,10 @@ fn start_anim_ticker(app: &mut App, tx: mpsc::UnboundedSender) { fn spawn_create_sandbox(app: &mut App, tx: mpsc::UnboundedSender) { let mut client = app.client.clone(); - let Some((name, image, command, selected_providers, ports)) = app.create_form_data() else { + let Some((name, image, selected_providers, ports)) = app.create_form_data() else { return; }; - // Stash command so we can exec after sandbox creation + Ready. - app.pending_exec_command = command; // Stash ports so we can include them in the status text. app.pending_forward_ports.clone_from(&ports); @@ -3130,6 +3132,46 @@ fn format_age(epoch_ms: i64) -> String { } } +#[cfg(all(test, unix))] +mod exec_command_tests { + use super::build_exec_command; + use std::io::Write; + use std::process::{Command, Stdio}; + + fn run_command(command: &str, stdin: &[u8]) -> std::process::Output { + let args = shell_words::split(command).unwrap(); + let mut child = Command::new("/bin/sh") + .args(["-c", &build_exec_command(&args)]) + .stdin(Stdio::piped()) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()) + .spawn() + .unwrap(); + child.stdin.take().unwrap().write_all(stdin).unwrap(); + child.wait_with_output().unwrap() + } + + #[test] + fn quoted_script_runs_and_reads_stdin() { + let output = run_command(r#"/bin/sh -c "echo GOOD; read x; echo $x""#, b"entered\n"); + assert!(output.status.success(), "{:?}", output.stderr); + assert_eq!(output.stdout, b"GOOD\nentered\n"); + } + + #[test] + fn arguments_remain_literal_at_remote_shell_boundary() { + let output = run_command( + r#"printf '%s\n' 'hello world' "" "it's" a\ b $HOME '$(echo BAD)' ';' '|' '>' '*.txt'"#, + b"", + ); + assert!(output.status.success(), "{:?}", output.stderr); + assert_eq!( + output.stdout, + b"hello world\n\nit's\na b\n$HOME\n$(echo BAD)\n;\n|\n>\n*.txt\n" + ); + } +} + #[cfg(test)] mod draft_approve_all_message_tests { use super::*; diff --git a/docs/how-it-works/sandboxes/overview.mdx b/docs/how-it-works/sandboxes/overview.mdx index e38f925327..b084590eee 100644 --- a/docs/how-it-works/sandboxes/overview.mdx +++ b/docs/how-it-works/sandboxes/overview.mdx @@ -763,6 +763,8 @@ The dashboard has three panels stacked vertically: Gateways, Providers (or Globa The sandbox table’s NOTES column shows `Invalid config` when policy or provider configuration blocks provisioning. Open the sandbox detail view for the full rejection reason, or run `openshell sandbox get -o json`. The note clears after the configuration is repaired; active port forwards remain listed. +Press `c` in the Sandboxes panel to create a sandbox. In the optional Command field, use quotes or backslashes to group arguments containing spaces. For example, `/bin/sh -c "echo GOOD; read x"` prints `GOOD` and waits for Enter. Invalid quoting keeps the form open without creating a sandbox. The field does not expand variables or evaluate shell operators; invoke a shell explicitly with `sh -c` when you need that behavior. + ## Port Forwarding Forward a local port to a running sandbox to access services inside it, such as a web server or database: From 8983642e280e7b0f2ed423edbffd705ef259e7c0 Mon Sep 17 00:00:00 2001 From: Shiju Date: Sat, 3 Oct 2026 20:48:18 +0000 Subject: [PATCH 12/13] fix(gateway): give image preparation its own deadline (#4038) * feat(server): separate image preparation and admission deadlines Give sandbox image preparation its own deadline and start the admission deadline after preparation finishes. Persist both phases across gateway restarts and show recovery guidance for the phase that expired. Preserve the current service authorization schema and regenerate the Go bindings with the preparation timestamps. Fixes #3952 Related to #3955 Signed-off-by: Shiju * fix(cli): simplify preparation timeout fallback selection Use lazy Option fallbacks while preserving timeout messages and retained sandbox behavior. Signed-off-by: Shiju * docs(server): clarify admission timer prerequisites Signed-off-by: Shiju * fix(compute): enforce deadlines during initial sandbox create Release stalled create operations after preparation expires and preserve the timeout diagnosis across late driver results. Keep failed-create cleanup bound to its original attempt so another replica can retry safely. Signed-off-by: Shiju * test(server): satisfy deadline regression lints Drop the create-error mutex guard before matching its cloned value and use idiomatic iteration and timeout matching in the deadline fixtures. Signed-off-by: Shiju * fix(server): retain ownership of pending provisioning operations Keep submitted create and start requests alive after caller cancellation, monitor failure, or preparation timeout. Persist request ownership before dispatch and retain staged uploads while the driver response is pending. Require timeout cleanup to stop compute after driver settlement without discarding an active cleanup claim. Fence late result handling against newer operations, preserve failed-start recovery, and defer automatic restart while another request owns the sandbox. Expose pending ownership in CLI JSON. Add ordered multi-replica and cancellation regressions for the review findings. Signed-off-by: Shiju * fix(openshell): preserve tracing and accept ready create responses Carry the request span into the detached provisioning worker so compute driver calls remain attached to their parent trace after task handoff. Update the compensation regression to require no backend DELETE when the durable cleanup claim fails, matching the operation ownership requirement. Accept the gateway's current Ready snapshot when a sandbox becomes ready before CREATE returns. Do not require the client to observe an earlier provisioning phase. Cover the Ready-only watch and command attachment. Signed-off-by: Shiju --------- Signed-off-by: Shiju --- crates/openshell-cli/src/run.rs | 182 +- .../sandbox_create_lifecycle_integration.rs | 78 +- crates/openshell-core/src/config.rs | 5 + crates/openshell-server/src/cli.rs | 49 + crates/openshell-server/src/compute/mod.rs | 2576 ++++++++++++++--- .../src/compute/provisioning_deadline.rs | 237 +- .../src/compute/provisioning_operation.rs | 313 ++ .../src/compute/rootfs_tar.rs | 65 + crates/openshell-server/src/config_file.rs | 3 + crates/openshell-server/src/grpc/policy.rs | 147 +- crates/openshell-server/src/grpc/sandbox.rs | 7 +- crates/openshell-server/src/lib.rs | 3 + crates/openshell-server/src/storage_proto.rs | 10 +- crates/openshell-tui/src/lib.rs | 27 +- docs/how-it-works/gateways/configuration.mdx | 4 + .../how-it-works/policies/manage-policies.mdx | 7 +- docs/how-it-works/sandboxes/overview.mdx | 26 +- proto/openshell.proto | 22 +- sdk/go/proto/openshellv1/openshell.pb.go | 412 +-- 19 files changed, 3569 insertions(+), 604 deletions(-) create mode 100644 crates/openshell-server/src/compute/provisioning_operation.rs diff --git a/crates/openshell-cli/src/run.rs b/crates/openshell-cli/src/run.rs index 289bfa8bb5..afd5872a45 100644 --- a/crates/openshell-cli/src/run.rs +++ b/crates/openshell-cli/src/run.rs @@ -800,11 +800,8 @@ pub async fn sandbox_create( // Non-interactive mode: track start time for timestamps. let provision_start = Instant::now(); - // Don't use stop_on_terminal on the server — the Kubernetes CRD may - // briefly report a stale Ready status before the controller reconciles - // a newly created sandbox. Instead we handle termination client-side: - // we wait until we have observed at least one non-Ready phase followed - // by Ready (a genuine Provisioning → Ready transition). + // Handle terminal states here so a provisional container exit can wait + // for the supervisor's canonical-process result before cleanup. let sandbox_name = sandbox.object_name().to_string(); let sandbox_workspace = sandbox.object_workspace().to_string(); let mut stream = client @@ -832,8 +829,6 @@ pub async fn sandbox_create( let mut last_sandbox = sandbox.clone(); let mut last_error_reason = String::new(); let mut last_condition_message = ready_false_condition_message(sandbox.status.as_ref()); - // Track whether we have seen a non-Ready phase during the watch. - let mut saw_non_ready = SandboxPhase::try_from(sandbox.phase()) != Ok(SandboxPhase::Ready); let provision_timeout = Duration::from_secs( std::env::var("OPENSHELL_PROVISION_TIMEOUT") .ok() @@ -911,10 +906,6 @@ pub async fn sandbox_create( last_condition_message = Some(message); } - if phase != SandboxPhase::Ready { - saw_non_ready = true; - } - let main_process_result = has_main_process_result(&s); if matches!( phase, @@ -949,9 +940,10 @@ pub async fn sandbox_create( break; } - // Only accept Ready as terminal after we've observed a - // non-Ready phase, proving the controller has reconciled. - if saw_non_ready && phase == SandboxPhase::Ready { + // The gateway owns readiness. Its initial watch snapshot may + // already be Ready if provisioning finished before CREATE + // returned; requiring an earlier phase would miss that state. + if phase == SandboxPhase::Ready { if let Some(d) = display.as_interactive_mut() { d.clear(); } @@ -1218,25 +1210,31 @@ pub async fn sandbox_create( SandboxPhase::Error => { drop(stream); drop(client); - let provisioning_timed_out = last_sandbox + let timed_out_provisioning = last_sandbox .status .as_ref() .and_then(|status| status.provisioning.as_ref()) - .is_some_and(|record| record.timeout_time.is_some()); - let create_result = if provisioning_timed_out { - Err(miette::miette!( - "{last_error_reason}\nSandbox '{sandbox_name}' was retained. Inspect it with `openshell sandbox get {sandbox_name}`; repair its configuration, then run `openshell sandbox start {sandbox_name}` after cleanup completes." - )) - } else if last_error_reason.is_empty() { - Err(miette::miette!( - "sandbox entered error phase while provisioning" - )) - } else { - Err(miette::miette!( - "sandbox entered error phase while provisioning: {}", - last_error_reason - )) - }; + .filter(|record| record.timeout_time.is_some()); + let create_result = timed_out_provisioning.map_or_else( + || { + if last_error_reason.is_empty() { + Err(miette::miette!( + "sandbox entered error phase while provisioning" + )) + } else { + Err(miette::miette!( + "sandbox entered error phase while provisioning: {}", + last_error_reason + )) + } + }, + |record| { + Err(miette::miette!( + "{}", + retained_sandbox_timeout_message(&sandbox_name, &last_error_reason, record) + )) + }, + ); finalize_sandbox_create_session( &effective_server, &sandbox_name, @@ -1259,6 +1257,24 @@ pub async fn sandbox_create( } } +/// Use the persisted phase to select recovery guidance. Preparation may expire +/// before any policy is evaluated, so it must not tell the user to repair policy. +fn retained_sandbox_timeout_message( + sandbox_name: &str, + error_reason: &str, + record: &openshell_core::proto::SandboxProvisioning, +) -> String { + let recovery = if record.preparation_deadline.is_some() && record.admission_start_time.is_none() + { + "check image preparation and supervisor startup diagnostics and the gateway's `image_preparation_timeout_seconds` budget" + } else { + "repair its configuration" + }; + format!( + "{error_reason}\nSandbox '{sandbox_name}' was retained. Inspect it with `openshell sandbox get {sandbox_name}`; {recovery}, then run `openshell sandbox start {sandbox_name}` after cleanup completes." + ) +} + /// Resolved source for the `--from` flag on `sandbox create`. #[derive(Debug)] enum ResolvedSource { @@ -2924,11 +2940,22 @@ fn sandbox_to_json(sandbox: &Sandbox) -> serde_json::Value { "configuration_change_id": record.configuration_change_id, "configuration_change_time": record.configuration_change_time.as_ref().map(ToString::to_string), "first_rejection_time": record.first_rejection_time.as_ref().map(ToString::to_string), + "phase": if record.deadline.is_none() && record.timeout_time.is_none() { + "ready" + } else if record.preparation_deadline.is_some() && record.admission_start_time.is_none() { + "preparation" + } else { + "admission" + }, + "preparation_deadline": record.preparation_deadline.as_ref().map(ToString::to_string), + "admission_start_time": record.admission_start_time.as_ref().map(ToString::to_string), "deadline": record.deadline.as_ref().map(ToString::to_string), "timeout_time": record.timeout_time.as_ref().map(ToString::to_string), "cleanup_completed_time": record.cleanup_completed_time.as_ref().map(ToString::to_string), "cleanup_error": record.cleanup_error, "cleanup_retry_time": record.cleanup_retry_time.as_ref().map(ToString::to_string), + "driver_operation_pending": record.driver_operation_pending, + "driver_operation_id": record.driver_operation_id, })); serde_json::json!({ "id": sandbox.object_id(), @@ -8131,6 +8158,101 @@ mod tests { ); } + #[test] + fn retained_sandbox_timeout_message_matches_expired_phase() { + let mut record = openshell_core::proto::SandboxProvisioning { + preparation_deadline: openshell_core::time::timestamp_from_millis(1_800_000).ok(), + timeout_time: openshell_core::time::timestamp_from_millis(1_800_000).ok(), + ..Default::default() + }; + let message = super::retained_sandbox_timeout_message( + "cold-image", + "ImagePreparationTimedOut: preparation expired", + &record, + ); + assert!(message.starts_with("ImagePreparationTimedOut: preparation expired\n")); + assert!(message.contains("Sandbox 'cold-image' was retained")); + assert!(message.contains("image preparation and supervisor startup diagnostics")); + assert!(message.contains("image_preparation_timeout_seconds")); + assert!(!message.contains("repair its configuration")); + assert!(message.contains("openshell sandbox get cold-image")); + assert!(message.contains("openshell sandbox start cold-image` after cleanup completes")); + + // An admission timeout retains preparation timestamps. Its completed + // transition must select configuration repair rather than a larger budget. + record.admission_start_time = openshell_core::time::timestamp_from_millis(600_000).ok(); + let admission_without_preparation = openshell_core::proto::SandboxProvisioning { + timeout_time: record.timeout_time, + ..Default::default() + }; + for admission_record in [&record, &admission_without_preparation] { + let message = super::retained_sandbox_timeout_message( + "invalid-policy", + "ProvisioningTimedOut: repair window expired", + admission_record, + ); + assert!(message.contains("repair its configuration")); + assert!(!message.contains("image_preparation_timeout_seconds")); + assert!( + message.contains("openshell sandbox start invalid-policy` after cleanup completes") + ); + } + } + + #[test] + fn provisioning_json_exposes_pending_driver_operation() { + for pending in [true, false] { + let mut sandbox = Sandbox::default(); + sandbox.set_phase(SandboxPhase::Provisioning.into()); + sandbox.status.as_mut().unwrap().provisioning = + Some(openshell_core::proto::SandboxProvisioning { + driver_operation_pending: pending, + driver_operation_id: "operation-1".into(), + ..Default::default() + }); + assert_eq!( + super::sandbox_to_json(&sandbox)["provisioning"]["driver_operation_pending"], + pending + ); + assert_eq!( + super::sandbox_to_json(&sandbox)["provisioning"]["driver_operation_id"], + "operation-1" + ); + } + } + + #[test] + fn provisioning_json_distinguishes_preparation_and_admission() { + let mut sandbox = Sandbox::default(); + sandbox.set_phase(SandboxPhase::Provisioning.into()); + let ceiling = openshell_core::time::timestamp_from_millis(1_800_000).ok(); + sandbox.status.as_mut().unwrap().provisioning = + Some(openshell_core::proto::SandboxProvisioning { + preparation_deadline: ceiling, + deadline: ceiling, + ..Default::default() + }); + let json = super::sandbox_to_json(&sandbox); + assert_eq!(json["provisioning"]["phase"], "preparation"); + assert_eq!( + json["provisioning"]["preparation_deadline"], + "1970-01-01T00:30:00Z" + ); + assert!(json["provisioning"]["admission_start_time"].is_null()); + sandbox + .status + .as_mut() + .unwrap() + .provisioning + .as_mut() + .unwrap() + .admission_start_time = openshell_core::time::timestamp_from_millis(600_000).ok(); + assert_eq!( + super::sandbox_to_json(&sandbox)["provisioning"]["phase"], + "admission" + ); + } + #[test] fn sandbox_json_exposes_repair_diagnostic_and_accepted_generation() { use openshell_core::proto::{ConfigurationAdmissionState, SandboxConfigurationAdmission}; diff --git a/crates/openshell-cli/tests/sandbox_create_lifecycle_integration.rs b/crates/openshell-cli/tests/sandbox_create_lifecycle_integration.rs index e21f747112..6723a25bc7 100644 --- a/crates/openshell-cli/tests/sandbox_create_lifecycle_integration.rs +++ b/crates/openshell-cli/tests/sandbox_create_lifecycle_integration.rs @@ -70,6 +70,7 @@ struct SandboxState { vm_error_with_observed_exit: Arc, vm_slow_progress_before_ready: Arc, vm_log_churn_before_ready: Arc, + ready_before_create_returns: Arc, terminal_before_relay: Arc, terminal_after_provisional_container_exit: Arc, provisional_container_exit_without_result: Arc, @@ -194,7 +195,17 @@ impl OpenShell for TestOpenShell { }), ..Sandbox::default() }; - sandbox.set_phase(SandboxPhase::Provisioning as i32); + sandbox.set_phase( + if self + .state + .ready_before_create_returns + .load(Ordering::SeqCst) + { + SandboxPhase::Ready as i32 + } else { + SandboxPhase::Provisioning as i32 + }, + ); Ok(Response::new(SandboxResponse { sandbox: Some(sandbox), service_urls, @@ -677,6 +688,10 @@ impl OpenShell for TestOpenShell { .vm_slow_progress_before_ready .load(Ordering::SeqCst); let vm_log_churn_before_ready = self.state.vm_log_churn_before_ready.load(Ordering::SeqCst); + let ready_before_create_returns = self + .state + .ready_before_create_returns + .load(Ordering::SeqCst); let terminal_before_relay = self.state.terminal_before_relay.load(Ordering::SeqCst); let terminal_after_provisional_container_exit = self .state @@ -723,6 +738,18 @@ impl OpenShell for TestOpenShell { } let mut ready = provisioning.clone(); ready.set_phase(SandboxPhase::Ready as i32); + if ready_before_create_returns { + // A watch starts with the current snapshot. Keep it open so + // stream closure cannot hide a client that ignores Ready. + let _ = tx + .send(Ok(SandboxStreamEvent { + payload: Some(sandbox_stream_event::Payload::Sandbox(ready)), + cursor: String::new(), + })) + .await; + tx.closed().await; + return; + } let mut completed = provisioning.clone(); completed.status = Some(SandboxStatus { exit_code: Some(0), @@ -2372,6 +2399,55 @@ async fn sandbox_create_preserves_vm_error_when_exit_code_is_observed() { assert!(rendered.contains("ProcessExited: VM process exited with status 0")); } +#[tokio::test] +async fn sandbox_create_accepts_ready_before_create_returns() { + let server = run_server().await; + server + .openshell + .state + .ready_before_create_returns + .store(true, Ordering::SeqCst); + let fake_ssh_dir = tempfile::tempdir().unwrap(); + let xdg_dir = tempfile::tempdir().unwrap(); + let _env = test_env_with( + &fake_ssh_dir, + &xdg_dir, + &[("OPENSHELL_PROVISION_TIMEOUT", "1".to_string())], + ); + let tls = test_tls(&server); + install_fake_ssh(&fake_ssh_dir); + + let exit_code = tokio::time::timeout( + Duration::from_secs(10), + run::sandbox_create( + &server.endpoint, + "openshell", + run::SandboxCreateConfig { + name: Some("already-ready"), + command: &["echo".into(), "OK".into()], + ..test_config() + }, + "default", + &tls, + ), + ) + .await + .expect("creation must finish while the watch remains open") + .expect("an already-Ready sandbox must not wait for a new provisioning transition"); + + assert_eq!(exit_code, 0); + assert_eq!(create_requests(&server).await.len(), 1); + assert_eq!( + server + .openshell + .state + .ssh_session_requests + .load(Ordering::SeqCst), + 1, + "the initial Ready snapshot must allow the command to attach" + ); +} + #[tokio::test] async fn sandbox_create_keeps_waiting_while_vm_progress_arrives() { let server = run_server().await; diff --git a/crates/openshell-core/src/config.rs b/crates/openshell-core/src/config.rs index 820b4a9476..35785f20f9 100644 --- a/crates/openshell-core/src/config.rs +++ b/crates/openshell-core/src/config.rs @@ -240,6 +240,10 @@ pub struct Config { /// TTL for SSH session tokens, in seconds. 0 disables expiry. pub ssh_session_ttl_secs: u64, + /// Absolute image preparation and initial supervisor startup budget for new + /// sandbox attempts, in seconds. Must be between 1 and 86400, inclusive. + pub image_preparation_timeout_seconds: u32, + /// Maximum gRPC requests allowed per rate-limit window. /// /// When paired with [`Self::grpc_rate_limit_window_secs`], positive values @@ -865,6 +869,7 @@ impl Config { credential_drivers: Vec::new(), default_credential_driver: None, ssh_session_ttl_secs: default_ssh_session_ttl_secs(), + image_preparation_timeout_seconds: 1800, grpc_rate_limit_requests: None, grpc_rate_limit_window_secs: None, service_routing: ServiceRoutingConfig::default(), diff --git a/crates/openshell-server/src/cli.rs b/crates/openshell-server/src/cli.rs index 93d98ed105..bdba1e32c0 100644 --- a/crates/openshell-server/src/cli.rs +++ b/crates/openshell-server/src/cli.rs @@ -556,6 +556,18 @@ fn prepare_server_config_with_drivers( config.policy_validation_failure_mode = mode; } + if let Some(seconds) = file + .as_ref() + .and_then(|f| f.openshell.gateway.image_preparation_timeout_seconds) + { + if !(1..=86_400).contains(&seconds) { + return Err(miette::miette!( + "image_preparation_timeout_seconds must be between 1 and 86400" + )); + } + config.image_preparation_timeout_seconds = seconds; + } + if let Some(issuer) = args.oidc_issuer.clone() { config = config.with_oidc(openshell_core::OidcConfig { issuer, @@ -3307,6 +3319,7 @@ version = 2 [openshell.gateway] policy_validation_failure_mode = "retain_last_valid" +image_preparation_timeout_seconds = 2400 [openshell.drivers.docker] unknown_docker_key = true @@ -3332,6 +3345,7 @@ mem_mib = "not-a-number" super::prepare_server_config(&mut args, &matches).expect("server config is prepared"); assert_eq!(prepared.config.compute_driver.as_deref(), Some("podman")); + assert_eq!(prepared.config.image_preparation_timeout_seconds, 2400); assert_eq!( prepared.config.policy_validation_failure_mode, openshell_core::PolicyValidationFailureMode::RetainLastValid @@ -3340,4 +3354,39 @@ mem_mib = "not-a-number" assert!(file.openshell.drivers.contains_key("docker")); assert!(file.openshell.drivers.contains_key("vm")); } + + #[test] + fn server_config_rejects_unbounded_image_preparation() { + let _lock = ENV_LOCK + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner); + let state = tempfile::tempdir().unwrap(); + let tls = tempfile::tempdir().unwrap(); + let _state = EnvVarGuard::set("XDG_STATE_HOME", state.path().to_str().unwrap()); + let _tls = EnvVarGuard::set("OPENSHELL_LOCAL_TLS_DIR", tls.path().to_str().unwrap()); + let config_path = state.path().join("gateway.toml"); + for seconds in [0, 86_401] { + std::fs::write(&config_path, format!( + "[openshell]\nversion = 2\n[openshell.gateway]\nimage_preparation_timeout_seconds = {seconds}\n" + )).unwrap(); + let (mut args, matches) = parse_with_args(&[ + "openshell-gateway", + "--config", + config_path.to_str().unwrap(), + "--db-url", + "sqlite::memory:", + "--compute-driver", + "podman", + "--disable-tls", + ]); + let Err(error) = super::prepare_server_config(&mut args, &matches) else { + panic!("unbounded preparation must be rejected"); + }; + assert!( + error + .to_string() + .contains("image_preparation_timeout_seconds must be between 1 and 86400") + ); + } + } } diff --git a/crates/openshell-server/src/compute/mod.rs b/crates/openshell-server/src/compute/mod.rs index 088231f6e1..52a15c6300 100644 --- a/crates/openshell-server/src/compute/mod.rs +++ b/crates/openshell-server/src/compute/mod.rs @@ -6,6 +6,7 @@ pub mod driver_config; pub mod lease; pub mod provisioning_deadline; +mod provisioning_operation; pub mod rootfs_tar; use crate::grpc::policy::SANDBOX_SETTINGS_OBJECT_TYPE; @@ -73,6 +74,7 @@ pub type DriverWatchStream = pub type SharedComputeDriver = Arc + Send + Sync>; +use provisioning_operation::ProvisioningOperationError; use traced_driver::TracedDriver; const LIFECYCLE_SWEEP_PAGE_SIZE: u32 = 1000; @@ -156,6 +158,23 @@ mod traced_driver { .await } + /// Keep a submitted call alive independently of the request handler. + /// The owned future retains the driver, arguments, and tracing scope. + pub(super) fn call_owned( + &self, + rpc: openshell_otel::ComputeDriverRpc, + sandbox_id: Option<&str>, + call: impl FnOnce(SharedComputeDriver) -> Fut + Send + 'static, + ) -> impl Future> + Send + 'static + where + T: Send + 'static, + Fut: Future> + Send + 'static, + { + let driver = self.clone(); + let sandbox_id = sandbox_id.map(str::to_owned); + async move { driver.call(rpc, sandbox_id.as_deref(), call).await } + } + /// Open a driver watch while keeping the client span alive with the stream. pub(super) async fn watch(&self) -> Result, Status> { let span = self.span(openshell_otel::rpc::WATCH_SANDBOXES, None); @@ -644,6 +663,7 @@ pub struct ComputeRuntime { telemetry_compute_driver: TelemetryComputeDriver, driver_process: Option>, default_image: String, + image_preparation_timeout_seconds: u32, store: Arc, sandbox_index: SandboxIndex, sandbox_watch_bus: SandboxWatchBus, @@ -745,6 +765,7 @@ impl ComputeRuntime { telemetry_compute_driver: TelemetryComputeDriver::custom(), driver_process, default_image, + image_preparation_timeout_seconds: 1800, store, sandbox_index, sandbox_watch_bus, @@ -826,6 +847,16 @@ impl ComputeRuntime { &self.default_image } + /// Validate the budget once at startup. Persisted attempts keep their + /// original deadline even if the operator changes this value on restart. + pub(crate) fn with_image_preparation_timeout(mut self, seconds: u32) -> Result { + if !(1..=86_400).contains(&seconds) { + return Err("image_preparation_timeout_seconds must be between 1 and 86400".into()); + } + self.image_preparation_timeout_seconds = seconds; + Ok(self) + } + #[must_use] pub fn driver_info_snapshots(&self) -> &[ComputeDriverInfoSnapshot] { std::slice::from_ref(&self.driver_info) @@ -1109,40 +1140,65 @@ impl ComputeRuntime { spec.await_main_process_attachment = await_main_process_attachment; spec.launch_authentication = launch_authentication.unwrap_or_default(); } - match self - .driver - .call( + let result = Box::pin(self.await_provisioning_operation( + &sandbox, + self.driver.call_owned( openshell_otel::rpc::CREATE_SANDBOX, Some(sandbox.object_id()), - |driver| async move { - driver + move |driver| async move { + // The owned task keeps this input through monitor or + // caller cancellation. A crash leaves its sweep marker. + if let Some(staged) = staged.as_mut() { + staged.prepare_dispatch().await?; + staged.disarm(); + } + let response = driver .create_sandbox(Request::new(CreateSandboxRequest { sandbox: Some(driver_sandbox), })) - .await + .await; + if let Some(staged) = staged.as_mut() { + staged.finish_driver_operation(response.is_ok()).await; + } + response }, - ) + ), + )) + .await; + let global_guard = self.lock_global_for_lifecycle(&lifecycle_guard).await; + // Result ownership comes from settlement, never from a newer row read + // after the driver returned. Recovery may reuse the same attempt. + let owned = match &result { + Ok((_, settled)) | Err(ProvisioningOperationError::Driver { settled, .. }) => settled, + Err( + ProvisioningOperationError::Monitor(status) + | ProvisioningOperationError::Unsettled(status), + ) => return Err(status.clone()), + }; + // The scanner can expire preparation while create owns the lifecycle + // gate. Every driver outcome must observe that durable decision before + // deleting records, publishing status, or compensating a failed create. + let current = self + .store + .get_message::(&sandbox_id) .await - { - Ok(response) => { + .map_err(|error| Status::internal(format!("fetch created sandbox failed: {error}")))? + .ok_or_else(|| Status::not_found("sandbox removed during create"))?; + provisioning_operation::ensure_current_result(¤t, owned)?; + sandbox = current; + if provisioning_deadline::timed_out(&sandbox) { + return Err(Status::deadline_exceeded( + "image preparation deadline expired", + )); + } + match result { + Ok((response, _)) => { let runtime_identity = response.into_inner().runtime_identity; - // The driver now owns the staged archive and removes the - // request directory once it has built the disk. - if let Some(staged) = staged.as_mut() { - staged.disarm(); - } - let global_guard = self.lock_global_for_lifecycle(&lifecycle_guard).await; if self.supports_sandbox_authentication() && runtime_identity.is_empty() { let status = Status::internal("compute driver did not return a runtime identity"); return Err(self - .compensate_failed_create( - &sandbox_id, - sandbox.object_name(), - lifecycle_guard, - global_guard, - status, - ) + .compensate_failed_create(&sandbox, lifecycle_guard, global_guard, status) .await); } if self.supports_sandbox_authentication() { @@ -1152,7 +1208,13 @@ impl ComputeRuntime { &sandbox, self.configured_driver_name(), &runtime_identity, - &[SandboxPhase::Provisioning, SandboxPhase::Ready], + // A main-process exit can schedule automatic + // restart before this owned CREATE settles. + &[ + SandboxPhase::Provisioning, + SandboxPhase::Ready, + SandboxPhase::Starting, + ], ) .await; sandbox = match persisted { @@ -1163,8 +1225,7 @@ impl ComputeRuntime { )); return Err(self .compensate_failed_create( - &sandbox_id, - sandbox.object_name(), + &sandbox, lifecycle_guard, global_guard, status, @@ -1177,46 +1238,72 @@ impl ComputeRuntime { self.sandbox_watch_bus.notify(sandbox.object_id()); Ok(sandbox) } - Err(status) if status.code() == Code::AlreadyExists => { - let _ = self - .store - .delete(Sandbox::object_type(), sandbox.object_id()) - .await; - self.sandbox_index.remove_sandbox(sandbox.object_id()); - Err(Status::already_exists("sandbox already exists")) - } - Err(status) if status.code() == Code::FailedPrecondition => { - let _ = self - .store - .delete(Sandbox::object_type(), sandbox.object_id()) - .await; - self.sandbox_index.remove_sandbox(sandbox.object_id()); - Err(Status::failed_precondition(status.message().to_string())) + Err( + ProvisioningOperationError::Monitor(status) + | ProvisioningOperationError::Unsettled(status), + ) => { + // A monitor failure says nothing about the submitted create. + // Preserve its record even if the owner has since returned. + Err(status) } - Err(err) => { - let _ = self + Err(ProvisioningOperationError::Driver { status, .. }) => { + // Another replica can expire this attempt after our read. + // Remove only the version inspected above, never a newer row. + match self .store - .delete(Sandbox::object_type(), sandbox.object_id()) - .await; - self.sandbox_index.remove_sandbox(sandbox.object_id()); - Err(Status::internal(format!( - "create sandbox failed: {}", - err.message() - ))) + .delete_if( + Sandbox::object_type(), + &sandbox_id, + sandbox_resource_version(&sandbox), + ) + .await + { + Ok(_) => self.sandbox_index.remove_sandbox(&sandbox_id), + Err(crate::persistence::PersistenceError::Conflict { .. }) => { + if let Some(current) = self + .store + .get_message::(&sandbox_id) + .await + .map_err(|error| Status::internal(error.to_string()))? + && provisioning_deadline::timed_out(¤t) + { + return Err(Status::deadline_exceeded( + "image preparation deadline expired", + )); + } + return Err(Status::aborted( + "sandbox changed during failed create cleanup", + )); + } + Err(error) => { + return Err(Status::internal(format!("clean up failed create: {error}"))); + } + } + match status.code() { + Code::AlreadyExists => Err(Status::already_exists("sandbox already exists")), + Code::FailedPrecondition => { + Err(Status::failed_precondition(status.message().to_string())) + } + _ => Err(Status::internal(format!( + "create sandbox failed: {}", + status.message() + ))), + } } } } async fn compensate_failed_create( &self, - sandbox_id: &str, - sandbox_name: &str, + created: &Sandbox, lifecycle_guard: SandboxLifecycleGuard, global_guard: tokio::sync::OwnedMutexGuard<()>, original: Status, ) -> Status { + let sandbox_id = created.object_id(); + let sandbox_name = created.object_name(); let transition = match self - .begin_sandbox_delete_with_initial_snapshot(sandbox_id, None) + .begin_sandbox_delete_with_initial_snapshot(sandbox_id, None, Some(created)) .await { Ok(BeginDelete::Started(transition)) => *transition, @@ -1229,25 +1316,24 @@ impl ComputeRuntime { ), ); } + Err(error) + if matches!( + error.code(), + Code::DeadlineExceeded | Code::Aborted | Code::FailedPrecondition + ) => + { + return error; + } Err(error) => { - drop(global_guard); - let delete_result = self - .delete_backend_after_failed_create(sandbox_id, sandbox_name) - .await; - let cleanup_detail = match delete_result { - Ok(_) => String::new(), - Err(delete_error) => format!( - "; best-effort backend cleanup also failed: {}", - delete_error.message() - ), - }; + // A failed store read or CAS cannot establish cleanup + // ownership. Keep the record for authoritative cleanup; an + // unclaimed DELETE could destroy a newer same-ID runtime. return Status::new( original.code(), format!( - "{}; cleanup after successful create could not claim the sandbox record: {}{}", + "{}; cleanup after successful create could not claim the sandbox record: {}", original.message(), - error.message(), - cleanup_detail + error.message() ), ); } @@ -1350,6 +1436,7 @@ impl ComputeRuntime { )); } + provisioning_operation::ensure_operation_settled(¤t)?; let phase = SandboxPhase::try_from(current.phase()).unwrap_or(SandboxPhase::Unknown); if matches!(phase, SandboxPhase::Stopped | SandboxPhase::Completed) || is_failed_main_process_result(¤t) @@ -1506,6 +1593,7 @@ impl ComputeRuntime { .await .map_err(|e| Status::internal(format!("fetch sandbox failed: {e}")))? .ok_or_else(|| Status::not_found("sandbox not found"))?; + provisioning_operation::ensure_operation_settled(&candidate)?; if provisioning_deadline::timed_out(&candidate) && candidate .status @@ -1542,6 +1630,7 @@ impl ComputeRuntime { let mut attempts = 0; let (previous, starting, launch_authentication) = loop { + provisioning_operation::ensure_operation_settled(¤t)?; let phase = SandboxPhase::try_from(current.phase()).unwrap_or(SandboxPhase::Unknown); if phase == SandboxPhase::Ready { return Ok(current); @@ -1618,6 +1707,7 @@ impl ComputeRuntime { SandboxPhase::Starting, "Starting", "Sandbox start requested", + self.image_preparation_timeout_seconds, ); }, ) @@ -1682,30 +1772,6 @@ impl ComputeRuntime { })? } - /// Release the lifecycle gate after the durable deadline expires, allowing - /// cleanup to cancel partial startup. Configuration repairs can extend this - /// deadline, so a fixed timeout around the driver RPC would expire too soon. - async fn await_provisioning_operation( - &self, - starting: &Sandbox, - operation: impl std::future::Future>, - ) -> Result { - tokio::pin!(operation); - loop { - tokio::select! { - result = &mut operation => return result, - () = tokio::time::sleep(Duration::from_secs(1)) => { - let current = self.store.get_message::(starting.object_id()) - .await.map_err(|error| Status::internal(error.to_string()))? - .ok_or_else(|| Status::not_found("sandbox removed during startup"))?; - if provisioning_deadline::timed_out(¤t) { - return Err(Status::deadline_exceeded("provisioning repair window expired")); - } - } - } - } - } - async fn complete_sandbox_start( &self, sandbox_id: String, @@ -1725,71 +1791,68 @@ impl ComputeRuntime { .map_err(Status::failed_precondition)? .into_string(); let expected_runtime_identity = sandbox_compute_runtime_identity(&previous); - let authentication_for_recreate = launch_authentication.clone(); - let mut result = self - .await_provisioning_operation( - &starting, - self.driver.call( - openshell_otel::rpc::START_SANDBOX, - Some(&sandbox_id), - |driver| { - let sandbox_id = sandbox_id.clone(); - let sandbox_name = sandbox_name.clone(); - async move { - driver - .start_sandbox(Request::new(StartSandboxRequest { - sandbox_id, - name: sandbox_name, - launch_authentication, - generation_id, - expected_runtime_identity, - })) - .await - } - }, - ), - ) - .await; - - if provisioning_deadline::timed_out(&previous) - && matches!(&result, Err(error) if error.code() == Code::NotFound) - { - // A partial provisioning attempt may have been canceled before a - // restartable backend object existed. Keep the API identity/spec. + // One owned operation covers both calls. A NotFound response alone + // does not end ownership while the recovery create can still run. + let recreate = if provisioning_deadline::timed_out(&previous) { let mut driver_sandbox = driver_sandbox_from_public(&starting, &self.driver_info.name) .map_err(|status| *status)?; if let Some(spec) = driver_sandbox.spec.as_mut() { - spec.launch_authentication = authentication_for_recreate; + spec.launch_authentication + .clone_from(&launch_authentication); } - result = self - .await_provisioning_operation( - &starting, - self.driver.call( - openshell_otel::rpc::CREATE_SANDBOX, - Some(&sandbox_id), - |driver| async move { - driver - .create_sandbox(Request::new(CreateSandboxRequest { - sandbox: Some(driver_sandbox), - })) - .await - .map(|response| { - tonic::Response::new( - openshell_core::proto::compute::v1::StartSandboxResponse { - runtime_identity: response - .into_inner() - .runtime_identity, - }, - ) - }) - }, - ), + Some(driver_sandbox) + } else { + None + }; + let driver = self.driver.clone(); + let operation_id = sandbox_id.clone(); + let result = Box::pin(self.await_provisioning_operation(&starting, async move { + let request_id = operation_id.clone(); + let response = driver + .call_owned( + openshell_otel::rpc::START_SANDBOX, + Some(&operation_id), + move |driver| async move { + driver + .start_sandbox(Request::new(StartSandboxRequest { + sandbox_id: request_id, + name: sandbox_name, + launch_authentication, + generation_id, + expected_runtime_identity, + })) + .await + }, ) .await; - } + match (response, recreate) { + (Err(error), Some(driver_sandbox)) if error.code() == Code::NotFound => { + driver + .call_owned( + openshell_otel::rpc::CREATE_SANDBOX, + Some(&operation_id), + move |driver| async move { + use openshell_core::proto::compute::v1::StartSandboxResponse; + let response = driver + .create_sandbox(Request::new(CreateSandboxRequest { + sandbox: Some(driver_sandbox), + })) + .await?; + Ok(tonic::Response::new(StartSandboxResponse { + runtime_identity: response.into_inner().runtime_identity, + })) + }, + ) + .await + } + (response, _) => response, + } + })) + .await; match result { - Ok(response) => { + Ok((response, starting)) => { + provisioning_operation::ensure_current_result(&starting, &starting)?; let runtime_identity = response.into_inner().runtime_identity; if self.supports_sandbox_authentication() && runtime_identity.is_empty() { let status = @@ -1832,18 +1895,29 @@ impl ComputeRuntime { } } } else { - self.store + let latest = self + .store .get_message::(&sandbox_id) .await .map_err(|e| Status::internal(format!("fetch sandbox failed: {e}")))? - .ok_or_else(|| Status::not_found("sandbox not found"))? + .ok_or_else(|| Status::not_found("sandbox not found"))?; + provisioning_operation::ensure_current_result(&latest, &starting)?; + latest }; self.sandbox_index.update_from_sandbox(&latest); self.sandbox_watch_bus.notify(&sandbox_id); Ok(latest) } - Err(err) => { - self.recover_failed_lifecycle(&lifecycle_guard, &starting, &previous, false) + Err( + ProvisioningOperationError::Monitor(error) + | ProvisioningOperationError::Unsettled(error), + ) => Err(error), + Err(ProvisioningOperationError::Driver { + status: err, + settled, + }) => { + provisioning_operation::ensure_current_result(&settled, &settled)?; + self.recover_failed_lifecycle(&lifecycle_guard, &settled, &previous, false) .await; Err(Status::new( err.code(), @@ -1861,6 +1935,8 @@ impl ComputeRuntime { runtime_identity: &str, allowed_phases: &[SandboxPhase], ) -> Result { + provisioning_operation::ensure_current_result(starting, starting) + .map_err(|error| error.to_string())?; let expected_generation = sandbox_runtime_generation(starting)?; let mut expected_resource_version = sandbox_resource_version(starting); @@ -1901,7 +1977,10 @@ impl ComputeRuntime { let current_generation = sandbox_runtime_generation(¤t)?; let phase = SandboxPhase::try_from(current.phase()).unwrap_or(SandboxPhase::Unknown); - if current_generation != expected_generation || !allowed_phases.contains(&phase) + if current_generation != expected_generation + || provisioning_operation::ensure_current_result(¤t, starting) + .is_err() + || !allowed_phases.contains(&phase) { return Err(format!( "sandbox changed lifecycle ownership while persisting runtime identity (phase: {phase:?})" @@ -1929,40 +2008,51 @@ impl ComputeRuntime { previous: &Sandbox, original: Status, ) -> Status { - let sandbox_id = starting.object_id(); - let sandbox_name = starting.object_name(); + let sandbox_id = starting.object_id().to_string(); + let sandbox_name = starting.object_name().to_string(); + let request_id = sandbox_id.clone(); + // Claim compensation against the settled START operation before STOP. + // A newer operation or timeout must reject it without touching compute. let stop_result = self - .driver - .call( - openshell_otel::rpc::STOP_SANDBOX, - Some(sandbox_id), - |driver| { - let sandbox_id = sandbox_id.to_string(); - let sandbox_name = sandbox_name.to_string(); - async move { - driver - .stop_sandbox(Request::new(StopSandboxRequest { - sandbox_id, - name: sandbox_name, - })) - .await - } - }, + .await_provisioning_operation( + starting, + self.driver.call_owned( + openshell_otel::rpc::STOP_SANDBOX, + Some(&sandbox_id), + move |driver| { + let sandbox_id = request_id; + async move { + driver + .stop_sandbox(Request::new(StopSandboxRequest { + sandbox_id, + name: sandbox_name, + })) + .await + } + }, + ), ) .await; - if let Err(error) = stop_result { - return Status::new( - original.code(), - format!( - "{}; rollback after successful start failed: {}", - original.message(), - error.message() - ), - ); - } + let settled = match stop_result { + Ok((_, settled)) => settled, + Err( + error @ (ProvisioningOperationError::Monitor(_) + | ProvisioningOperationError::Unsettled(_)), + ) => return error.into(), + Err(error) => { + return Status::new( + original.code(), + format!( + "{}; rollback after successful start failed: {}", + original.message(), + Status::from(error).message() + ), + ); + } + }; let _global_guard = self.lock_global_for_lifecycle(lifecycle_guard).await; - if self.restore_lifecycle_snapshot(starting, previous).await { + if self.restore_lifecycle_snapshot(&settled, previous).await { original } else { Status::new( @@ -2104,7 +2194,13 @@ impl ComputeRuntime { &sandbox_id, expected_resource_version, move |sandbox| { - apply_lifecycle_phase(sandbox, phase, &reason, &message); + apply_lifecycle_phase( + sandbox, + phase, + &reason, + &message, + self.image_preparation_timeout_seconds, + ); }, ) .await @@ -2112,8 +2208,23 @@ impl ComputeRuntime { } async fn restore_lifecycle_snapshot(&self, owned: &Sandbox, previous: &Sandbox) -> bool { + if provisioning_deadline::timed_out(owned) + || provisioning_deadline::driver_operation_pending(owned) + { + return false; + } let sandbox_id = owned.object_id().to_string(); - let previous = previous.clone(); + let mut previous = previous.clone(); + // Restore lifecycle state, not historical driver ownership. In + // particular a failed retry must not resurrect its prior timed-out + // attempt or reauthorize callbacks using an old operation ID. + previous + .status + .get_or_insert_with(Default::default) + .provisioning = owned + .status + .as_ref() + .and_then(|status| status.provisioning.clone()); match self .store .update_message_cas::( @@ -2246,7 +2357,7 @@ impl ComputeRuntime { // `Deleting` row used to fence recovery, and the prior row used only // for exact-version rollback after an ambiguous driver failure. let transition = match self - .begin_sandbox_delete_with_initial_snapshot(&target.sandbox_id, Some(current)) + .begin_sandbox_delete_with_initial_snapshot(&target.sandbox_id, Some(current), None) .await? { BeginDelete::AlreadyDeleting => { @@ -2325,6 +2436,7 @@ impl ComputeRuntime { &self, sandbox_id: &str, mut initial_snapshot: Option, + failed_create: Option<&Sandbox>, ) -> Result { let operation = "set sandbox phase to Deleting"; @@ -2338,6 +2450,14 @@ impl ComputeRuntime { .map_err(|e| Status::internal(format!("fetch sandbox failed: {e}")))? .ok_or_else(|| Status::not_found("sandbox not found"))?, }; + provisioning_operation::ensure_operation_settled(&sandbox)?; + + // Failed-create compensation owns only its original attempt. + // Preserve its timeout diagnosis and any replacement attempt on + // every CAS retry. Explicit deletion requires a settled operation. + if let Some(created) = failed_create { + provisioning_operation::ensure_current_result(&sandbox, created)?; + } if SandboxPhase::try_from(sandbox.phase()).unwrap_or(SandboxPhase::Unknown) == SandboxPhase::Deleting @@ -2485,6 +2605,9 @@ impl ComputeRuntime { } let sandbox = decode_sandbox_record(&record)?; + if provisioning_deadline::driver_operation_pending(&sandbox) { + return Ok(false); + } self.cleanup_sandbox_owned_records(&sandbox).await?; match self @@ -2997,6 +3120,13 @@ impl ComputeRuntime { } }; + if provisioning_deadline::driver_operation_pending(&sandbox) { + warn!( + sandbox_id, + "Retaining pending driver operation during gateway recovery" + ); + continue; + } let phase = SandboxPhase::try_from(sandbox.phase()).unwrap_or(SandboxPhase::Unknown); let recoverable_error = phase == SandboxPhase::Error && is_recoverable_error_reason(&sandbox); @@ -3057,34 +3187,36 @@ impl ComputeRuntime { } }; let expected_runtime_identity = sandbox_compute_runtime_identity(&sandbox); - match self - .await_provisioning_operation( - &sandbox, - self.driver.call( - openshell_otel::rpc::START_SANDBOX, - Some(&sandbox_id), - |driver| { - let sandbox_id = sandbox_id.clone(); - let sandbox_name = sandbox_name.clone(); - let launch_authentication = launch_authentication.clone(); - let expected_runtime_identity = expected_runtime_identity.clone(); - async move { - driver - .start_sandbox(Request::new(StartSandboxRequest { - sandbox_id, - name: sandbox_name, - launch_authentication, - generation_id, - expected_runtime_identity, - })) - .await - } - }, - ), - ) - .await + let request_id = sandbox_id.clone(); + let request_name = sandbox_name.clone(); + match Box::pin(self.await_provisioning_operation( + &sandbox, + self.driver.call_owned( + openshell_otel::rpc::START_SANDBOX, + Some(&sandbox_id), + move |driver| { + let sandbox_id = request_id; + let sandbox_name = request_name; + let launch_authentication = launch_authentication.clone(); + let expected_runtime_identity = expected_runtime_identity.clone(); + async move { + driver + .start_sandbox(Request::new(StartSandboxRequest { + sandbox_id, + name: sandbox_name, + launch_authentication, + generation_id, + expected_runtime_identity, + })) + .await + } + }, + ), + )) + .await { - Ok(response) => { + Ok((response, settled)) => { + let sandbox = settled; let mut recovered_sandbox = sandbox.clone(); if self.supports_sandbox_authentication() { let runtime_identity = response.into_inner().runtime_identity; @@ -3114,7 +3246,7 @@ impl ComputeRuntime { ) .await { - Ok(updated) => recovered_sandbox = updated, + Ok(updated) => *recovered_sandbox = updated, Err(error) => { warn!( sandbox_id = %sandbox.object_id(), @@ -3151,7 +3283,18 @@ impl ComputeRuntime { recovered += 1; } } - Err(err) if err.code() == Code::NotFound => { + Err( + ProvisioningOperationError::Monitor(error) + | ProvisioningOperationError::Unsettled(error), + ) => { + warn!(sandbox_id, %error, "Gateway recovery stopped monitoring an owned driver operation"); + failed += 1; + } + Err(ProvisioningOperationError::Driver { + status: err, + settled, + }) if err.code() == Code::NotFound => { + let sandbox = settled; authentication_failed(sandbox.object_id()); // Backend resource is gone but the store still // remembers the sandbox. Mark Error so the UI @@ -3173,7 +3316,11 @@ impl ComputeRuntime { } missing += 1; } - Err(err) => { + Err(ProvisioningOperationError::Driver { + status: err, + settled, + }) => { + let sandbox = settled; authentication_failed(sandbox.object_id()); warn!( sandbox_id = %sandbox.object_id(), @@ -3227,6 +3374,13 @@ impl ComputeRuntime { continue; } }; + if provisioning_deadline::driver_operation_pending(&sandbox) { + warn!( + sandbox_id, + "Retaining pending driver operation during gateway recovery" + ); + continue; + } let phase = SandboxPhase::try_from(sandbox.phase()).unwrap_or(SandboxPhase::Unknown); match phase { SandboxPhase::Stopped | SandboxPhase::Completed => { @@ -3312,22 +3466,27 @@ impl ComputeRuntime { } }; let expected_runtime_identity = sandbox_compute_runtime_identity(&sandbox); + // Recovery retains the original attempt and deadline. A + // stalled driver must release this lifecycle gate after + // expiry so the deadline worker can reclaim its compute. if let Err(err) = self - .driver - .call( - openshell_otel::rpc::START_SANDBOX, - Some(&sandbox_id), - |driver| async move { - driver - .start_sandbox(Request::new(StartSandboxRequest { - sandbox_id: driver_sandbox_id, - name: sandbox_name, - launch_authentication: Vec::new(), - generation_id, - expected_runtime_identity, - })) - .await - }, + .await_provisioning_operation( + &sandbox, + self.driver.call_owned( + openshell_otel::rpc::START_SANDBOX, + Some(&sandbox_id), + |driver| async move { + driver + .start_sandbox(Request::new(StartSandboxRequest { + sandbox_id: driver_sandbox_id, + name: sandbox_name, + launch_authentication: Vec::new(), + generation_id, + expected_runtime_identity, + })) + .await + }, + ), ) .await { @@ -3356,7 +3515,10 @@ impl ComputeRuntime { match self .store .update_message_cas::(&sandbox_id, 0, |s| { - if provisioning_deadline::timed_out(s) { + if provisioning_deadline::timed_out(s) + || provisioning_deadline::driver_operation_pending(s) + || !provisioning_operation::same_operation(s, sandbox) + { return; } s.set_phase(SandboxPhase::Error as i32); @@ -3668,6 +3830,14 @@ impl ComputeRuntime { } async fn restart_sandbox_runtime(&self, sandbox_id: &str) -> Result<(), String> { + use openshell_core::proto::compute::v1::StartSandboxResponse; + + enum RestartOutcome { + Started(StartSandboxResponse), + Superseded, + DriverError(&'static str, Status), + } + let lifecycle_guard = self.lifecycle_gates.lock_for(sandbox_id).await; let global_guard = self.lock_global_for_lifecycle(&lifecycle_guard).await; let Some(current) = self @@ -3679,12 +3849,18 @@ impl ComputeRuntime { return Ok(()); }; let phase = SandboxPhase::try_from(current.phase()).unwrap_or(SandboxPhase::Unknown); - let now_ms = openshell_core::time::now_ms(); let due = current .status .as_ref() - .is_some_and(|status| restart_is_due(status, now_ms)); - if phase != SandboxPhase::Starting || !is_automatic_restart_transition(¤t) || !due { + .is_some_and(|status| restart_is_due(status, openshell_core::time::now_ms())); + if phase != SandboxPhase::Starting + || !is_automatic_restart_transition(¤t) + || !due + || provisioning_deadline::driver_operation_pending(¤t) + || provisioning_deadline::timed_out(¤t) + { + // Main-process exit may schedule restart while CREATE is pending. + // Keep that schedule intact so settlement makes it eligible again. return Ok(()); } @@ -3702,10 +3878,13 @@ impl ComputeRuntime { .transpose() .map_err(|status| status.to_string())? }; - let claimed = if already_claimed { + let claimed = if already_claimed && sandbox_provisioning_attempt_id(¤t).is_none() { current.clone() } else { - self.store + // Claim the restart schedule and driver ownership in one CAS. A + // pending check followed by an unclaimed STOP races another replica. + match self + .store .update_message_cas::( sandbox_id, sandbox_resource_version(¤t), @@ -3715,10 +3894,8 @@ impl ComputeRuntime { { identity.write(&mut metadata.annotations); } + provisioning_operation::claim_record(sandbox); let status = sandbox.status.get_or_insert_with(Default::default); - // Zero durably claims this due attempt across gateway - // replicas. The readiness watchdog starts only after the - // old runtime has stopped. set_next_restart_at_ms(status, 0); upsert_ready_condition( &mut sandbox.status, @@ -3735,77 +3912,97 @@ impl ComputeRuntime { }, ) .await - .map_err(|err| err.to_string())? + { + Ok(claimed) => claimed, + Err(crate::persistence::PersistenceError::Conflict { .. }) => return Ok(()), + Err(error) => return Err(error.to_string()), + } }; self.sandbox_index.update_from_sandbox(&claimed); self.sandbox_watch_bus.notify(sandbox_id); drop(global_guard); - let stop_result = self - .driver - .call( - openshell_otel::rpc::STOP_SANDBOX, - Some(sandbox_id), - |driver| { - let sandbox_id = sandbox_id.to_string(); - let sandbox_name = sandbox_name.clone(); - async move { - driver - .stop_sandbox(Request::new(StopSandboxRequest { - sandbox_id, - name: sandbox_name, - })) - .await - } - }, - ) - .await; - if let Err(status) = stop_result { - return self - .record_restart_driver_failure(&lifecycle_guard, sandbox_id, "stop", status) - .await; - } - - self.cleanup_stopped_sandbox_sessions(&claimed).await?; - - let Some(armed) = self.arm_restart_readiness_watchdog(&claimed).await? else { - // A concurrent stop or delete changed durable intent while the - // old runtime was stopping. Do not recreate it. - return Ok(()); - }; - - let launch_authentication = serialize_persisted_launch_authentication(authority, &armed) - .map_err(|status| status.to_string())?; - let generation_id = sandbox_runtime_generation(&armed)?.into_string(); + let runtime = self.clone(); + let owned = claimed.clone(); + let operation_id = sandbox_id.to_string(); + let operation_name = sandbox_name.clone(); let expected_runtime_identity = sandbox_compute_runtime_identity(¤t); + // One detached worker owns STOP and START. Caller cancellation cannot + // strand a raw START, and the monitor may release the local gate while + // the durable pending flag still protects the unresolved operation. + let result = self + .await_claimed_provisioning_operation(&claimed, async move { + let request_id = operation_id.clone(); + let request_name = operation_name.clone(); + let stop = runtime + .driver + .call_owned( + openshell_otel::rpc::STOP_SANDBOX, + Some(&operation_id), + move |driver| async move { + driver + .stop_sandbox(Request::new(StopSandboxRequest { + sandbox_id: request_id, + name: request_name, + })) + .await + }, + ) + .await; + if let Err(status) = stop { + return Ok(RestartOutcome::DriverError("stop", status)); + } + runtime + .cleanup_stopped_sandbox_sessions(&owned) + .await + .map_err(Status::internal)?; + let Some(armed) = runtime + .arm_restart_readiness_watchdog(&owned) + .await + .map_err(Status::internal)? + else { + return Ok(RestartOutcome::Superseded); + }; + let authority = runtime.restart_authority.get().and_then(Option::as_deref); + let launch_authentication = + serialize_persisted_launch_authentication(authority, &armed)?; + let generation_id = sandbox_runtime_generation(&armed) + .map_err(Status::failed_precondition)? + .into_string(); + let request_id = operation_id.clone(); + let start = runtime + .driver + .call_owned( + openshell_otel::rpc::START_SANDBOX, + Some(&operation_id), + move |driver| async move { + driver + .start_sandbox(Request::new(StartSandboxRequest { + sandbox_id: request_id, + name: operation_name, + launch_authentication, + generation_id, + expected_runtime_identity, + })) + .await + }, + ) + .await; + Ok(match start { + Ok(response) => RestartOutcome::Started(response.into_inner()), + Err(status) => RestartOutcome::DriverError("start", status), + }) + }) + .await + .map_err(|error| error.to_string())?; - let start_result = self - .driver - .call( - openshell_otel::rpc::START_SANDBOX, - Some(sandbox_id), - |driver| { - let sandbox_id = sandbox_id.to_string(); - let sandbox_name = sandbox_name.clone(); - async move { - driver - .start_sandbox(Request::new(StartSandboxRequest { - sandbox_id, - name: sandbox_name, - launch_authentication, - generation_id, - expected_runtime_identity, - })) - .await - } - }, - ) - .await; - let response = match start_result { - Ok(response) => response.into_inner(), - Err(status) => { + let (outcome, settled) = result; + let response = match outcome { + RestartOutcome::Started(response) => response, + RestartOutcome::Superseded => return Ok(()), + RestartOutcome::DriverError(stage, status) => { return self - .record_restart_driver_failure(&lifecycle_guard, sandbox_id, "start", status) + .record_restart_driver_failure(&lifecycle_guard, &settled, stage, status) .await; } }; @@ -3817,16 +4014,14 @@ impl ComputeRuntime { } self.persist_runtime_binding( sandbox_id, - &armed, + &settled, self.configured_driver_name(), &response.runtime_identity, &[SandboxPhase::Starting, SandboxPhase::Ready], ) .await?; } - - self.enforce_lifecycle_after_restart_start(&armed).await?; - + self.enforce_lifecycle_after_restart_start(&settled).await?; info!( sandbox_id, sandbox_name, "Sandbox runtime restarted; waiting for replacement supervisor" @@ -3847,6 +4042,9 @@ impl ComputeRuntime { for _ in 0..DELETE_PHASE_CAS_RETRY_LIMIT { let phase = SandboxPhase::try_from(current.phase()).unwrap_or(SandboxPhase::Unknown); let claim_owned = phase == SandboxPhase::Starting + && provisioning_operation::same_operation(¤t, claimed) + && sandbox_runtime_generation(¤t) == sandbox_runtime_generation(claimed) + && !provisioning_deadline::timed_out(¤t) && is_automatic_restart_transition(¤t) && current .status @@ -3905,6 +4103,16 @@ impl ComputeRuntime { .get_message::(&sandbox_id) .await .map_err(|err| err.to_string())?; + if sandbox_provisioning_attempt_id(armed).is_some() + && latest.as_ref().is_none_or(|current| { + !provisioning_operation::same_operation(current, armed) + || sandbox_runtime_generation(current) != sandbox_runtime_generation(armed) + || provisioning_deadline::timed_out(current) + }) + { + // Newer operations and timeout reclamation own their own cleanup. + return Ok(()); + } let phase = latest.as_ref().map(|sandbox| { SandboxPhase::try_from(sandbox.phase()).unwrap_or(SandboxPhase::Unknown) }); @@ -3975,10 +4183,11 @@ impl ComputeRuntime { async fn record_restart_driver_failure( &self, lifecycle_guard: &SandboxLifecycleGuard, - sandbox_id: &str, + settled: &Sandbox, operation: &str, driver_status: Status, ) -> Result<(), String> { + let sandbox_id = settled.object_id(); let _global_guard = self.lock_global_for_lifecycle(lifecycle_guard).await; let Some(current) = self .store @@ -3988,7 +4197,9 @@ impl ComputeRuntime { else { return Ok(()); }; - if !is_automatic_restart_transition(¤t) { + if !is_automatic_restart_transition(¤t) + || provisioning_operation::ensure_current_result(¤t, settled).is_err() + { return Ok(()); } let terminal = driver_status.code() == Code::NotFound; @@ -4799,8 +5010,9 @@ impl ComputeRuntime { .await .map_err(|e| e.to_string())?; if let Some(sandbox) = sandbox.as_ref() { - if provisioning_deadline::timed_out(sandbox) - && sandbox.phase() == i32::from(SandboxPhase::Error) + if provisioning_deadline::driver_operation_pending(sandbox) + || (provisioning_deadline::timed_out(sandbox) + && sandbox.phase() == i32::from(SandboxPhase::Error)) { return Ok(()); } @@ -4810,8 +5022,16 @@ impl ComputeRuntime { // processed sequentially, so this must not block on the driver // call itself, only on the (instant, non-blocking) decision to // make it. - self.spawn_driver_sandbox_cleanup(sandbox.object_id(), sandbox.object_name()); - self.cleanup_sandbox_owned_records(sandbox).await?; + if self + .remove_sandbox_record_if_version_locked( + sandbox_id, + sandbox_resource_version(sandbox), + ) + .await? + { + self.spawn_driver_sandbox_cleanup(sandbox.object_id(), sandbox.object_name()); + } + return Ok(()); } let _ = self @@ -4828,6 +5048,9 @@ impl ComputeRuntime { sandbox: &Sandbox, expected_resource_version: u64, ) -> Result<(), String> { + if provisioning_deadline::driver_operation_pending(sandbox) { + return Ok(()); + } let sandbox_id = sandbox.object_id(); self.remove_sandbox_record_if_version_locked(sandbox_id, expected_resource_version) .await?; @@ -5143,6 +5366,9 @@ impl ComputeRuntime { } let sandbox = decode_sandbox_record(¤t_record)?; + if provisioning_deadline::driver_operation_pending(&sandbox) { + return Ok(()); + } let age_ms = openshell_core::time::now_ms() .saturating_sub(current_record.created_at_ms) .max(0); @@ -5954,6 +6180,14 @@ fn decode_sandbox_record(record: &ObjectRecord) -> Result { Sandbox::decode(record.payload.as_slice()).map_err(|e| e.to_string()) } +fn sandbox_provisioning_attempt_id(sandbox: &Sandbox) -> Option<&str> { + sandbox + .status + .as_ref() + .and_then(|status| status.provisioning.as_ref()) + .map(|record| record.attempt_id.as_str()) +} + fn sandbox_resource_version(sandbox: &Sandbox) -> u64 { sandbox .metadata @@ -6553,7 +6787,13 @@ fn is_recoverable_error_reason(sandbox: &Sandbox) -> bool { .is_some_and(|c| c.reason == CONDITION_RUNTIME_RESTART || c.reason == CONDITION_STOPPED) } -fn apply_lifecycle_phase(sandbox: &mut Sandbox, phase: SandboxPhase, reason: &str, message: &str) { +fn apply_lifecycle_phase( + sandbox: &mut Sandbox, + phase: SandboxPhase, + reason: &str, + message: &str, + image_preparation_timeout_seconds: u32, +) { sandbox.set_phase(phase as i32); if matches!(phase, SandboxPhase::Stopping | SandboxPhase::Starting) { let status = sandbox.status.get_or_insert_with(Default::default); @@ -6564,8 +6804,9 @@ fn apply_lifecycle_phase(sandbox: &mut Sandbox, phase: SandboxPhase, reason: &st set_next_restart_at_ms(status, 0); set_main_process_started_at_ms(status, 0); if phase == SandboxPhase::Starting { - status.provisioning = Some(provisioning_deadline::new_record( + status.provisioning = Some(provisioning_deadline::new_preparation_record( openshell_core::time::now_ms(), + image_preparation_timeout_seconds, )); status.configuration_admission = Some(openshell_core::proto::SandboxConfigurationAdmission { @@ -6940,6 +7181,7 @@ pub fn new_test_runtime_with_driver( telemetry_compute_driver: TelemetryComputeDriver::custom(), driver_process: None, default_image: "openshell/sandbox:test".to_string(), + image_preparation_timeout_seconds: 1800, store, sandbox_index: SandboxIndex::new(), sandbox_watch_bus: SandboxWatchBus::new(), @@ -7517,8 +7759,12 @@ mod tests { delete_requests: TestMutex>, delete_outcome: TestMutex, create_started: Notify, + create_finished: Notify, create_release: Semaphore, create_blocked: AtomicBool, + create_error: TestMutex>, + track_compute: AtomicBool, + compute_exists: AtomicBool, stop_started: Notify, stop_finished: Notify, stop_release: Semaphore, @@ -7556,8 +7802,12 @@ mod tests { delete_requests: TestMutex::new(Vec::new()), delete_outcome: TestMutex::new(ControlledDeleteOutcome::Ok(true)), create_started: Notify::new(), + create_finished: Notify::new(), create_release: Semaphore::new(0), create_blocked: AtomicBool::new(false), + create_error: TestMutex::new(None), + track_compute: AtomicBool::new(false), + compute_exists: AtomicBool::new(false), stop_started: Notify::new(), stop_finished: Notify::new(), stop_release: Semaphore::new(0), @@ -7811,6 +8061,18 @@ mod tests { .expect("create release semaphore closed") .forget(); } + self.create_finished.notify_one(); + let create_error = self + .create_error + .lock() + .expect("create error lock poisoned") + .clone(); + if let Some(error) = create_error { + return Err(error); + } + if self.track_compute.load(Ordering::SeqCst) { + self.compute_exists.store(true, Ordering::SeqCst); + } Ok(tonic::Response::new(CreateSandboxResponse { runtime_identity: self .runtime_identity @@ -7830,6 +8092,12 @@ mod tests { .expect("stop requests lock poisoned") .push((request.sandbox_id, request.name)); self.stop_calls.fetch_add(1, Ordering::SeqCst); + // Observe backend absence before the barrier, as a STOP may do + // before a concurrently submitted CREATE materializes compute. + let tracked_exists = self + .track_compute + .load(Ordering::SeqCst) + .then(|| self.compute_exists.swap(false, Ordering::SeqCst)); self.stop_started.notify_one(); if self.stop_blocked.load(Ordering::SeqCst) { self.stop_release @@ -7839,11 +8107,15 @@ mod tests { .forget(); } self.stop_finished.notify_one(); - let outcome = self - .stop_outcome - .lock() - .expect("stop outcome lock poisoned") - .clone(); + let outcome = match tracked_exists { + Some(true) => ControlledLifecycleOutcome::Ok, + Some(false) => ControlledLifecycleOutcome::NotFound, + None => self + .stop_outcome + .lock() + .expect("stop outcome lock poisoned") + .clone(), + }; match outcome { ControlledLifecycleOutcome::Ok => Ok(tonic::Response::new(StopSandboxResponse {})), ControlledLifecycleOutcome::NotFound => Err(Status::not_found("sandbox not found")), @@ -7989,6 +8261,7 @@ mod tests { telemetry_compute_driver: TelemetryComputeDriver::custom(), driver_process: None, default_image: "openshell/sandbox:test".to_string(), + image_preparation_timeout_seconds: 1800, store, sandbox_index: SandboxIndex::new(), sandbox_watch_bus: SandboxWatchBus::new(), @@ -8162,7 +8435,7 @@ mod tests { } #[tokio::test] - async fn create_compensation_deletes_backend_when_delete_transition_cannot_be_stored() { + async fn create_compensation_retains_backend_when_delete_transition_cannot_be_stored() { let directory = tempfile::tempdir().expect("temporary database directory"); let database_url = format!("sqlite://{}", directory.path().join("gateway.db").display()); let store = Arc::new(Store::connect(&database_url).await.expect("connect store")); @@ -8205,14 +8478,10 @@ mod tests { .message() .contains("could not claim the sandbox record") ); - assert_eq!(driver.delete_calls(), 1); - assert_eq!( - driver.delete_requests(), - vec![( - sandbox.object_id().to_string(), - sandbox.object_name().to_string() - )] - ); + // A rejected durable claim leaves cleanup ownership unknown. An + // unclaimed DELETE could destroy newer compute with the same ID. + assert_eq!(driver.delete_calls(), 0); + assert!(driver.delete_requests().is_empty()); let retained = runtime .store .get_message::(sandbox.object_id()) @@ -8510,12 +8779,13 @@ mod tests { .await .expect("start task must finish") .expect_err("a replaced generation must reject the prior runtime binding"); - assert!( - error - .message() - .contains("changed lifecycle ownership while persisting runtime identity") + assert_eq!(error.code(), Code::Aborted); + assert!(error.message().contains("runtime generation changed")); + assert_eq!( + driver.stop_calls(), + 0, + "stale START cannot stop replacement compute" ); - assert_eq!(driver.stop_calls(), 1); let retained = runtime .store .get_message::(sandbox.object_id()) @@ -10612,18 +10882,19 @@ mod tests { assert!(!crate::policy_store::permits_initial_static_policy_repair( &blocked )); - // Simulate the supervisor's successful exact-generation admission report. + // The supervisor report records the preparation-to-admission transition + // together with acceptance of the exact configuration generation. runtime .store .update_message_cas::(sandbox.object_id(), 0, |sandbox| { - sandbox - .status - .as_mut() - .unwrap() - .configuration_admission - .as_mut() - .unwrap() - .state = openshell_core::proto::ConfigurationAdmissionState::Accepted.into(); + let status = sandbox.status.as_mut().unwrap(); + provisioning_deadline::record_admission_start( + status.provisioning.as_mut().unwrap(), + openshell_core::time::now_ms(), + ) + .unwrap(); + status.configuration_admission.as_mut().unwrap().state = + openshell_core::proto::ConfigurationAdmissionState::Accepted.into(); }) .await .unwrap(); @@ -11015,8 +11286,8 @@ mod tests { assert_eq!(stored_identity.auth_epoch.get(), 2); assert_eq!( sandbox_resource_version(&stored), - sandbox_resource_version(&before) + 1, - "identity rotation and Starting must be one CAS write" + sandbox_resource_version(&before) + 3, + "identity/Starting transition, operation claim, and settlement are separate CAS writes" ); runtime @@ -11050,8 +11321,8 @@ mod tests { .unwrap(); assert_eq!( sandbox_resource_version(&after_retry), - sandbox_resource_version(&stored), - "idempotent Starting retry must reuse the persisted identity" + sandbox_resource_version(&stored) + 2, + "Starting retry claims and settles another operation without rotating identity" ); } @@ -11080,6 +11351,8 @@ mod tests { .await .expect("detached start worker did not finish the driver call"); + wait_driver_pending(&runtime, sandbox.object_id(), false).await; + driver.release_start(); let starting = tokio::time::timeout( Duration::from_secs(1), @@ -11666,7 +11939,7 @@ mod tests { .unwrap(); let transition = runtime - .begin_sandbox_delete_with_initial_snapshot("sb-1", Some(stale_snapshot)) + .begin_sandbox_delete_with_initial_snapshot("sb-1", Some(stale_snapshot), None) .await .unwrap(); let BeginDelete::Started(transition) = transition else { @@ -15369,6 +15642,1615 @@ mod tests { .unwrap() } + async fn wait_driver_pending(runtime: &ComputeRuntime, id: &str, pending: bool) -> Sandbox { + tokio::time::timeout(Duration::from_secs(3), async { + loop { + let sandbox = runtime + .store + .get_message::(id) + .await + .unwrap() + .unwrap(); + if provisioning_deadline::driver_operation_pending(&sandbox) == pending { + break sandbox; + } + tokio::task::yield_now().await; + } + }) + .await + .expect("driver ownership must reach the expected durable state") + } + + fn stage_create_test_upload( + runtime: &mut ComputeRuntime, + sandbox: &mut Sandbox, + root: &Path, + ) -> PathBuf { + runtime.rootfs_tar_staging = Arc::new(rootfs_tar::RootfsTarStagingRegistry::new( + Some(root.to_path_buf()), + 1024, + )); + runtime.admission.allow_driver_config = true; + let slot = runtime + .rootfs_tar_staging + .begin("default", "test", "rootfs.tar", 7) + .unwrap(); + std::fs::write(&slot.upload_path, b"archive").unwrap(); + sandbox + .spec + .get_or_insert_with(SandboxSpec::default) + .template + .get_or_insert_with(SandboxTemplate::default) + .driver_config = Some(prost_types::Struct { + fields: std::iter::once(( + runtime.driver_info.name.clone(), + struct_value([("rootfs_tar_staging_token", string_value(&slot.token))]), + )) + .collect(), + }); + slot.upload_path + } + + async fn assert_late_create_preserves_recovery(create_error: bool, create_identity: &str) { + for recovery_pending in [true, false] { + let driver = ControlledDriver::new(); + driver.block_create(); + driver.set_runtime_identity(create_identity); + if create_error { + *driver.create_error.lock().unwrap() = + Some(Status::failed_precondition("create rejected")); + } + let mut runtime = test_runtime(driver.clone()).await; + enable_runtime_identity_binding(&mut runtime); + let mut other = runtime.clone(); + other.sync_lock = Arc::new(Mutex::new(())); + other.lifecycle_gates = Arc::new(LifecycleGateRegistry::default()); + other.driver_info.gateway_manages_lifecycle = true; + let mut sandbox = sandbox_record( + "sb-result-owner", + "result-owner", + SandboxPhase::Provisioning, + ); + sandbox.status.as_mut().unwrap().provisioning = Some( + provisioning_deadline::new_preparation_record(openshell_core::time::now_ms(), 1800), + ); + let creating = runtime.clone(); + let create = + tokio::spawn(async move { creating.create_sandbox(sandbox, None, false).await }); + driver.create_started.notified().await; + let held = runtime.sync_lock.lock().await; + driver.release_create(); + wait_driver_pending(&other, "sb-result-owner", false).await; + driver.set_runtime_identity("recovery-runtime"); + if recovery_pending { + driver.block_start(); + } + let recovering = other.clone(); + let recovery = + tokio::spawn(async move { recovering.start_persisted_sandboxes().await }); + driver.start_started.notified().await; + if !recovery_pending { + wait_driver_pending(&other, "sb-result-owner", false).await; + // Wait for the binding callback, not just the driver response. + tokio::time::timeout(Duration::from_secs(2), async { + loop { + let row = other + .store + .get_message::("sb-result-owner") + .await + .unwrap() + .unwrap(); + if sandbox_compute_runtime_identity(&row) == "recovery-runtime" { + break; + } + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + } + drop(held); + let result = create.await.unwrap(); + let retained = other + .store + .get_message::("sb-result-owner") + .await + .unwrap() + .expect("newer recovery operation must retain its row"); + assert_eq!( + driver.delete_calls(), + 0, + "old CREATE must not compensate against recovery" + ); + if !recovery_pending { + assert_eq!( + sandbox_compute_runtime_identity(&retained), + "recovery-runtime" + ); + } + assert!( + result.is_err(), + "superseded CREATE must not publish success" + ); + if recovery_pending { + driver.release_start(); + } + recovery.await.unwrap().unwrap(); + } + } + + #[tokio::test] + async fn late_create_error_preserves_newer_recovery_operation() { + assert_late_create_preserves_recovery(true, "").await; + } + + #[tokio::test] + async fn late_create_success_preserves_newer_recovery_binding() { + assert_late_create_preserves_recovery(false, "create-runtime").await; + } + + #[tokio::test] + async fn late_create_compensation_preserves_newer_recovery_operation() { + assert_late_create_preserves_recovery(false, "").await; + } + + #[tokio::test] + async fn driver_settlement_preserves_active_cleanup_stop_lease() { + let driver = ControlledDriver::new(); + driver.block_create(); + driver.block_stop(); + let runtime = test_runtime(driver.clone()).await; + let mut other = runtime.clone(); + other.sync_lock = Arc::new(Mutex::new(())); + other.lifecycle_gates = Arc::new(LifecycleGateRegistry::default()); + let now = openshell_core::time::now_ms(); + let mut sandbox = sandbox_record("sb-stop-lease", "stop-lease", SandboxPhase::Provisioning); + sandbox.status.as_mut().unwrap().provisioning = + Some(provisioning_deadline::new_preparation_record(now, 1)); + let creating = runtime.clone(); + let create = + tokio::spawn(async move { creating.create_sandbox(sandbox, None, false).await }); + driver.create_started.notified().await; + other + .reconcile_provisioning_deadlines(now + 1000) + .await + .unwrap(); + driver.stop_started.notified().await; + let before = other + .store + .get_message::("sb-stop-lease") + .await + .unwrap() + .unwrap(); + let lease = before + .status + .as_ref() + .unwrap() + .provisioning + .as_ref() + .unwrap() + .cleanup_retry_time; + assert!(lease.is_some()); + driver.release_create(); + assert_eq!( + create.await.unwrap().unwrap_err().code(), + Code::DeadlineExceeded + ); + let settled = wait_driver_pending(&runtime, "sb-stop-lease", false).await; + assert_eq!( + settled + .status + .as_ref() + .unwrap() + .provisioning + .as_ref() + .unwrap() + .cleanup_retry_time, + lease, + "settlement cannot erase an active STOP lease" + ); + let gate = runtime.lifecycle_gates.lock_for("sb-stop-lease").await; + runtime + .reclaim_provisioning_timeout(&settled, &gate) + .await + .unwrap(); + assert_eq!( + driver.stop_calls(), + 1, + "second replica must honor active STOP lease" + ); + let current = runtime + .store + .get_message::("sb-stop-lease") + .await + .unwrap() + .unwrap(); + assert!( + current + .status + .as_ref() + .unwrap() + .provisioning + .as_ref() + .unwrap() + .cleanup_completed_time + .is_none() + ); + drop(gate); + driver.release_stop(); + let completed = other.lifecycle_gates.lock_for("sb-stop-lease").await; + drop(completed); + } + + #[tokio::test] + async fn automatic_restart_waits_for_pending_driver_and_preserves_schedule() { + let driver = ControlledDriver::new(); + driver.block_create(); + driver.set_runtime_identity("create-runtime"); + let mut runtime = test_runtime(driver.clone()).await; + enable_runtime_identity_binding(&mut runtime); + let mut other = runtime.clone(); + other.sync_lock = Arc::new(Mutex::new(())); + other.lifecycle_gates = Arc::new(LifecycleGateRegistry::default()); + let mut sandbox = + sandbox_record("sb-restart-pending", "restart-pending", SandboxPhase::Ready); + sandbox + .spec + .get_or_insert_with(SandboxSpec::default) + .restart_policy = SandboxRestartPolicy::OnFailure as i32; + let status = sandbox.status.as_mut().unwrap(); + status.main_process_instance_id = "instance-1".into(); + status.provisioning = Some(provisioning_deadline::new_preparation_record( + openshell_core::time::now_ms(), + 1800, + )); + let creating = runtime.clone(); + let create = + tokio::spawn(async move { creating.create_sandbox(sandbox, None, false).await }); + driver.create_started.notified().await; + other + .main_process_exited("sb-restart-pending", "instance-1", 9) + .await + .unwrap(); + other + .finalize_main_process_exit("sb-restart-pending", "instance-1") + .await + .unwrap(); + let scheduled = other + .store + .get_message::("sb-restart-pending") + .await + .unwrap() + .unwrap(); + other + .restart_sandbox_runtime("sb-restart-pending") + .await + .unwrap(); + assert_eq!( + driver.stop_calls(), + 0, + "automatic restart cannot STOP an owned driver operation" + ); + assert_eq!(driver.start_calls(), 0); + let deferred = other + .store + .get_message::("sb-restart-pending") + .await + .unwrap() + .unwrap(); + assert_eq!( + deferred, scheduled, + "pending restart keeps its durable schedule" + ); + driver.release_create(); + create.await.unwrap().unwrap(); + other + .restart_sandbox_runtime("sb-restart-pending") + .await + .unwrap(); + assert_eq!(driver.stop_calls(), 1); + assert_eq!(driver.start_calls(), 1); + } + + #[tokio::test] + async fn actual_start_error_restores_authoritative_stopped_state() { + let driver = ControlledDriver::new(); + driver.set_start_outcome(ControlledLifecycleOutcome::Error("rejected start")); + let sandbox = sandbox_record("sb-rejected-start", "rejected-start", SandboxPhase::Stopped); + let mut stopped = ready_driver_sandbox(sandbox.object_id(), sandbox.object_name()); + stopped.status = Some(make_driver_status(make_driver_condition( + "ContainerStopped", + "Stopped", + ))); + stopped + .status + .as_mut() + .unwrap() + .conditions + .push(DriverCondition { + r#type: "Suspended".into(), + status: "True".into(), + ..Default::default() + }); + driver.set_get_outcome(ControlledGetOutcome::Sandbox(Box::new(stopped))); + let runtime = test_runtime(driver).await; + runtime.store.put_message(&sandbox).await.unwrap(); + runtime + .start_sandbox("default", "rejected-start") + .await + .unwrap_err(); + let restored = runtime + .store + .get_message::(sandbox.object_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(restored.phase(), i32::from(SandboxPhase::Stopped)); + } + + #[tokio::test] + async fn automatic_restart_retains_owned_start_after_caller_cancellation() { + let driver = ControlledDriver::new(); + driver.block_start(); + let runtime = test_runtime(driver.clone()).await; + let mut other = runtime.clone(); + other.sync_lock = Arc::new(Mutex::new(())); + other.lifecycle_gates = Arc::new(LifecycleGateRegistry::default()); + let mut sandbox = sandbox_record("sb-auto-owned", "auto-owned", SandboxPhase::Ready); + sandbox + .spec + .get_or_insert_with(SandboxSpec::default) + .restart_policy = SandboxRestartPolicy::OnFailure as i32; + let status = sandbox.status.as_mut().unwrap(); + status.main_process_instance_id = "instance-1".into(); + status.provisioning = Some(provisioning_deadline::new_record( + openshell_core::time::now_ms(), + )); + runtime.store.put_message(&sandbox).await.unwrap(); + runtime + .main_process_exited(sandbox.object_id(), "instance-1", 9) + .await + .unwrap(); + runtime + .finalize_main_process_exit(sandbox.object_id(), "instance-1") + .await + .unwrap(); + let restarting = runtime.clone(); + let restart = + tokio::spawn(async move { restarting.restart_sandbox_runtime("sb-auto-owned").await }); + driver.start_started.notified().await; + let owned = wait_driver_pending(&other, sandbox.object_id(), true).await; + restart.abort(); + assert!(restart.await.unwrap_err().is_cancelled()); + let error = other + .await_provisioning_operation(&owned, async { Ok(()) }) + .await + .unwrap_err(); + assert_eq!(Status::from(error).code(), Code::FailedPrecondition); + other + .restart_sandbox_runtime(sandbox.object_id()) + .await + .unwrap(); + assert_eq!(driver.stop_calls(), 1); + assert_eq!(driver.start_calls(), 1); + driver.release_start(); + let settled = wait_driver_pending(&other, sandbox.object_id(), false).await; + assert!(provisioning_operation::same_operation(&owned, &settled)); + } + + #[tokio::test] + async fn start_settlement_cannot_restore_over_timeout() { + for failure in [false, true] { + let driver = ControlledDriver::new(); + driver.block_start(); + if failure { + driver.set_start_outcome(ControlledLifecycleOutcome::Error("late rejection")); + } + let runtime = test_runtime(driver.clone()).await; + let sandbox = + sandbox_record("sb-start-timeout", "start-timeout", SandboxPhase::Stopped); + runtime.store.put_message(&sandbox).await.unwrap(); + let starting = runtime.clone(); + let start = + tokio::spawn( + async move { starting.start_sandbox("default", "start-timeout").await }, + ); + driver.start_started.notified().await; + let pending = wait_driver_pending(&runtime, sandbox.object_id(), true).await; + let record = pending + .status + .as_ref() + .unwrap() + .provisioning + .as_ref() + .unwrap(); + let deadline = + openshell_core::time::timestamp_to_millis(record.deadline.as_ref().unwrap()) + .unwrap(); + runtime + .reconcile_provisioning_deadlines(deadline) + .await + .unwrap(); + driver.release_start(); + assert_eq!( + start.await.unwrap().unwrap_err().code(), + Code::DeadlineExceeded + ); + let retained = wait_driver_pending(&runtime, sandbox.object_id(), false).await; + assert!(provisioning_deadline::timed_out(&retained)); + assert_eq!(retained.phase(), i32::from(SandboxPhase::Error)); + assert!(provisioning_operation::same_operation(&pending, &retained)); + } + } + + #[tokio::test] + async fn stale_start_compensation_and_rollback_cannot_reuse_operation_identity() { + let driver = ControlledDriver::new(); + let runtime = test_runtime(driver.clone()).await; + let mut previous = sandbox_record("sb-start-owner", "start-owner", SandboxPhase::Stopped); + let mut record = + provisioning_deadline::new_preparation_record(openshell_core::time::now_ms(), 1800); + record.driver_operation_id = "old-operation".into(); + previous.status.as_mut().unwrap().provisioning = Some(record); + runtime.store.put_message(&previous).await.unwrap(); + let current = runtime + .store + .get_message::(previous.object_id()) + .await + .unwrap() + .unwrap(); + let ((), old_settled) = runtime + .await_provisioning_operation(¤t, async { Ok(()) }) + .await + .unwrap(); + let ((), settled) = runtime + .await_provisioning_operation(&old_settled, async { Ok(()) }) + .await + .unwrap(); + let gate = runtime.lifecycle_gates.lock_for(previous.object_id()).await; + let error = runtime + .compensate_successful_start( + &gate, + &old_settled, + &previous, + Status::internal("missing binding"), + ) + .await; + assert_eq!(error.code(), Code::Aborted); + assert_eq!( + driver.stop_calls(), + 0, + "stale compensation cannot reach backend" + ); + assert!( + runtime + .restore_lifecycle_snapshot(&settled, &previous) + .await + ); + let restored = runtime + .store + .get_message::(previous.object_id()) + .await + .unwrap() + .unwrap(); + assert!(provisioning_operation::same_operation(&restored, &settled)); + assert!(!provisioning_operation::same_operation( + &restored, &previous + )); + assert!( + !runtime + .restore_lifecycle_snapshot(&old_settled, &previous) + .await + ); + } + + #[tokio::test] + async fn failed_create_compensation_cannot_delete_without_store_ownership() { + for generation_changed in [false, true] { + let driver = ControlledDriver::new(); + let directory = tempfile::tempdir().unwrap(); + let database_url = + format!("sqlite://{}", directory.path().join("gateway.db").display()); + let mut runtime = test_runtime(driver.clone()).await; + runtime.store = Arc::new(Store::connect(&database_url).await.unwrap()); + let pool = sqlx::SqlitePool::connect(&database_url).await.unwrap(); + let mut sandbox = sandbox_record( + "sb-cleanup-owner", + "cleanup-owner", + SandboxPhase::Provisioning, + ); + sandbox.status.as_mut().unwrap().provisioning = Some( + provisioning_deadline::new_record(openshell_core::time::now_ms()), + ); + runtime.store.put_message(&sandbox).await.unwrap(); + let owned = runtime + .store + .get_message::(sandbox.object_id()) + .await + .unwrap() + .unwrap(); + let retained = if generation_changed { + runtime + .store + .update_message_cas::( + sandbox.object_id(), + sandbox_resource_version(&owned), + |current| { + current.metadata.as_mut().unwrap().annotations.insert( + crate::auth::sandbox_session::RUNTIME_GENERATION_ANNOTATION.into(), + "replacement-generation".into(), + ); + }, + ) + .await + .unwrap() + } else { + sqlx::query("CREATE TRIGGER reject_cleanup_claim BEFORE UPDATE ON objects BEGIN SELECT RAISE(ABORT, 'transient store failure'); END").execute(&pool).await.unwrap(); + owned.clone() + }; + let gate = runtime.lifecycle_gates.lock_for(sandbox.object_id()).await; + let global = runtime.lock_global_for_lifecycle(&gate).await; + let error = runtime + .compensate_failed_create(&owned, gate, global, Status::internal("missing binding")) + .await; + if generation_changed { + assert_eq!(error.code(), Code::Aborted); + } else { + assert!( + error + .message() + .contains("could not claim the sandbox record") + ); + } + assert_eq!( + driver.delete_calls(), + 0, + "unknown or changed ownership forbids backend cleanup" + ); + assert_eq!( + runtime + .store + .get_message::(sandbox.object_id()) + .await + .unwrap() + .unwrap(), + retained + ); + } + } + + #[tokio::test] + async fn automatic_restart_rejects_generation_change_during_stop() { + let driver = ControlledDriver::new(); + driver.block_stop(); + let runtime = test_runtime(driver.clone()).await; + let mut sandbox = + sandbox_record("sb-auto-generation", "auto-generation", SandboxPhase::Ready); + sandbox + .spec + .get_or_insert_with(SandboxSpec::default) + .restart_policy = SandboxRestartPolicy::OnFailure as i32; + let status = sandbox.status.as_mut().unwrap(); + status.main_process_instance_id = "instance-1".into(); + status.provisioning = Some(provisioning_deadline::new_record( + openshell_core::time::now_ms(), + )); + runtime.store.put_message(&sandbox).await.unwrap(); + runtime + .main_process_exited(sandbox.object_id(), "instance-1", 9) + .await + .unwrap(); + runtime + .finalize_main_process_exit(sandbox.object_id(), "instance-1") + .await + .unwrap(); + let restarting = runtime.clone(); + let restart = tokio::spawn(async move { + restarting + .restart_sandbox_runtime("sb-auto-generation") + .await + }); + driver.stop_started.notified().await; + let claimed = wait_driver_pending(&runtime, sandbox.object_id(), true).await; + runtime + .store + .update_message_cas::( + sandbox.object_id(), + sandbox_resource_version(&claimed), + |current| { + current.metadata.as_mut().unwrap().annotations.insert( + crate::auth::sandbox_session::RUNTIME_GENERATION_ANNOTATION.into(), + "replacement-generation".into(), + ); + }, + ) + .await + .unwrap(); + driver.release_stop(); + assert!(restart.await.unwrap().is_err()); + let settled = wait_driver_pending(&runtime, sandbox.object_id(), false).await; + assert_eq!( + driver.start_calls(), + 0, + "watchdog cannot authorize a replacement generation" + ); + assert!(provisioning_operation::same_operation(&claimed, &settled)); + assert_eq!( + sandbox_runtime_generation(&settled).unwrap().as_str(), + "replacement-generation" + ); + } + + #[tokio::test] + async fn transient_create_monitor_failure_retains_record_after_driver_response() { + let driver = ControlledDriver::new(); + driver.block_create(); + *driver.create_error.lock().unwrap() = + Some(Status::failed_precondition("ordinary rejection")); + let directory = tempfile::tempdir().unwrap(); + let database_url = format!("sqlite://{}", directory.path().join("gateway.db").display()); + let mut runtime = test_runtime(driver.clone()).await; + runtime.store = Arc::new(Store::connect(&database_url).await.unwrap()); + let pool = sqlx::SqlitePool::connect(&database_url).await.unwrap(); + let now = openshell_core::time::now_ms(); + let mut sandbox = sandbox_record("sb-monitor", "monitor", SandboxPhase::Provisioning); + sandbox.status.as_mut().unwrap().provisioning = + Some(provisioning_deadline::new_preparation_record(now, 60)); + let creating = runtime.clone(); + let create = + tokio::spawn(async move { creating.create_sandbox(sandbox, None, false).await }); + driver.create_started.notified().await; + let held = runtime.sync_lock.lock().await; + sqlx::query("ALTER TABLE objects RENAME TO temporarily_hidden_objects") + .execute(&pool) + .await + .unwrap(); + // The one-second monitor must observe a real transient SELECT error. + // The held lock prevents failed-create handling until the table returns. + tokio::time::sleep(Duration::from_millis(1500)).await; + sqlx::query("ALTER TABLE temporarily_hidden_objects RENAME TO objects") + .execute(&pool) + .await + .unwrap(); + driver.release_create(); + wait_driver_pending(&runtime, "sb-monitor", false).await; + drop(held); + let result = tokio::time::timeout(Duration::from_secs(3), create) + .await + .unwrap() + .unwrap() + .unwrap_err(); + assert!( + result.message().starts_with("monitor provisioning:"), + "{result}" + ); + let retained = runtime + .store + .get_message::("sb-monitor") + .await + .unwrap() + .expect("monitor interruption must not delete the record"); + assert!(!provisioning_deadline::driver_operation_pending(&retained)); + assert_eq!(retained.phase(), i32::from(SandboxPhase::Provisioning)); + runtime + .reconcile_provisioning_deadlines(now + 60_000) + .await + .unwrap(); + driver.stop_finished.notified().await; + assert_eq!(driver.delete_calls(), 0); + } + + #[tokio::test] + async fn canceled_create_retains_upload_and_cannot_be_redispatched() { + let driver = ControlledDriver::new(); + driver.block_create(); + let mut runtime = test_runtime(driver.clone()).await; + let now = openshell_core::time::now_ms(); + let mut sandbox = sandbox_record("sb-owned", "owned", SandboxPhase::Provisioning); + sandbox.status.as_mut().unwrap().provisioning = + Some(provisioning_deadline::new_preparation_record(now, 60)); + let root = tempfile::tempdir().unwrap(); + let upload = stage_create_test_upload(&mut runtime, &mut sandbox, root.path()); + let creating = runtime.clone(); + let create = + tokio::spawn(async move { creating.create_sandbox(sandbox, None, false).await }); + driver.create_started.notified().await; + create.abort(); + let _ = create.await; + let owned = wait_driver_pending(&runtime, "sb-owned", true).await; + assert!(upload.exists()); + let polled = Arc::new(AtomicBool::new(false)); + let observed = polled.clone(); + let error = runtime + .await_provisioning_operation(&owned, async move { + observed.store(true, Ordering::SeqCst); + Ok(()) + }) + .await + .unwrap_err(); + assert_eq!(Status::from(error).code(), Code::FailedPrecondition); + assert!( + !polled.load(Ordering::SeqCst), + "second owner must not dispatch" + ); + let mut restarted = runtime.clone(); + restarted.lifecycle_gates = Arc::new(LifecycleGateRegistry::default()); + restarted.sync_lock = Arc::new(Mutex::new(())); + restarted.driver_info.gateway_manages_lifecycle = true; + restarted + .start_persisted_sandboxes_with_authentication( + |_| async { panic!("pending recovery must not rotate authentication") }, + |_| async { Ok(()) }, + |_| panic!("pending recovery must not revoke authentication"), + ) + .await + .unwrap(); + assert_eq!(driver.start_calls(), 0); + let record = restarted + .store + .get(Sandbox::object_type(), "sb-owned") + .await + .unwrap() + .unwrap(); + restarted + .prune_missing_sandbox(record, openshell_core::time::now_ms(), 0) + .await + .unwrap(); + assert!( + restarted + .store + .get_message::("sb-owned") + .await + .unwrap() + .is_some() + ); + assert_eq!(driver.delete_calls(), 0); + assert_eq!( + restarted + .delete_sandbox("default", "owned") + .await + .unwrap_err() + .code(), + Code::FailedPrecondition + ); + driver.release_create(); + wait_driver_pending(&runtime, "sb-owned", false).await; + assert!(upload.exists(), "successful driver owns the archive"); + } + + #[tokio::test] + async fn actual_create_rejection_clears_pending_and_preserves_error_cleanup() { + let driver = ControlledDriver::new(); + *driver.create_error.lock().unwrap() = + Some(Status::failed_precondition("volume no longer exists")); + let mut runtime = test_runtime(driver).await; + let mut sandbox = sandbox_record("sb-rejected", "rejected", SandboxPhase::Provisioning); + sandbox.status.as_mut().unwrap().provisioning = Some( + provisioning_deadline::new_preparation_record(openshell_core::time::now_ms(), 60), + ); + let root = tempfile::tempdir().unwrap(); + let upload = stage_create_test_upload(&mut runtime, &mut sandbox, root.path()); + let error = runtime + .create_sandbox(sandbox, None, false) + .await + .unwrap_err(); + assert_eq!(error.code(), Code::FailedPrecondition); + assert!( + runtime + .store + .get_message::("sb-rejected") + .await + .unwrap() + .is_none() + ); + tokio::time::timeout(Duration::from_secs(1), async { + while upload.exists() { + tokio::task::yield_now().await; + } + }) + .await + .expect("actual rejection cleans the upload"); + } + + #[tokio::test] + async fn concurrent_provisioning_claims_dispatch_only_one_driver_operation() { + let runtime = test_runtime(ControlledDriver::new()).await; + let sandbox = seed_provisioning_attempt(&runtime).await; + let dispatched = Arc::new(AtomicUsize::new(0)); + let release = Arc::new(Semaphore::new(0)); + let spawn = || { + let runtime = runtime.clone(); + let sandbox = sandbox.clone(); + let dispatched = dispatched.clone(); + let release = release.clone(); + tokio::spawn(async move { + runtime + .await_provisioning_operation(&sandbox, async move { + dispatched.fetch_add(1, Ordering::SeqCst); + release.acquire().await.unwrap().forget(); + Ok(()) + }) + .await + }) + }; + let mut first = spawn(); + let mut second = spawn(); + let (loser, winner) = tokio::time::timeout(Duration::from_secs(3), async { + tokio::select! { + result = &mut first => (result, second), + result = &mut second => (result, first), + } + }) + .await + .unwrap(); + assert_eq!( + Status::from(loser.unwrap().unwrap_err()).code(), + Code::FailedPrecondition + ); + assert_eq!(dispatched.load(Ordering::SeqCst), 1); + release.add_permits(1); + assert!(winner.await.unwrap().is_ok()); + wait_driver_pending(&runtime, "sb-ttl", false).await; + } + + #[tokio::test] + async fn driver_completion_retries_transient_store_failure() { + let driver = ControlledDriver::new(); + driver.block_create(); + let directory = tempfile::tempdir().unwrap(); + let database_url = format!("sqlite://{}", directory.path().join("gateway.db").display()); + let mut runtime = test_runtime(driver.clone()).await; + runtime.store = Arc::new(Store::connect(&database_url).await.unwrap()); + let pool = sqlx::SqlitePool::connect(&database_url).await.unwrap(); + let mut sandbox = sandbox_record("sb-settlement", "settlement", SandboxPhase::Provisioning); + sandbox.status.as_mut().unwrap().provisioning = Some( + provisioning_deadline::new_preparation_record(openshell_core::time::now_ms(), 60), + ); + let creating = runtime.clone(); + let create = + tokio::spawn(async move { creating.create_sandbox(sandbox, None, false).await }); + driver.create_started.notified().await; + sqlx::query("CREATE TRIGGER reject_settlement BEFORE UPDATE ON objects BEGIN SELECT RAISE(ABORT, 'transient write failure'); END").execute(&pool).await.unwrap(); + sqlx::query("ALTER TABLE objects RENAME TO temporarily_hidden_objects") + .execute(&pool) + .await + .unwrap(); + driver.release_create(); + driver.create_finished.notified().await; + tokio::time::sleep(Duration::from_millis(50)).await; + assert!( + !create.is_finished(), + "owner retains the result after a settlement read failure" + ); + sqlx::query("ALTER TABLE temporarily_hidden_objects RENAME TO objects") + .execute(&pool) + .await + .unwrap(); + // The retry now reaches the injected write failure. Its result must + // remain owned across both failed reads and failed CAS persistence. + tokio::time::sleep(Duration::from_millis(1100)).await; + let pending = runtime + .store + .get_message::("sb-settlement") + .await + .unwrap() + .unwrap(); + assert!(provisioning_deadline::driver_operation_pending(&pending)); + assert!(!create.is_finished()); + sqlx::query("DROP TRIGGER reject_settlement") + .execute(&pool) + .await + .unwrap(); + assert!( + tokio::time::timeout(Duration::from_secs(3), create) + .await + .unwrap() + .unwrap() + .is_ok() + ); + let settled = runtime + .store + .get_message::("sb-settlement") + .await + .unwrap() + .unwrap(); + assert!(!provisioning_deadline::driver_operation_pending(&settled)); + } + + #[tokio::test] + async fn initial_create_preparation_expiry_releases_cleanup_gate() { + let driver = ControlledDriver::new(); + driver.block_create(); + let mut runtime = test_runtime(driver.clone()).await; + let now = openshell_core::time::now_ms(); + let preparation = provisioning_deadline::new_preparation_record(now, 1); + let mut sandbox = sandbox_record("sb-create-ttl", "create-ttl", SandboxPhase::Provisioning); + sandbox.status.as_mut().unwrap().provisioning = Some(preparation.clone()); + let upload_root = tempfile::tempdir().unwrap(); + let upload = stage_create_test_upload(&mut runtime, &mut sandbox, upload_root.path()); + let creating_runtime = runtime.clone(); + let mut create = + tokio::spawn( + async move { creating_runtime.create_sandbox(sandbox, None, false).await }, + ); + tokio::time::timeout(Duration::from_secs(1), driver.create_started.notified()) + .await + .unwrap(); + + runtime + .reconcile_provisioning_deadlines(now + 1_000) + .await + .unwrap(); + assert_eq!( + driver.stop_calls(), + 0, + "create still owns the lifecycle gate" + ); + let result = if let Ok(result) = + tokio::time::timeout(Duration::from_secs(3), &mut create).await + { + result.unwrap() + } else { + create.abort(); + let _ = create.await; + panic!("expired initial create kept the lifecycle gate while the driver was blocked"); + }; + assert_eq!(result.unwrap_err().code(), Code::DeadlineExceeded); + assert!(upload.exists(), "owned create still needs the upload"); + runtime + .reconcile_provisioning_deadlines(now + 1_001) + .await + .unwrap(); + driver.stop_finished.notified().await; + let cleanup_finished = runtime.lifecycle_gates.lock_for("sb-create-ttl").await; + drop(cleanup_finished); + let pending = wait_driver_pending(&runtime, "sb-create-ttl", true).await; + assert!( + pending + .status + .as_ref() + .unwrap() + .provisioning + .as_ref() + .unwrap() + .cleanup_completed_time + .is_none() + ); + driver.release_create(); + let settled = wait_driver_pending(&runtime, "sb-create-ttl", false).await; + // The earlier STOP has returned and released its gate. Advance only + // its retry backoff; settlement must preserve any still-active lease. + runtime + .store + .update_message_cas::( + "sb-create-ttl", + sandbox_resource_version(&settled), + |sandbox| { + sandbox + .status + .as_mut() + .unwrap() + .provisioning + .as_mut() + .unwrap() + .cleanup_retry_time = None; + }, + ) + .await + .unwrap(); + runtime + .reconcile_provisioning_deadlines(now + 1_002) + .await + .unwrap(); + let retained = tokio::time::timeout(Duration::from_secs(1), async { + loop { + let sandbox = runtime + .store + .get_message::("sb-create-ttl") + .await + .unwrap() + .expect("retained timeout record"); + if sandbox + .status + .as_ref() + .unwrap() + .provisioning + .as_ref() + .unwrap() + .cleanup_completed_time + .is_some() + { + break sandbox; + } + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + assert_eq!(driver.stop_calls(), 2); + assert_eq!(driver.delete_calls(), 0); + assert_eq!(retained.phase(), i32::from(SandboxPhase::Error)); + let status = retained.status.as_ref().unwrap(); + let record = status.provisioning.as_ref().unwrap(); + assert_eq!(record.attempt_id, preparation.attempt_id); + assert_eq!( + record.preparation_deadline, + preparation.preparation_deadline + ); + assert!( + status + .conditions + .iter() + .any(|c| c.reason == "ImagePreparationTimedOut") + ); + } + + #[tokio::test] + async fn failed_create_cleanup_preserves_timeout_after_cas_conflict() { + let driver = ControlledDriver::new(); + let runtime = test_runtime(driver.clone()).await; + let now = openshell_core::time::now_ms(); + let mut sandbox = sandbox_record( + "sb-compensation-ttl", + "compensation-ttl", + SandboxPhase::Provisioning, + ); + sandbox.status.as_mut().unwrap().provisioning = + Some(provisioning_deadline::new_preparation_record(now, 1)); + runtime.store.put_message(&sandbox).await.unwrap(); + let stale = runtime + .store + .get_message::(sandbox.object_id()) + .await + .unwrap() + .unwrap(); + let lifecycle_guard = runtime.lifecycle_gates.lock_for(sandbox.object_id()).await; + runtime + .reconcile_provisioning_deadlines(now + 1_000) + .await + .unwrap(); + let _global_guard = runtime.lock_global_for_lifecycle(&lifecycle_guard).await; + let result = runtime + .begin_sandbox_delete_with_initial_snapshot( + sandbox.object_id(), + Some(stale.clone()), + Some(&stale), + ) + .await; + assert!(matches!(result, Err(error) if error.code() == Code::DeadlineExceeded)); + let retained = runtime + .store + .get_message::(sandbox.object_id()) + .await + .unwrap() + .unwrap(); + assert!(provisioning_deadline::timed_out(&retained)); + assert_eq!(driver.delete_calls(), 0); + } + + #[tokio::test] + async fn late_initial_create_requires_cleanup_after_owned_response() { + let driver = ControlledDriver::new(); + driver.block_create(); + driver.block_stop(); + driver.track_compute.store(true, Ordering::SeqCst); + let runtime = test_runtime(driver.clone()).await; + let mut other = runtime.clone(); + other.sync_lock = Arc::new(Mutex::new(())); + other.lifecycle_gates = Arc::new(LifecycleGateRegistry::default()); + other.replica_id = "other-replica".into(); + let now = openshell_core::time::now_ms(); + let mut sandbox = sandbox_record( + "sb-create-retry", + "create-retry", + SandboxPhase::Provisioning, + ); + sandbox.status.as_mut().unwrap().provisioning = + Some(provisioning_deadline::new_preparation_record(now, 1)); + let original_attempt = sandbox_provisioning_attempt_id(&sandbox) + .unwrap() + .to_owned(); + let creating = runtime.clone(); + let create = + tokio::spawn(async move { creating.create_sandbox(sandbox, None, false).await }); + driver.create_started.notified().await; + other + .reconcile_provisioning_deadlines(now + 1_000) + .await + .unwrap(); + driver.stop_started.notified().await; + assert_eq!( + tokio::time::timeout(Duration::from_secs(3), create) + .await + .unwrap() + .unwrap() + .unwrap_err() + .code(), + Code::DeadlineExceeded + ); + assert_eq!( + runtime + .start_sandbox("default", "create-retry") + .await + .unwrap_err() + .code(), + Code::FailedPrecondition + ); + assert_eq!( + runtime + .stop_sandbox("default", "create-retry") + .await + .unwrap_err() + .code(), + Code::FailedPrecondition + ); + assert_eq!( + runtime + .delete_sandbox("default", "create-retry") + .await + .unwrap_err() + .code(), + Code::FailedPrecondition + ); + runtime.apply_deleted("sb-create-retry").await.unwrap(); + assert!( + runtime + .store + .get_message::("sb-create-retry") + .await + .unwrap() + .is_some() + ); + + // STOP has observed absence. Keep its final reread blocked while the + // original CREATE succeeds and durably clears pending ownership. + let held = other.sync_lock.lock().await; + driver.release_stop(); + driver.stop_finished.notified().await; + driver.release_create(); + let settled = wait_driver_pending(&runtime, "sb-create-retry", false).await; + assert!(driver.compute_exists.load(Ordering::SeqCst)); + assert_eq!( + sandbox_provisioning_attempt_id(&settled), + Some(original_attempt.as_str()) + ); + drop(held); + let gate = other.lifecycle_gates.lock_for("sb-create-retry").await; + let retained = other + .store + .get_message::("sb-create-retry") + .await + .unwrap() + .unwrap(); + let record = retained + .status + .as_ref() + .unwrap() + .provisioning + .as_ref() + .unwrap(); + assert!( + record.cleanup_completed_time.is_none(), + "pre-settlement STOP cannot complete cleanup" + ); + assert_eq!( + runtime + .start_sandbox("default", "create-retry") + .await + .unwrap_err() + .code(), + Code::FailedPrecondition + ); + // Advance only the cleanup retry timer; preserve the original deadline. + let retry = other + .store + .update_message_cas::( + "sb-create-retry", + sandbox_resource_version(&retained), + |sandbox| { + sandbox + .status + .as_mut() + .unwrap() + .provisioning + .as_mut() + .unwrap() + .cleanup_retry_time = None; + }, + ) + .await + .unwrap(); + driver.stop_blocked.store(false, Ordering::SeqCst); + other + .reclaim_provisioning_timeout(&retry, &gate) + .await + .unwrap(); + drop(gate); + let reclaimed = other + .store + .get_message::("sb-create-retry") + .await + .unwrap() + .unwrap(); + assert!( + reclaimed + .status + .as_ref() + .unwrap() + .provisioning + .as_ref() + .unwrap() + .cleanup_completed_time + .is_some() + ); + assert!(!driver.compute_exists.load(Ordering::SeqCst)); + assert_eq!(driver.stop_calls(), 2); + let retry = other + .start_sandbox("default", "create-retry") + .await + .unwrap(); + assert_ne!( + sandbox_provisioning_attempt_id(&retry), + Some(original_attempt.as_str()) + ); + assert_eq!(driver.delete_calls(), 0); + } + + #[tokio::test] + async fn failed_create_cleanup_preserves_retry_after_cas_conflict() { + let runtime = test_runtime(ControlledDriver::new()).await; + let mut sandbox = sandbox_record("sb-retry-cas", "retry-cas", SandboxPhase::Provisioning); + sandbox.status.as_mut().unwrap().provisioning = Some( + provisioning_deadline::new_preparation_record(openshell_core::time::now_ms(), 1), + ); + runtime.store.put_message(&sandbox).await.unwrap(); + let stale = runtime + .store + .get_message::(sandbox.object_id()) + .await + .unwrap() + .unwrap(); + let retry = runtime + .store + .update_message_cas::( + sandbox.object_id(), + sandbox_resource_version(&stale), + |current| { + apply_lifecycle_phase( + current, + SandboxPhase::Starting, + "Starting", + "Sandbox start requested", + runtime.image_preparation_timeout_seconds, + ); + }, + ) + .await + .unwrap(); + assert!(sandbox_resource_version(&retry) > sandbox_resource_version(&stale)); + let result = runtime + .begin_sandbox_delete_with_initial_snapshot( + sandbox.object_id(), + Some(stale.clone()), + Some(&stale), + ) + .await; + assert!(matches!(result, Err(error) if error.code() == Code::Aborted)); + let retained = runtime + .store + .get_message::(sandbox.object_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(retained.status, retry.status); + } + + #[tokio::test] + async fn create_runtime_binding_rejects_retry_after_cas_conflict() { + let runtime = test_runtime(ControlledDriver::new()).await; + let mut sandbox = sandbox_record( + "sb-binding-retry", + "binding-retry", + SandboxPhase::Provisioning, + ); + sandbox.status.as_mut().unwrap().provisioning = Some( + provisioning_deadline::new_preparation_record(openshell_core::time::now_ms(), 1), + ); + runtime.store.put_message(&sandbox).await.unwrap(); + let stale = runtime + .store + .get_message::(sandbox.object_id()) + .await + .unwrap() + .unwrap(); + let retry = runtime + .store + .update_message_cas::( + sandbox.object_id(), + sandbox_resource_version(&stale), + |current| { + apply_lifecycle_phase( + current, + SandboxPhase::Starting, + "Starting", + "Sandbox start requested", + runtime.image_preparation_timeout_seconds, + ); + current.set_phase(SandboxPhase::Ready.into()); + set_compute_runtime_binding(current, "retry-runtime"); + }, + ) + .await + .unwrap(); + assert_eq!( + sandbox_runtime_generation(&stale).unwrap(), + sandbox_runtime_generation(&retry).unwrap() + ); + let result = runtime + .persist_runtime_binding( + sandbox.object_id(), + &stale, + "test-driver", + "expired-create-runtime", + &[SandboxPhase::Provisioning, SandboxPhase::Ready], + ) + .await; + assert!( + result.is_err(), + "old create must not bind a replacement attempt" + ); + let retained = runtime + .store + .get_message::(sandbox.object_id()) + .await + .unwrap() + .unwrap(); + assert_eq!(retained, retry); + } + + #[tokio::test] + async fn initial_create_preserves_timeout_when_driver_returns_late() { + // Exercise every former error-deletion branch and the authenticated + // success path that would otherwise compensate for missing identity. + for error in [ + None, + Some(Status::already_exists("late duplicate")), + Some(Status::failed_precondition("late rejection")), + Some(Status::internal("late failure")), + ] { + let driver = ControlledDriver::new(); + driver.block_create(); + let accepted_upload = error.is_none(); + *driver.create_error.lock().unwrap() = error; + let mut runtime = test_runtime(driver.clone()).await; + runtime.driver_info.supports_sandbox_authentication = true; + let now = openshell_core::time::now_ms(); + let preparation = provisioning_deadline::new_preparation_record(now, 1); + let mut sandbox = + sandbox_record("sb-late-create", "late-create", SandboxPhase::Provisioning); + sandbox.status.as_mut().unwrap().provisioning = Some(preparation.clone()); + let upload_root = tempfile::tempdir().unwrap(); + let upload = stage_create_test_upload(&mut runtime, &mut sandbox, upload_root.path()); + let creating_runtime = runtime.clone(); + let create = + tokio::spawn( + async move { creating_runtime.create_sandbox(sandbox, None, false).await }, + ); + tokio::time::timeout(Duration::from_secs(1), driver.create_started.notified()) + .await + .unwrap(); + runtime + .reconcile_provisioning_deadlines(now + 1_000) + .await + .unwrap(); + driver.release_create(); + tokio::time::timeout(Duration::from_secs(1), driver.create_finished.notified()) + .await + .expect("driver result must reach the create path before cancellation"); + let result = tokio::time::timeout(Duration::from_secs(1), create) + .await + .unwrap() + .unwrap(); + assert_eq!(result.unwrap_err().code(), Code::DeadlineExceeded); + if accepted_upload { + assert!( + upload.exists(), + "successful driver owns the accepted upload" + ); + } else { + tokio::time::timeout(Duration::from_secs(1), async { + while upload.exists() { + tokio::task::yield_now().await; + } + }) + .await + .expect("failed create cleans up the upload"); + } + let retained = runtime + .store + .get_message::("sb-late-create") + .await + .unwrap() + .expect("late result must retain timeout diagnosis"); + assert!(provisioning_deadline::timed_out(&retained)); + let status = retained.status.as_ref().unwrap(); + assert_eq!( + status.provisioning.as_ref().unwrap().attempt_id, + preparation.attempt_id + ); + assert!( + status + .conditions + .iter() + .any(|c| c.reason == "ImagePreparationTimedOut") + ); + assert_eq!( + driver.delete_calls(), + 0, + "timeout cleanup must not enter create compensation" + ); + } + } + + #[tokio::test] + async fn preparation_expiry_releases_stalled_start_recovery_for_cleanup() { + let driver = ControlledDriver::new(); + driver.block_start(); + let runtime = test_runtime(driver.clone()).await; + let now = openshell_core::time::now_ms(); + let preparation = provisioning_deadline::new_preparation_record(now, 1800); + let mut sandbox = sandbox_record("sb-recovery-ttl", "recovery-ttl", SandboxPhase::Starting); + sandbox.status.as_mut().unwrap().provisioning = Some(preparation.clone()); + runtime.store.put_message(&sandbox).await.unwrap(); + let recovered_runtime = runtime.clone(); + let mut recovery = tokio::spawn(async move { + recovered_runtime + .recover_persisted_lifecycle_transitions() + .await + }); + tokio::time::timeout(Duration::from_secs(1), driver.start_started.notified()) + .await + .unwrap(); + runtime + .reconcile_provisioning_deadlines(now + 1_800_000) + .await + .unwrap(); + let expired = runtime + .store + .get_message::("sb-recovery-ttl") + .await + .unwrap() + .unwrap(); + assert_eq!(expired.phase(), i32::from(SandboxPhase::Error)); + let status = expired.status.as_ref().unwrap(); + let record = status.provisioning.as_ref().unwrap(); + assert_eq!(record.attempt_id, preparation.attempt_id); + assert_eq!( + record.preparation_deadline, + preparation.preparation_deadline + ); + assert!( + status + .conditions + .iter() + .any(|condition| condition.reason == "ImagePreparationTimedOut") + ); + // The recovery RPC never receives its semaphore permit. The persisted + // expiry must cancel its waiter and release the lifecycle gate itself. + if let Ok(result) = tokio::time::timeout(Duration::from_secs(3), &mut recovery).await { + result.unwrap().unwrap(); + } else { + recovery.abort(); + let _ = recovery.await; + panic!("expired recovery kept the lifecycle gate while the driver was blocked"); + } + assert!(provisioning_deadline::driver_operation_pending(&expired)); + driver.release_start(); + wait_driver_pending(&runtime, "sb-recovery-ttl", false).await; + runtime + .reconcile_provisioning_deadlines(now + 1_800_001) + .await + .unwrap(); + tokio::time::timeout(Duration::from_secs(1), async { + loop { + let sandbox = runtime + .store + .get_message::("sb-recovery-ttl") + .await + .unwrap() + .unwrap(); + let record = sandbox + .status + .as_ref() + .unwrap() + .provisioning + .as_ref() + .unwrap(); + if record.cleanup_completed_time.is_some() { + assert_eq!(record.attempt_id, preparation.attempt_id); + break; + } + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + assert_eq!(driver.stop_calls(), 1); + } + + #[tokio::test] + async fn preparation_timeout_retains_reason_and_uses_existing_cleanup() { + let driver = ControlledDriver::new(); + let runtime = test_runtime(driver.clone()).await; + let mut sandbox = sandbox_record("sb-prepare", "prepare", SandboxPhase::Provisioning); + sandbox.status.as_mut().unwrap().provisioning = + Some(provisioning_deadline::new_preparation_record(0, 1800)); + runtime.store.put_message(&sandbox).await.unwrap(); + let sandbox = runtime + .store + .get_message::("sb-prepare") + .await + .unwrap() + .unwrap(); + let gate = runtime.lifecycle_gates.lock_for("sb-prepare").await; + let global = runtime.lock_global_for_lifecycle(&gate).await; + assert!( + runtime + .claim_provisioning_timeout(&sandbox, 600_000) + .await + .unwrap() + .is_none() + ); + let expired = runtime + .claim_provisioning_timeout(&sandbox, 1_800_000) + .await + .unwrap() + .unwrap(); + assert_eq!(expired.phase(), i32::from(SandboxPhase::Error)); + assert!( + expired + .status + .as_ref() + .unwrap() + .conditions + .iter() + .any(|condition| { condition.reason == "ImagePreparationTimedOut" }) + ); + let record = expired + .status + .as_ref() + .unwrap() + .provisioning + .as_ref() + .unwrap(); + assert!(record.admission_start_time.is_none()); + assert!(record.preparation_deadline.is_some()); + assert!(record.cleanup_completed_time.is_none()); + drop(global); + runtime + .reclaim_provisioning_timeout(&expired, &gate) + .await + .unwrap(); + let reclaimed = runtime + .store + .get_message::("sb-prepare") + .await + .unwrap() + .unwrap(); + assert!( + reclaimed + .status + .as_ref() + .unwrap() + .provisioning + .as_ref() + .unwrap() + .cleanup_completed_time + .is_some() + ); + assert_eq!(driver.stop_calls(), 1); + } + #[tokio::test] async fn provisioning_timeout_persists_error_before_cleanup_and_preserves_record() { let driver = ControlledDriver::new(); @@ -15687,7 +17569,7 @@ mod tests { .await .expect("expired startup releases its lifecycle gate") .unwrap_err(); - assert_eq!(interrupted.code(), Code::DeadlineExceeded); + assert_eq!(Status::from(interrupted).code(), Code::DeadlineExceeded); runtime .mark_sandbox_error(&sandbox, "StartFailed", "late startup failure") .await; diff --git a/crates/openshell-server/src/compute/provisioning_deadline.rs b/crates/openshell-server/src/compute/provisioning_deadline.rs index 54d1f83682..1724e3c984 100644 --- a/crates/openshell-server/src/compute/provisioning_deadline.rs +++ b/crates/openshell-server/src/compute/provisioning_deadline.rs @@ -1,11 +1,12 @@ // SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. // SPDX-License-Identifier: Apache-2.0 -//! Gateway-owned provisioning repair-window transitions. +//! Gateway-owned image preparation and admission repair deadlines. //! //! Callers must persist each transition under the sandbox lifecycle fence. Times //! are gateway-assigned Unix milliseconds, never supervisor-supplied values. -//! The timer remains armed after admission acceptance until compute becomes Ready. +//! Preparation has an absolute ceiling. The first authenticated supervisor +//! configuration report starts admission repair, which remains armed until Ready. use openshell_core::proto::SandboxProvisioning; use openshell_core::time::{timestamp_from_millis, timestamp_to_millis}; @@ -26,6 +27,8 @@ pub(super) struct ProvisioningDeadline { attempt_id: String, change: ConfigurationChange, first_rejection_at_ms: Option, + preparation_deadline_at_ms: Option, + admission_start_at_ms: Option, state: DeadlineState, } @@ -50,6 +53,21 @@ impl ProvisioningDeadline { if record.deadline.is_some() && record.timeout_time.is_some() { return Err("provisioning cannot be armed and expired".into()); } + let preparation_deadline_at_ms = record + .preparation_deadline + .as_ref() + .map(|value| millis(Some(value), "preparation deadline")) + .transpose()?; + let admission_start_at_ms = record + .admission_start_time + .as_ref() + .map(|value| millis(Some(value), "admission start time")) + .transpose()?; + if admission_start_at_ms.is_some_and(|started| { + preparation_deadline_at_ms.is_none_or(|ceiling| started >= ceiling) + }) { + return Err("admission must start before its preparation deadline".into()); + } let state = if record.timeout_time.is_some() { DeadlineState::Expired { expired_at_ms: millis(record.timeout_time.as_ref(), "timeout time")?, @@ -61,6 +79,13 @@ impl ProvisioningDeadline { } else { DeadlineState::Ready }; + if let Some(ceiling) = preparation_deadline_at_ms + && admission_start_at_ms.is_none() + && !matches!(state, DeadlineState::Expired { .. }) + && !matches!(state, DeadlineState::Armed { deadline_at_ms } if deadline_at_ms == ceiling) + { + return Err("preparation must retain its absolute deadline until admission".into()); + } Ok(Self { attempt_id: record.attempt_id.clone(), change: ConfigurationChange { @@ -72,6 +97,8 @@ impl ProvisioningDeadline { .as_ref() .map(|value| millis(Some(value), "rejection time")) .transpose()?, + preparation_deadline_at_ms, + admission_start_at_ms, state, }) } @@ -84,6 +111,12 @@ impl ProvisioningDeadline { record.first_rejection_time = self .first_rejection_at_ms .and_then(|value| timestamp_from_millis(value).ok()); + record.preparation_deadline = self + .preparation_deadline_at_ms + .and_then(|value| timestamp_from_millis(value).ok()); + record.admission_start_time = self + .admission_start_at_ms + .and_then(|value| timestamp_from_millis(value).ok()); record.deadline = self .deadline_at_ms() .and_then(|value| timestamp_from_millis(value).ok()); @@ -98,12 +131,38 @@ impl ProvisioningDeadline { attempt_id, change, first_rejection_at_ms: None, + preparation_deadline_at_ms: None, + admission_start_at_ms: None, state: DeadlineState::Armed { deadline_at_ms: now_ms.saturating_add(REPAIR_WINDOW_MS), }, } } + fn is_preparing(&self) -> bool { + self.preparation_deadline_at_ms.is_some() && self.admission_start_at_ms.is_none() + } + + /// Only an authenticated report for the current supervisor may call this. + /// Persist it with the report: duplicate delivery or restart must not grant + /// another admission window, and a late registration cannot revive compute. + fn start_admission(&mut self, attempt_id: &str, now_ms: i64) -> bool { + if attempt_id != self.attempt_id + || !self.is_preparing() + || now_ms < self.change.committed_at_ms + || self + .deadline_at_ms() + .is_none_or(|deadline| now_ms >= deadline) + { + return false; + } + self.admission_start_at_ms = Some(now_ms); + self.state = DeadlineState::Armed { + deadline_at_ms: now_ms.saturating_add(REPAIR_WINDOW_MS), + }; + true + } + pub fn deadline_at_ms(&self) -> Option { match self.state { DeadlineState::Armed { deadline_at_ms } => Some(deadline_at_ms), @@ -124,10 +183,12 @@ impl ProvisioningDeadline { { return false; } - self.state = DeadlineState::Armed { - deadline_at_ms: deadline_at_ms - .max(change.committed_at_ms.saturating_add(REPAIR_WINDOW_MS)), - }; + if !self.is_preparing() { + self.state = DeadlineState::Armed { + deadline_at_ms: deadline_at_ms + .max(change.committed_at_ms.saturating_add(REPAIR_WINDOW_MS)), + }; + } self.change = change; self.first_rejection_at_ms = None; true @@ -140,6 +201,7 @@ impl ProvisioningDeadline { return false; }; if !self.matches(attempt_id, change_id) + || self.is_preparing() || self.first_rejection_at_ms.is_some() || now_ms < self.change.committed_at_ms || now_ms >= deadline_at_ms @@ -173,6 +235,7 @@ impl ProvisioningDeadline { /// must not call this method. Late readiness requires an explicit retry. pub fn ready(&mut self, attempt_id: &str, change_id: &str, now_ms: i64) -> bool { if !self.matches(attempt_id, change_id) + || self.is_preparing() || self .deadline_at_ms() .is_none_or(|deadline| now_ms >= deadline) @@ -197,7 +260,17 @@ pub fn timed_out(sandbox: &openshell_core::proto::Sandbox) -> bool { .is_some_and(|record| record.timeout_time.is_some()) } -/// Create an independent attempt. Supervisor reconnects must never call this. +/// A submitted operation still owns the possibility of a later backend commit. +pub(super) fn driver_operation_pending(sandbox: &openshell_core::proto::Sandbox) -> bool { + sandbox + .status + .as_ref() + .and_then(|status| status.provisioning.as_ref()) + .is_some_and(|record| record.driver_operation_pending) +} + +/// Adopt an existing untimed attempt without granting it a new preparation phase. +/// New create/start operations use `new_preparation_record` instead. pub fn new_record(now_ms: i64) -> SandboxProvisioning { let mut record = SandboxProvisioning::default(); ProvisioningDeadline::new( @@ -212,6 +285,27 @@ pub fn new_record(now_ms: i64) -> SandboxProvisioning { record } +/// Create an independent attempt with a fixed preparation budget. The gateway +/// validates the configured seconds before constructing the runtime. Reconnects +/// and driver progress must never call this function. +pub fn new_preparation_record(now_ms: i64, timeout_seconds: u32) -> SandboxProvisioning { + let mut record = new_record(now_ms); + let ceiling = now_ms.saturating_add(i64::from(timeout_seconds) * 1_000); + record.preparation_deadline = timestamp_from_millis(ceiling).ok(); + record.deadline.clone_from(&record.preparation_deadline); + record +} + +/// The caller has authenticated the supervisor and checked its instance fence. +/// A Pending registration may start timing before configuration validation. +/// Persist this transition in the same CAS as the supervisor report. +pub fn record_admission_start(record: &mut SandboxProvisioning, now_ms: i64) -> Result<(), String> { + let mut deadline = ProvisioningDeadline::from_record(record)?; + deadline.start_admission(&record.attempt_id, now_ms); + deadline.write_record(record); + Ok(()) +} + /// Apply the first accepted rejection under the same CAS as admission evidence. /// The report's generation and supervisor instance must already be validated. pub fn record_rejection(record: &mut SandboxProvisioning, now_ms: i64) -> Result<(), String> { @@ -250,6 +344,10 @@ pub(super) fn reconcile_readiness(sandbox: &mut openshell_core::proto::Sandbox, if deadline.ready(&record.attempt_id, &record.configuration_change_id, now_ms) { deadline.write_record(record); } else { + let awaiting_registration = deadline.is_preparing() + && deadline + .deadline_at_ms() + .is_some_and(|value| now_ms < value); status.phase = SandboxPhase::Provisioning.into(); status .conditions @@ -257,8 +355,18 @@ pub(super) fn reconcile_readiness(sandbox: &mut openshell_core::proto::Sandbox, status.conditions.push(SandboxCondition { r#type: "Ready".into(), status: "False".into(), - reason: "ProvisioningDeadlineElapsed".into(), - message: "Provisioning deadline elapsed; awaiting compute reclamation".into(), + reason: if awaiting_registration { + "ConfigurationPending" + } else { + "ProvisioningDeadlineElapsed" + } + .into(), + message: if awaiting_registration { + "Waiting for an authenticated supervisor configuration report" + } else { + "Provisioning deadline elapsed; awaiting compute reclamation" + } + .into(), ..Default::default() }); } @@ -442,6 +550,7 @@ impl super::ComputeRuntime { return Ok(None); }; let mut deadline = ProvisioningDeadline::from_record(record)?; + let preparation_expired = deadline.is_preparing(); if !deadline.expire(&record.attempt_id, &record.configuration_change_id, now_ms) { return Ok(None); } @@ -463,7 +572,9 @@ impl super::ComputeRuntime { .configuration_admission .as_ref() .map_or("", |admission| admission.error.as_str()); - let message = if diagnostic.is_empty() { + let message = if preparation_expired { + "Image preparation or initial supervisor startup exceeded its absolute deadline".to_string() + } else if diagnostic.is_empty() { "Provisioning repair window expired after 300 seconds".to_string() } else { format!( @@ -476,7 +587,11 @@ impl super::ComputeRuntime { status.conditions.push(SandboxCondition { r#type: "Ready".into(), status: "False".into(), - reason: "ProvisioningTimedOut".into(), + reason: if preparation_expired { + "ImagePreparationTimedOut" + } else { + "ProvisioningTimedOut" + }.into(), message, transition_time: timestamp_from_millis(now_ms).ok(), }); @@ -488,7 +603,8 @@ impl super::ComputeRuntime { self.sandbox_watch_bus.notify(current.object_id()); tracing::warn!( sandbox_id = current.object_id(), - "Sandbox provisioning repair window expired" + preparation_expired, + "Sandbox provisioning deadline expired" ); Ok(Some(updated)) } @@ -540,6 +656,10 @@ impl super::ComputeRuntime { if record.timeout_time.is_none() || record.cleanup_completed_time.is_some() { return Ok(()); } + // A stop can observe NotFound before an in-flight create materializes + // compute. Even if that create finishes before the final read below, + // only a later stop issued after settlement can complete cleanup. + let pending_before_stop = record.driver_operation_pending; let now_ms = openshell_core::time::now_ms(); if record .cleanup_retry_time @@ -593,8 +713,9 @@ impl super::ComputeRuntime { ), ) .await; - let reclaimed = matches!(&result, Ok(Ok(_))) - || matches!(&result, Ok(Err(error)) if error.code() == tonic::Code::NotFound); + let reclaimed = !pending_before_stop + && (matches!(&result, Ok(Ok(_))) + || matches!(&result, Ok(Err(error)) if error.code() == tonic::Code::NotFound)); let _global_guard = self.lock_global_for_lifecycle(lifecycle_guard).await; let Some(current) = self .store @@ -614,6 +735,7 @@ impl super::ComputeRuntime { { return Ok(()); } + let reclaimed = reclaimed && !driver_operation_pending(¤t); let completed_at_ms = openshell_core::time::now_ms(); let updated = self .store @@ -654,6 +776,93 @@ impl super::ComputeRuntime { mod tests { use super::*; + #[test] + fn existing_wire_record_does_not_gain_preparation_time() { + use prost::Message; + // Encoded before preparation timestamps existed: attempt a, change c, + // configuration at epoch 0, and an admission deadline at 300 seconds. + let bytes = [0x0a, 1, b'a', 0x12, 1, b'c', 0x1a, 0, 0x2a, 3, 8, 0xac, 2]; + let mut record = SandboxProvisioning::decode(bytes.as_slice()).unwrap(); + assert!(!record.driver_operation_pending); + assert!(record.driver_operation_id.is_empty()); + let before = record.clone(); + record_admission_start(&mut record, 299_999).unwrap(); + assert_eq!(record, before); + assert!(record.preparation_deadline.is_none()); + assert!(!allows_admission(&record, 300_000)); + } + + #[test] + fn preparation_over_five_minutes_gets_a_full_admission_repair_window() { + let record = new_preparation_record(0, 1800); + let mut timer = ProvisioningDeadline::from_record(&record).unwrap(); + let attempt = record.attempt_id; + let change_id = record.configuration_change_id; + assert!(!timer.expire(&attempt, &change_id, 600_000)); + assert!(!timer.ready(&attempt, &change_id, 600_000)); + assert!(timer.start_admission(&attempt, 600_000)); + assert_eq!(timer.deadline_at_ms(), Some(900_000)); + assert!(timer.rejected(&attempt, &change_id, 601_000)); + assert_eq!(timer.deadline_at_ms(), Some(901_000)); + assert!(!timer.start_admission(&attempt, 800_000)); + assert!(timer.configuration_changed(&attempt, change("updated", 800_000))); + assert_eq!(timer.deadline_at_ms(), Some(1_100_000)); + assert_eq!(timer.admission_start_at_ms, Some(600_000)); + assert_eq!(timer.preparation_deadline_at_ms, Some(1_800_000)); + } + + #[test] + fn preparation_config_changes_and_restart_preserve_the_absolute_ceiling() { + use prost::Message; + let mut record = new_preparation_record(0, 1800); + record.driver_operation_pending = true; + record.driver_operation_id = "operation-1".into(); + let mut timer = ProvisioningDeadline::from_record(&record).unwrap(); + let attempt = record.attempt_id.clone(); + assert!(timer.configuration_changed(&attempt, change("first", 600_000))); + assert!(timer.configuration_changed(&attempt, change("last", 1_799_000))); + assert!(!timer.configuration_changed(&attempt, change("last", 1_799_500))); + assert!(!timer.rejected(&attempt, "last", 1_799_500)); + timer.write_record(&mut record); + let bytes = record.encode_to_vec(); + let restored = SandboxProvisioning::decode(bytes.as_slice()).unwrap(); + assert!(restored.driver_operation_pending); + assert_eq!(restored.driver_operation_id, "operation-1"); + let mut timer = ProvisioningDeadline::from_record(&restored).unwrap(); + assert_eq!(timer.deadline_at_ms(), Some(1_800_000)); + assert!(!timer.start_admission("previous-attempt", 1_799_999)); + assert!(!timer.start_admission(&attempt, 1_800_000)); + assert!(timer.expire(&attempt, "last", 1_800_000)); + assert!(!timer.start_admission(&attempt, 1_799_999)); + assert!(!timer.ready(&attempt, "last", 1_799_999)); + assert!(!timer.configuration_changed(&attempt, change("late", 1_799_999))); + } + + #[test] + fn admission_registration_roundtrip_does_not_restart_repair() { + use prost::Message; + let mut record = new_preparation_record(0, 1800); + record_admission_start(&mut record, 600_000).unwrap(); + let before = record.clone(); + let bytes = record.encode_to_vec(); + let mut restored = SandboxProvisioning::decode(bytes.as_slice()).unwrap(); + record_admission_start(&mut restored, 899_999).unwrap(); + assert_eq!(restored, before); + assert!(!allows_admission(&restored, 900_000)); + } + + #[test] + fn malformed_preparation_timing_cannot_grant_admission() { + let mut record = new_preparation_record(0, 1800); + record.deadline = timestamp_from_millis(1_800_001).ok(); + assert!(ProvisioningDeadline::from_record(&record).is_err()); + record.deadline = None; + assert!(ProvisioningDeadline::from_record(&record).is_err()); + record.deadline = record.preparation_deadline; + record.admission_start_time = timestamp_from_millis(1_800_000).ok(); + assert!(ProvisioningDeadline::from_record(&record).is_err()); + } + #[test] fn protobuf_roundtrip_retains_deadline_and_cleanup_progress() { use prost::Message; diff --git a/crates/openshell-server/src/compute/provisioning_operation.rs b/crates/openshell-server/src/compute/provisioning_operation.rs new file mode 100644 index 0000000000..764c136cc4 --- /dev/null +++ b/crates/openshell-server/src/compute/provisioning_operation.rs @@ -0,0 +1,313 @@ +// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +//! Retain ownership of submitted driver work until its outcome is known. +//! +//! Local cancellation and backend absence cannot prove that an external create +//! stopped. A durable claim prevents another replica from declaring cleanup +//! complete, retrying, or deleting the only record of an unresolved request. + +use std::future::Future; +use std::time::Duration; + +use openshell_core::{ObjectId, proto::Sandbox}; +use tonic::Status; +use tracing::Instrument as _; + +use super::{ + ComputeRuntime, provisioning_deadline, sandbox_provisioning_attempt_id, + sandbox_resource_version, sandbox_runtime_generation, +}; +use crate::persistence::PersistenceError; + +/// An attempt may contain several recovery operations on the same generation. +/// Keep the last operation ID after settlement so a delayed callback cannot +/// adopt a newer operation merely because it has already finished. +pub(super) fn same_operation(left: &Sandbox, right: &Sandbox) -> bool { + sandbox_provisioning_attempt_id(left) == sandbox_provisioning_attempt_id(right) + && operation_id(left) == operation_id(right) +} + +fn operation_id(sandbox: &Sandbox) -> Option<&str> { + sandbox + .status + .as_ref()? + .provisioning + .as_ref() + .map(|record| record.driver_operation_id.as_str()) +} + +/// Called only inside a CAS that has checked the prior operation is settled. +/// Legacy rows without an attempt retain their existing lifecycle contract. +pub(super) fn claim_record(sandbox: &mut Sandbox) { + if let Some(record) = sandbox + .status + .as_mut() + .and_then(|status| status.provisioning.as_mut()) + { + record.driver_operation_pending = true; + record.driver_operation_id = uuid::Uuid::new_v4().to_string(); + } +} + +pub(super) fn ensure_current_result(current: &Sandbox, owned: &Sandbox) -> Result<(), Status> { + if !same_operation(current, owned) + || sandbox_runtime_generation(current) != sandbox_runtime_generation(owned) + { + return Err(Status::aborted( + "sandbox driver operation changed before result handling", + )); + } + if provisioning_deadline::timed_out(current) { + return Err(Status::deadline_exceeded("provisioning deadline expired")); + } + ensure_operation_settled(current) +} + +pub(super) fn ensure_operation_settled(sandbox: &Sandbox) -> Result<(), Status> { + if provisioning_deadline::driver_operation_pending(sandbox) { + return Err(Status::failed_precondition( + "previous driver operation is still pending; retry and deletion are blocked", + )); + } + Ok(()) +} + +/// Keep monitor interruption separate from an actual driver response. Only +/// the latter may enter the caller's existing failure recovery path. +#[derive(Debug, thiserror::Error)] +pub(super) enum ProvisioningOperationError { + #[error("{status}")] + Driver { + status: Status, + settled: Box, + }, + #[error("{0}")] + Monitor(Status), + #[error("{0}")] + Unsettled(Status), +} + +impl From for Status { + fn from(error: ProvisioningOperationError) -> Self { + match error { + ProvisioningOperationError::Driver { status, .. } + | ProvisioningOperationError::Monitor(status) + | ProvisioningOperationError::Unsettled(status) => status, + } + } +} + +impl ComputeRuntime { + /// Claim before polling the operation, then retain its task even when the + /// caller or deadline monitor leaves. Pending describes the owned future, + /// not backend cancellation: transport-error ambiguity remains governed by + /// the compute driver's existing error contract. + pub(super) async fn await_provisioning_operation( + &self, + starting: &Sandbox, + operation: impl Future> + Send + 'static, + ) -> Result<(T, Box), ProvisioningOperationError> { + let claimed = self + .claim_provisioning_operation(starting) + .await + .map_err(ProvisioningOperationError::Monitor)?; + self.await_claimed_provisioning_operation(&claimed, operation) + .await + } + + /// The caller has atomically claimed both its lifecycle transition and the + /// operation. The detached worker owns I/O, not the caller's local gate. + pub(super) async fn await_claimed_provisioning_operation( + &self, + starting: &Sandbox, + operation: impl Future> + Send + 'static, + ) -> Result<(T, Box), ProvisioningOperationError> { + let tracked = sandbox_provisioning_attempt_id(starting).is_some(); + let runtime = self.clone(); + let owned_attempt = starting.clone(); + // Dropping a JoinHandle detaches its task. Do not abort it on a monitor + // error or deadline: the remote side may still commit the request. + let worker = async move { + let result = operation.await; + let settled = if tracked { + let settled = runtime + .settle_provisioning_operation(&owned_attempt) + .await + .map_err(ProvisioningOperationError::Monitor)?; + // Settlement records a known response even if another writer + // rotated identity. That response cannot bind or compensate + // against the replacement generation in its returned snapshot. + if sandbox_runtime_generation(&settled) + != sandbox_runtime_generation(&owned_attempt) + { + return Err(ProvisioningOperationError::Monitor(Status::aborted( + "sandbox runtime generation changed during driver operation", + ))); + } + settled + } else { + owned_attempt + }; + match result { + Ok(response) => Ok((response, Box::new(settled))), + Err(status) => Err(ProvisioningOperationError::Driver { + status, + settled: Box::new(settled), + }), + } + }; + // Retain the request span so detaching ownership preserves the driver + // call's parent trace, including when the caller stops waiting. + let mut worker = tokio::spawn(worker.in_current_span()); + loop { + tokio::select! { + result = &mut worker => return result.map_err(|error| { + ProvisioningOperationError::Unsettled(Status::internal(format!( + "driver operation owner terminated; outcome remains unknown: {error}" + ))) + })?, + () = tokio::time::sleep(Duration::from_secs(1)) => { + let current = self.store.get_message::(starting.object_id()) + .await.map_err(|error| ProvisioningOperationError::Monitor( + Status::internal(format!("monitor provisioning: {error}")) + ))? + .ok_or_else(|| ProvisioningOperationError::Monitor( + Status::not_found("sandbox removed during startup") + ))?; + if !same_operation(¤t, starting) + || sandbox_runtime_generation(¤t) != sandbox_runtime_generation(starting) { + return Err(ProvisioningOperationError::Monitor(Status::aborted( + "sandbox provisioning attempt changed while waiting for compute" + ))); + } + if provisioning_deadline::timed_out(¤t) { + return Err(ProvisioningOperationError::Monitor(Status::deadline_exceeded( + "provisioning deadline expired; driver settlement may still be pending" + ))); + } + } + } + } + } + + async fn claim_provisioning_operation(&self, starting: &Sandbox) -> Result { + let Some(attempt_id) = sandbox_provisioning_attempt_id(starting) else { + // Legacy rows without an attempt do not gain a synthetic deadline. + return Ok(starting.clone()); + }; + if attempt_id.is_empty() { + return Err(Status::failed_precondition("provisioning attempt is empty")); + } + for _ in 0..super::START_PHASE_CAS_RETRY_LIMIT { + let current = self + .store + .get_message::(starting.object_id()) + .await + .map_err(|error| Status::internal(error.to_string()))? + .ok_or_else(|| Status::not_found("sandbox removed before driver dispatch"))?; + if provisioning_deadline::timed_out(¤t) { + return Err(Status::deadline_exceeded( + "provisioning deadline already expired", + )); + } + if provisioning_deadline::driver_operation_pending(¤t) { + return Err(Status::failed_precondition( + "previous driver operation may still complete; retry and deletion are blocked", + )); + } + if !same_operation(¤t, starting) + || sandbox_runtime_generation(¤t) != sandbox_runtime_generation(starting) + { + return Err(Status::aborted( + "provisioning operation changed before driver dispatch", + )); + } + if current.phase() != starting.phase() { + return Err(Status::aborted( + "sandbox phase changed before driver dispatch", + )); + } + match self + .store + .update_message_cas::( + starting.object_id(), + sandbox_resource_version(¤t), + claim_record, + ) + .await + { + Ok(updated) => { + self.sandbox_index.update_from_sandbox(&updated); + self.sandbox_watch_bus.notify(starting.object_id()); + return Ok(updated); + } + Err(PersistenceError::Conflict { .. }) => {} + Err(error) => return Err(Status::internal(error.to_string())), + } + } + Err(Status::aborted( + "sandbox kept changing before driver dispatch", + )) + } + + async fn settle_provisioning_operation(&self, starting: &Sandbox) -> Result { + // Keep the observed response in this task across transient persistence + // failures. A process crash still leaves the durable claim intact; + // neither lease expiry nor backend absence can safely clear it. + loop { + let result = self.try_settle_provisioning_operation(starting).await; + match result { + Ok(settled) => return Ok(settled), + Err(error) if error.code() == tonic::Code::Aborted => return Err(error), + Err(error) => { + tracing::warn!(sandbox_id = starting.object_id(), %error, + "Retaining driver response until operation ownership can be persisted"); + tokio::time::sleep(Duration::from_secs(1)).await; + } + } + } + } + + async fn try_settle_provisioning_operation( + &self, + starting: &Sandbox, + ) -> Result { + let current = self + .store + .get_message::(starting.object_id()) + .await + .map_err(|error| Status::internal(error.to_string()))? + .ok_or_else(|| Status::aborted("sandbox removed before driver settlement"))?; + if !same_operation(¤t, starting) { + return Err(Status::aborted( + "provisioning attempt changed before driver settlement", + )); + } + let updated = self + .store + .update_message_cas::( + starting.object_id(), + sandbox_resource_version(¤t), + |sandbox| { + if let Some(record) = sandbox + .status + .as_mut() + .and_then(|status| status.provisioning.as_mut()) + { + record.driver_operation_pending = false; + // Any stop before this response may have seen absence before + // the late create committed. Require another cleanup pass. + record.cleanup_completed_time = None; + // The retry timestamp may be another replica's active + // STOP lease. Preserve it through driver settlement. + } + }, + ) + .await + .map_err(|error| Status::internal(error.to_string()))?; + self.sandbox_index.update_from_sandbox(&updated); + self.sandbox_watch_bus.notify(starting.object_id()); + Ok(updated) + } +} diff --git a/crates/openshell-server/src/compute/rootfs_tar.rs b/crates/openshell-server/src/compute/rootfs_tar.rs index 2747d57374..544cdcafec 100644 --- a/crates/openshell-server/src/compute/rootfs_tar.rs +++ b/crates/openshell-server/src/compute/rootfs_tar.rs @@ -32,6 +32,9 @@ const MAX_SLOTS_PER_CALLER: usize = 4; /// let one caller starve everyone else. const MAX_TOTAL_SLOTS: usize = 64; const STAGING_DIR_PREFIX: &str = "req-"; +// A driver request can outlive its gateway. Age alone never proves that such +// a request stopped reading its input. +const PENDING_DRIVER_MARKER: &str = ".driver-request-pending"; const MAX_STAGED_FILE_NAME_LEN: usize = 128; /// `driver_config.` key the CLI sets to redeem a staging slot. @@ -86,6 +89,35 @@ impl StagedRootfsTar { pub fn disarm(&mut self) { self.dir = None; } + + /// Protect this input from the orphan sweep before dispatch. The armed + /// guard still removes it if dispatch is rejected before the driver runs. + pub async fn prepare_dispatch(&self) -> Result<(), Status> { + let directory = self + .dir + .as_ref() + .ok_or_else(|| Status::internal("rootfs upload ownership was already transferred"))?; + tokio::fs::write(directory.join(PENDING_DRIVER_MARKER), b"pending\n") + .await + .map_err(|error| Status::internal(format!("protect rootfs upload: {error}"))) + } + + /// Restore ordinary error cleanup after the owned driver future returns. + /// Successful responses leave cleanup with the driver; an interrupted + /// owner leaves the marker and archive for explicit reconciliation. + pub async fn finish_driver_operation(&mut self, accepted: bool) { + let Some(directory) = self.path.parent() else { + return; + }; + if let Err(error) = tokio::fs::remove_file(directory.join(PENDING_DRIVER_MARKER)).await + && error.kind() != std::io::ErrorKind::NotFound + { + warn!(%error, "Rootfs upload remains protected after driver response"); + } + if !accepted { + self.dir = Some(directory.to_path_buf()); + } + } } impl Drop for StagedRootfsTar { @@ -303,6 +335,12 @@ impl RootfsTarStagingRegistry { if !name.starts_with(STAGING_DIR_PREFIX) { continue; } + // Preserve uncertain ownership across gateway restarts. A missing + // response or old mtime is not permission to remove a live input. + match std::fs::symlink_metadata(entry.path().join(PENDING_DRIVER_MARKER)) { + Err(error) if error.kind() == std::io::ErrorKind::NotFound => {} + Ok(_) | Err(_) => continue, + } let stale = entry .metadata() .and_then(|meta| meta.modified()) @@ -690,4 +728,31 @@ mod tests { assert!(!fresh.exists()); assert!(unrelated.exists(), "unrelated entries must be left alone"); } + + #[tokio::test] + async fn orphan_sweep_preserves_dispatched_upload_after_owner_loss() { + let root = temp_root(); + let registry = RootfsTarStagingRegistry::new(Some(root.path().to_path_buf()), 1024); + let slot = registry.begin("default", "test", "rootfs.tar", 7).unwrap(); + std::fs::write(&slot.upload_path, b"archive").unwrap(); + let mut staged = registry.consume(&slot.token).unwrap(); + staged.prepare_dispatch().await.unwrap(); + staged.disarm(); + let restarted = RootfsTarStagingRegistry::with_ttl( + Some(root.path().to_path_buf()), + 1024, + Duration::ZERO, + ); + restarted.sweep_orphans(); + assert!( + slot.upload_path.exists(), + "age cannot settle an owned driver request" + ); + staged.finish_driver_operation(true).await; + restarted.sweep_orphans(); + assert!( + !slot.upload_path.exists(), + "settled uploads follow normal sweep rules" + ); + } } diff --git a/crates/openshell-server/src/config_file.rs b/crates/openshell-server/src/config_file.rs index e442607d9d..efa0182967 100644 --- a/crates/openshell-server/src/config_file.rs +++ b/crates/openshell-server/src/config_file.rs @@ -116,6 +116,9 @@ pub struct GatewayFileSection { // ── Sandbox / SSH ──────────────────────────────────────────────────── #[serde(default)] pub ssh_session_ttl_secs: Option, + /// Absolute preparation budget for new attempts; existing deadlines persist. + #[serde(default)] + pub image_preparation_timeout_seconds: Option, #[serde(default)] pub grpc_rate_limit_requests: Option, #[serde(default)] diff --git a/crates/openshell-server/src/grpc/policy.rs b/crates/openshell-server/src/grpc/policy.rs index ec65533bc5..bf115370b4 100644 --- a/crates/openshell-server/src/grpc/policy.rs +++ b/crates/openshell-server/src/grpc/policy.rs @@ -4600,7 +4600,7 @@ pub(super) async fn handle_report_sandbox_configuration( .map_err(Status::internal)?; if crate::compute::provisioning_deadline::timed_out(&sandbox) { return Err(Status::failed_precondition( - "provisioning repair window expired; explicitly start the sandbox after cleanup", + "provisioning deadline expired; explicitly start the sandbox after cleanup", )); } let current = sandbox @@ -4686,10 +4686,10 @@ pub(super) async fn handle_report_sandbox_configuration( .claim_provisioning_timeout(&sandbox, now_ms) .await .map_err(Status::internal)?; - return Err(Status::failed_precondition( - "provisioning repair window expired", - )); + return Err(Status::failed_precondition("provisioning deadline expired")); } + crate::compute::provisioning_deadline::record_admission_start(record, now_ms) + .map_err(Status::internal)?; if reported == ConfigurationAdmissionState::Rejected { crate::compute::provisioning_deadline::record_rejection(record, now_ms) .map_err(Status::internal)?; @@ -7906,6 +7906,143 @@ mod tests { request } + #[tokio::test] + async fn preparation_registration_starts_repair_once_after_a_cold_start() { + use crate::compute::provisioning_deadline::new_preparation_record; + use openshell_core::proto::{ + ConfigurationAdmissionState, ReportSandboxConfigurationRequest, + SandboxConfigurationAdmission, SandboxPhase, + }; + use openshell_core::time::timestamp_to_millis; + let state = test_server_state().await; + let sandbox_id = "sb-cold-registration"; + let mut sandbox = test_sandbox( + sandbox_id, + "cold-registration", + openshell_policy::restrictive_default_policy(), + Vec::new(), + ); + sandbox.set_phase(SandboxPhase::Provisioning.into()); + let preparation = new_preparation_record(current_time_ms() - 600_000, 1800); + sandbox.status.as_mut().unwrap().provisioning = Some(preparation.clone()); + state.store.put_message(&sandbox).await.unwrap(); + let instance_id = uuid::Uuid::new_v4().to_string(); + let request = || { + with_sandbox( + Request::new(ReportSandboxConfigurationRequest { + sandbox_id: sandbox_id.into(), + admission: Some(SandboxConfigurationAdmission { + instance_id: instance_id.clone(), + state: ConfigurationAdmissionState::Pending.into(), + ..Default::default() + }), + ..Default::default() + }), + sandbox_id, + ) + }; + // A duplicate report racing the initial registration must share one + // persisted transition; either may acquire the lifecycle fence first. + let (first, duplicate) = tokio::join!( + handle_report_sandbox_configuration(&state, request()), + handle_report_sandbox_configuration(&state, request()), + ); + first.unwrap(); + duplicate.unwrap(); + let saved = state + .store + .get_message::(sandbox_id) + .await + .unwrap() + .unwrap(); + let record = saved + .status + .as_ref() + .unwrap() + .provisioning + .as_ref() + .unwrap(); + assert_eq!( + record.preparation_deadline, + preparation.preparation_deadline + ); + let start = timestamp_to_millis(record.admission_start_time.as_ref().unwrap()).unwrap(); + let deadline = timestamp_to_millis(record.deadline.as_ref().unwrap()).unwrap(); + assert_eq!(deadline - start, 300_000); + handle_report_sandbox_configuration(&state, request()) + .await + .unwrap(); + let repeated = state + .store + .get_message::(sandbox_id) + .await + .unwrap() + .unwrap(); + assert_eq!( + repeated + .status + .as_ref() + .unwrap() + .provisioning + .as_ref() + .unwrap(), + record + ); + } + + #[tokio::test] + async fn preparation_expiry_rejects_registration_before_the_scanner_runs() { + use crate::compute::provisioning_deadline::new_preparation_record; + use openshell_core::proto::{ + ConfigurationAdmissionState, ReportSandboxConfigurationRequest, + SandboxConfigurationAdmission, SandboxPhase, + }; + let state = test_server_state().await; + let sandbox_id = "sb-expired-preparation"; + let mut sandbox = test_sandbox( + sandbox_id, + "expired-preparation", + openshell_policy::restrictive_default_policy(), + Vec::new(), + ); + sandbox.set_phase(SandboxPhase::Provisioning.into()); + sandbox.status.as_mut().unwrap().provisioning = Some(new_preparation_record(0, 1800)); + state.store.put_message(&sandbox).await.unwrap(); + let error = handle_report_sandbox_configuration( + &state, + with_sandbox( + Request::new(ReportSandboxConfigurationRequest { + sandbox_id: sandbox_id.into(), + admission: Some(SandboxConfigurationAdmission { + instance_id: uuid::Uuid::new_v4().to_string(), + state: ConfigurationAdmissionState::Pending.into(), + ..Default::default() + }), + ..Default::default() + }), + sandbox_id, + ), + ) + .await + .unwrap_err(); + assert_eq!(error.code(), Code::FailedPrecondition); + let expired = state + .store + .get_message::(sandbox_id) + .await + .unwrap() + .unwrap(); + assert_eq!(expired.phase(), i32::from(SandboxPhase::Error)); + let status = expired.status.unwrap(); + assert!(status.provisioning.unwrap().admission_start_time.is_none()); + assert!( + status + .conditions + .iter() + .any(|condition| condition.reason == "ImagePreparationTimedOut") + ); + } + #[tokio::test] async fn provisioning_timeout_rejects_supervisor_registration() { use openshell_core::proto::{ @@ -7944,7 +8081,7 @@ mod tests { .await .unwrap_err(); assert_eq!(error.code(), Code::FailedPrecondition); - assert!(error.message().contains("repair window expired")); + assert!(error.message().contains("provisioning deadline expired")); } #[tokio::test] diff --git a/crates/openshell-server/src/grpc/sandbox.rs b/crates/openshell-server/src/grpc/sandbox.rs index 594593a698..5dbae2cc7b 100644 --- a/crates/openshell-server/src/grpc/sandbox.rs +++ b/crates/openshell-server/src/grpc/sandbox.rs @@ -605,7 +605,12 @@ async fn handle_create_sandbox_inner( .status .as_mut() .expect("status initialized") - .provisioning = Some(crate::compute::provisioning_deadline::new_record(now_ms)); + .provisioning = Some( + crate::compute::provisioning_deadline::new_preparation_record( + now_ms, + state.config.image_preparation_timeout_seconds, + ), + ); crate::compute::provisioning_deadline::refresh_configuration( &state.store, &mut sandbox, diff --git a/crates/openshell-server/src/lib.rs b/crates/openshell-server/src/lib.rs index 353ca4396b..d17aa19b33 100644 --- a/crates/openshell-server/src/lib.rs +++ b/crates/openshell-server/src/lib.rs @@ -1726,6 +1726,9 @@ async fn build_compute_runtime( let runtime = runtime .with_admission_policy(admission) + .and_then(|runtime| { + runtime.with_image_preparation_timeout(config.image_preparation_timeout_seconds) + }) .map_err(Error::config)?; Ok(runtime.with_telemetry_compute_driver(telemetry_compute_driver)) } diff --git a/crates/openshell-server/src/storage_proto.rs b/crates/openshell-server/src/storage_proto.rs index e6d8730aa0..fb99042d63 100644 --- a/crates/openshell-server/src/storage_proto.rs +++ b/crates/openshell-server/src/storage_proto.rs @@ -126,12 +126,18 @@ mod tests { // inventories; the provider-environment file map is public-only. The // request has no provider-file capability field: older supervisors ignore // the additive file map while retaining the rest of the response. + // Preparation timing adds two optional timestamps to SandboxProvisioning. + // Existing rows decode with both absent and retain their active deadline; + // no stored attempt gains another phase or time budget on upgrade. // Service authorization also extends both schemas additively. Legacy // payloads retain the safe Strip default. + // Driver-operation ownership adds pending and a retained operation ID to + // SandboxProvisioning in both closures. Old rows decode false and empty; + // decoding or deadline updates cannot claim an existing attempt. const PUBLIC_RPC_SCHEMA_SHA256: &str = - "2e156c6ad3c8eb51bcd30dc13b173fe339b38207a1b1f98f7be2e0cad8e3bd45"; + "18206c52e68fdb0af60f8bb8dfaf47d9bc8021222cb49cacffab6352d3ad5549"; const DURABLE_SCHEMA_SHA256: &str = - "38165d9d76f49fcfe98a12f241e032838a2376c1d1a87ea2796fd33b9b1a3541"; + "76487ab369fc3a4b03a179bb5e7ea6be8d20e380ad5563075dff8ee50e539406"; const PUBLIC_DURABLE_OVERLAP_SHA256: &str = "761dea31a521b0650840fe2a823ad6e36a265ed323ba4506889781d630df0ee3"; // A persisted Sandbox without endpoint status retains its lifecycle fields; diff --git a/crates/openshell-tui/src/lib.rs b/crates/openshell-tui/src/lib.rs index 2265329bcf..2b99f77add 100644 --- a/crates/openshell-tui/src/lib.rs +++ b/crates/openshell-tui/src/lib.rs @@ -2722,7 +2722,12 @@ fn sandbox_notes_for_view( } else { "compute cleanup pending" }; - let mut notes = format!("Provisioning timed out; {cleanup}"); + let mut notes = + if record.preparation_deadline.is_some() && record.admission_start_time.is_none() { + format!("Image preparation timed out; {cleanup}") + } else { + format!("Provisioning timed out; {cleanup}") + }; if !forwards.is_empty() { notes.push_str("; "); notes.push_str(&forwards); @@ -3382,6 +3387,26 @@ mod sandbox_notes_tests { ); } + #[test] + fn preparation_timeout_notes_identify_the_expired_phase() { + let sandbox = Sandbox { + status: Some(SandboxStatus { + provisioning: Some(openshell_core::proto::SandboxProvisioning { + preparation_deadline: openshell_core::time::timestamp_from_millis(1_800_000) + .ok(), + timeout_time: openshell_core::time::timestamp_from_millis(1_800_000).ok(), + ..Default::default() + }), + ..Default::default() + }), + ..Default::default() + }; + assert_eq!( + sandbox_notes(&sandbox, String::new()), + "Image preparation timed out; compute cleanup pending" + ); + } + #[test] fn configuration_rejection_precedes_forwards_and_clears_after_repair() { let condition = SandboxCondition { diff --git a/docs/how-it-works/gateways/configuration.mdx b/docs/how-it-works/gateways/configuration.mdx index 52ebbed7fa..bdcd65a4bb 100644 --- a/docs/how-it-works/gateways/configuration.mdx +++ b/docs/how-it-works/gateways/configuration.mdx @@ -139,6 +139,10 @@ credential_drivers = ["kubernetes-secrets"] ssh_session_ttl_secs = 3600 +# Absolute image preparation and initial supervisor startup budget. +# Applies to new attempts only; allowed values are 1 through 86400 seconds. +image_preparation_timeout_seconds = 1800 + # Reject invalid policy generations securely by default. Set # "retain_last_valid" only when availability takes priority. policy_validation_failure_mode = "fail_closed" diff --git a/docs/how-it-works/policies/manage-policies.mdx b/docs/how-it-works/policies/manage-policies.mdx index f7416a62fc..c20af7b67b 100644 --- a/docs/how-it-works/policies/manage-policies.mdx +++ b/docs/how-it-works/policies/manage-policies.mdx @@ -368,14 +368,11 @@ openshell sandbox get my-sandbox --output json The same repair window applies when the gateway refuses an image-policy upload or a policy update that adds baseline filesystem paths with `FAILED_PRECONDITION` or `INVALID_ARGUMENT`. After the gateway acknowledges the rejection report, startup waits two seconds, reads a fresh configuration, and retries. If your repair reaches the gateway before that report, startup immediately reads the repaired configuration. The sandbox log records the gateway's reason; repeated refusals of the same write, status code, and configuration are logged once. Authentication and authorization failures still stop startup. -You have 300 seconds to fix the configuration. Replace the policy with -`openshell policy set`, or fix the provider configuration. Until the workload -first starts, you can also change filesystem, Landlock, and process settings. -Each change to the policy, providers, or settings restarts the 300 seconds. +You have 300 seconds to fix the configuration after the supervisor starts admission. Image preparation has its own deadline and does not consume that repair time. Replace the policy with `openshell policy set`, or fix the provider configuration. Until the workload first starts, you can also change filesystem, Landlock, and process settings. Each effective change to the policy, providers, or settings restarts the 300 seconds; writing an unchanged value does not. If the time runs out, the sandbox moves to `Error` with the reason `ProvisioningTimedOut`. Fixing the configuration does not restart it. After you -fix it, start the sandbox again, which begins a new 300-second window: +fix it, start the sandbox again. Admission gets a new 300-second window after preparation: ```shell openshell sandbox start my-sandbox diff --git a/docs/how-it-works/sandboxes/overview.mdx b/docs/how-it-works/sandboxes/overview.mdx index b084590eee..5be0d554ba 100644 --- a/docs/how-it-works/sandboxes/overview.mdx +++ b/docs/how-it-works/sandboxes/overview.mdx @@ -941,30 +941,24 @@ Management operations remain available while startup is blocked. After repair, the supervisor completes startup without recreating the sandbox. Starting a stopped sandbox repeats configuration admission before launching its workload. -The gateway enforces a 300-second provisioning repair window, independently of -the CLI wait timeout. An effective policy, settings, provider, profile, or -attachment change resets the window from its stored change time. The first -failed configuration load for that change grants another full window. Repeated -failures and reconnects do not extend it; reaching `Ready` clears it. +Image preparation and initial supervisor startup have an absolute deadline of 1800 seconds by default, independently of the CLI wait timeout. Operators can set `image_preparation_timeout_seconds` in `[openshell.gateway]` to any value from 1 through 86400. Each new create or explicit start stores its deadline. Progress events, configuration edits, and gateway or driver restarts cannot extend it. Changing the gateway setting affects new attempts only. -When the window expires, the sandbox enters `Error` with reason -`ProvisioningTimedOut`. The gateway stops its workload and supervisor compute, -retaining the sandbox record, diagnostic, and restartable storage. Cleanup can -remain pending if the backend is unavailable; the gateway retries it. Inspect -`provisioning` in JSON output for the deadline, timeout, and cleanup timestamps. -TUI NOTES distinguishes pending cleanup from reclaimed compute. +The first authenticated configuration report from the current supervisor ends preparation and starts a separate 300-second admission repair window. A slow image download therefore does not consume time reserved for fixing policy. During admission, an effective policy, settings, provider, profile, or attachment change resets the window from its stored change time. The first failed configuration load for that change grants another full window. Repeated failures, duplicate reports, and reconnects do not extend it; reaching `Ready` clears it. -Repair the configuration, wait for cleanup to complete, then explicitly restart: +Preparation expiry sets the sandbox to `Error` with reason `ImagePreparationTimedOut`; admission repair expiry uses `ProvisioningTimedOut`. The gateway stops its workload and supervisor compute, retaining the sandbox record, diagnostic, and restartable storage. Cleanup can remain pending if the backend is unavailable; the gateway retries it. Inspect `provisioning` in JSON output for the active `phase`, `deadline`, original `preparation_deadline`, `admission_start_time`, and cleanup timestamps. TUI NOTES identifies preparation timeout and distinguishes pending cleanup from reclaimed compute. + +The gateway keeps each submitted create or start operation running after its caller disconnects, its deadline expires, or monitoring fails. This includes recovery and the stop/start sequence for automatic restart. While `driver_operation_pending` is true, explicit start, stop, and deletion are blocked; automatic restart keeps its schedule and waits. Timeout cleanup can try to stop partial compute, but it cannot declare completion until the original driver call returns and a subsequent stop succeeds. This prevents a stop that observed missing compute from allowing a retry just before the original create finishes. Uploaded rootfs archives remain available to the owned operation. + +An actual driver response clears the pending flag, including ordinary rejection errors. The gateway retains `driver_operation_id` so a delayed response handler cannot change newer recovery work on the same provisioning attempt. Both fields appear in the `provisioning` JSON object. Backend behavior after a transport error still follows the driver's existing contract. If the owning gateway crashes before recording the response, the flag and any protected upload can remain indefinitely; restart and age alone cannot establish that the operation finished. There is currently no API to resolve that lost ownership, so automatic cleanup completion, retry, and deletion remain blocked for that record. + +For `ImagePreparationTimedOut`, inspect image preparation and supervisor startup diagnostics and check whether the configured preparation budget is sufficient. For `ProvisioningTimedOut`, repair the rejected configuration. In either case, wait for cleanup to complete, then explicitly restart: ```shell openshell sandbox get my-sandbox --output json openshell sandbox start my-sandbox ``` -Editing configuration after expiry does not restart compute. A retry gets a new -300-second window, while static-policy restrictions from any previous activation -remain in force. Timed-out records are retained even for ephemeral creates; use -`sandbox delete` when you no longer need the diagnostic or stored state. +Editing configuration after expiry does not restart compute. An explicit retry gets a new preparation deadline and a separate admission repair window, while static-policy restrictions from any previous activation remain in force. Attempts already active when the gateway is upgraded retain their stored deadline. Timed-out records are retained even for ephemeral creates; use `sandbox delete` when you no longer need the diagnostic or stored state. | Phase | Description | | ------------ | -------------------------------------------------------------------------------------------------------------------------------------------------------------- | diff --git a/proto/openshell.proto b/proto/openshell.proto index 83ffbe75ad..3d9cedad67 100644 --- a/proto/openshell.proto +++ b/proto/openshell.proto @@ -1184,7 +1184,7 @@ message SandboxStatus { SandboxConfigurationAdmission configuration_admission = 11; // Durable first-acceptance marker. Absent on legacy records; never reset by restart. optional bool configuration_activated = 12; - // Gateway-owned repair window. Retained after timeout for inspection and retry. + // Gateway-owned provisioning deadlines, retained after timeout for inspection and retry. SandboxProvisioning provisioning = 13; // Consecutive policy-driven restart number in the current crash loop. uint32 restart_count = 14; @@ -3769,7 +3769,7 @@ message SandboxProvisioning { string configuration_change_id = 2; google.protobuf.Timestamp configuration_change_time = 3; google.protobuf.Timestamp first_rejection_time = 4; - // Present only while the repair window is armed. + // Active preparation or admission-repair deadline. Absent after Ready/timeout. google.protobuf.Timestamp deadline = 5; google.protobuf.Timestamp timeout_time = 6; // Set only after both supervisor and workload compute have been reclaimed. @@ -3781,6 +3781,24 @@ message SandboxProvisioning { // Attachment edits have their own durable clock; status writes do not change it. string attachment_change_id = 10; google.protobuf.Timestamp attachment_change_time = 11; + // Absolute ceiling for image preparation and initial supervisor startup. + // Set once per new attempt; progress, configuration writes, and restarts + // cannot extend it. Absent on attempts created before preparation timing. + google.protobuf.Timestamp preparation_deadline = 12; + // First authenticated supervisor configuration report for this attempt. + // Starts the admission repair window. Absent while preparing, including + // after a preparation timeout. Duplicate reports never change this value. + google.protobuf.Timestamp admission_start_time = 13; + // An owned driver create/start future is still pending. Claimed before + // dispatch and cleared on its actual success or error response. Monitor + // interruption, owner loss, and StopSandbox/NotFound do not clear it. + // While set, cleanup cannot complete and retry/deletion remain blocked. + // This does not strengthen a driver's transport-error settlement contract. + bool driver_operation_pending = 14; + // Unique for each claimed driver operation, including recovery and automatic + // restart. Retained after settlement to reject delayed callbacks from an + // earlier operation on the same attempt and runtime generation. + string driver_operation_id = 15; } // Create-time request to expose one loopback HTTP service in a sandbox. diff --git a/sdk/go/proto/openshellv1/openshell.pb.go b/sdk/go/proto/openshellv1/openshell.pb.go index 109d2ef454..9de9c81318 100644 --- a/sdk/go/proto/openshellv1/openshell.pb.go +++ b/sdk/go/proto/openshellv1/openshell.pb.go @@ -3118,7 +3118,7 @@ type SandboxStatus struct { ConfigurationAdmission *SandboxConfigurationAdmission `protobuf:"bytes,11,opt,name=configuration_admission,json=configurationAdmission,proto3" json:"configuration_admission,omitempty"` // Durable first-acceptance marker. Absent on legacy records; never reset by restart. ConfigurationActivated *bool `protobuf:"varint,12,opt,name=configuration_activated,json=configurationActivated,proto3,oneof" json:"configuration_activated,omitempty"` - // Gateway-owned repair window. Retained after timeout for inspection and retry. + // Gateway-owned provisioning deadlines, retained after timeout for inspection and retry. Provisioning *SandboxProvisioning `protobuf:"bytes,13,opt,name=provisioning,proto3" json:"provisioning,omitempty"` // Consecutive policy-driven restart number in the current crash loop. RestartCount uint32 `protobuf:"varint,14,opt,name=restart_count,json=restartCount,proto3" json:"restart_count,omitempty"` @@ -17637,7 +17637,7 @@ type SandboxProvisioning struct { ConfigurationChangeId string `protobuf:"bytes,2,opt,name=configuration_change_id,json=configurationChangeId,proto3" json:"configuration_change_id,omitempty"` ConfigurationChangeTime *timestamppb.Timestamp `protobuf:"bytes,3,opt,name=configuration_change_time,json=configurationChangeTime,proto3" json:"configuration_change_time,omitempty"` FirstRejectionTime *timestamppb.Timestamp `protobuf:"bytes,4,opt,name=first_rejection_time,json=firstRejectionTime,proto3" json:"first_rejection_time,omitempty"` - // Present only while the repair window is armed. + // Active preparation or admission-repair deadline. Absent after Ready/timeout. Deadline *timestamppb.Timestamp `protobuf:"bytes,5,opt,name=deadline,proto3" json:"deadline,omitempty"` TimeoutTime *timestamppb.Timestamp `protobuf:"bytes,6,opt,name=timeout_time,json=timeoutTime,proto3" json:"timeout_time,omitempty"` // Set only after both supervisor and workload compute have been reclaimed. @@ -17649,8 +17649,26 @@ type SandboxProvisioning struct { // Attachment edits have their own durable clock; status writes do not change it. AttachmentChangeId string `protobuf:"bytes,10,opt,name=attachment_change_id,json=attachmentChangeId,proto3" json:"attachment_change_id,omitempty"` AttachmentChangeTime *timestamppb.Timestamp `protobuf:"bytes,11,opt,name=attachment_change_time,json=attachmentChangeTime,proto3" json:"attachment_change_time,omitempty"` - unknownFields protoimpl.UnknownFields - sizeCache protoimpl.SizeCache + // Absolute ceiling for image preparation and initial supervisor startup. + // Set once per new attempt; progress, configuration writes, and restarts + // cannot extend it. Absent on attempts created before preparation timing. + PreparationDeadline *timestamppb.Timestamp `protobuf:"bytes,12,opt,name=preparation_deadline,json=preparationDeadline,proto3" json:"preparation_deadline,omitempty"` + // First authenticated supervisor configuration report for this attempt. + // Starts the admission repair window. Absent while preparing, including + // after a preparation timeout. Duplicate reports never change this value. + AdmissionStartTime *timestamppb.Timestamp `protobuf:"bytes,13,opt,name=admission_start_time,json=admissionStartTime,proto3" json:"admission_start_time,omitempty"` + // An owned driver create/start future is still pending. Claimed before + // dispatch and cleared on its actual success or error response. Monitor + // interruption, owner loss, and StopSandbox/NotFound do not clear it. + // While set, cleanup cannot complete and retry/deletion remain blocked. + // This does not strengthen a driver's transport-error settlement contract. + DriverOperationPending bool `protobuf:"varint,14,opt,name=driver_operation_pending,json=driverOperationPending,proto3" json:"driver_operation_pending,omitempty"` + // Unique for each claimed driver operation, including recovery and automatic + // restart. Retained after settlement to reject delayed callbacks from an + // earlier operation on the same attempt and runtime generation. + DriverOperationId string `protobuf:"bytes,15,opt,name=driver_operation_id,json=driverOperationId,proto3" json:"driver_operation_id,omitempty"` + unknownFields protoimpl.UnknownFields + sizeCache protoimpl.SizeCache } func (x *SandboxProvisioning) Reset() { @@ -17760,6 +17778,34 @@ func (x *SandboxProvisioning) GetAttachmentChangeTime() *timestamppb.Timestamp { return nil } +func (x *SandboxProvisioning) GetPreparationDeadline() *timestamppb.Timestamp { + if x != nil { + return x.PreparationDeadline + } + return nil +} + +func (x *SandboxProvisioning) GetAdmissionStartTime() *timestamppb.Timestamp { + if x != nil { + return x.AdmissionStartTime + } + return nil +} + +func (x *SandboxProvisioning) GetDriverOperationPending() bool { + if x != nil { + return x.DriverOperationPending + } + return false +} + +func (x *SandboxProvisioning) GetDriverOperationId() string { + if x != nil { + return x.DriverOperationId + } + return "" +} + // Create-time request to expose one loopback HTTP service in a sandbox. type SandboxServiceExposure struct { state protoimpl.MessageState `protogen:"open.v1"` @@ -19141,7 +19187,7 @@ const file_openshell_proto_rawDesc = "" + "\x04path\x18\x04 \x01(\tR\x04path\x12=\n" + "\vlast_result\x18\x05 \x01(\x0e2\x1c.openshell.v1.EndpointResultR\n" + "lastResult\x12H\n" + - "\x12last_reported_time\x18j \x01(\v2\x1a.google.protobuf.TimestampR\x10lastReportedTimeJ\x04\b\x06\x10\aR\x10last_reported_at\"\xce\x05\n" + + "\x12last_reported_time\x18j \x01(\v2\x1a.google.protobuf.TimestampR\x10lastReportedTimeJ\x04\b\x06\x10\aR\x10last_reported_at\"\xd5\a\n" + "\x13SandboxProvisioning\x12\x1d\n" + "\n" + "attempt_id\x18\x01 \x01(\tR\tattemptId\x126\n" + @@ -19155,7 +19201,11 @@ const file_openshell_proto_rawDesc = "" + "\x12cleanup_retry_time\x18\t \x01(\v2\x1a.google.protobuf.TimestampR\x10cleanupRetryTime\x120\n" + "\x14attachment_change_id\x18\n" + " \x01(\tR\x12attachmentChangeId\x12P\n" + - "\x16attachment_change_time\x18\v \x01(\v2\x1a.google.protobuf.TimestampR\x14attachmentChangeTime\"\xaa\x01\n" + + "\x16attachment_change_time\x18\v \x01(\v2\x1a.google.protobuf.TimestampR\x14attachmentChangeTime\x12M\n" + + "\x14preparation_deadline\x18\f \x01(\v2\x1a.google.protobuf.TimestampR\x13preparationDeadline\x12L\n" + + "\x14admission_start_time\x18\r \x01(\v2\x1a.google.protobuf.TimestampR\x12admissionStartTime\x128\n" + + "\x18driver_operation_pending\x18\x0e \x01(\bR\x16driverOperationPending\x12.\n" + + "\x13driver_operation_id\x18\x0f \x01(\tR\x11driverOperationId\"\xaa\x01\n" + "\x16SandboxServiceExposure\x12\x18\n" + "\aservice\x18\x01 \x01(\tR\aservice\x12\x1f\n" + "\vtarget_port\x18\x02 \x01(\rR\n" + @@ -20108,180 +20158,182 @@ var file_openshell_proto_depIdxs = []int32{ 275, // 316: openshell.v1.SandboxProvisioning.cleanup_completed_time:type_name -> google.protobuf.Timestamp 275, // 317: openshell.v1.SandboxProvisioning.cleanup_retry_time:type_name -> google.protobuf.Timestamp 275, // 318: openshell.v1.SandboxProvisioning.attachment_change_time:type_name -> google.protobuf.Timestamp - 19, // 319: openshell.v1.SandboxServiceExposure.authorization_mode:type_name -> openshell.v1.ServiceAuthorizationMode - 275, // 320: openshell.v1.UpdateProviderRequest.CredentialExpirationTimesEntry.value:type_name -> google.protobuf.Timestamp - 275, // 321: openshell.v1.GetSandboxProviderEnvironmentResponse.CredentialExpirationTimesEntry.value:type_name -> google.protobuf.Timestamp - 127, // 322: openshell.v1.GetSandboxProviderEnvironmentResponse.DynamicCredentialsEntry.value:type_name -> openshell.v1.ProviderProfileCredential - 156, // 323: openshell.v1.GetSandboxProviderEnvironmentResponse.StaticCredentialBindingsEntry.value:type_name -> openshell.v1.StaticCredentialBinding - 24, // 324: openshell.v1.OpenShell.Health:input_type -> openshell.v1.HealthRequest - 26, // 325: openshell.v1.OpenShell.GetCurrentUser:input_type -> openshell.v1.GetCurrentUserRequest - 28, // 326: openshell.v1.OpenShell.GetGatewayInfo:input_type -> openshell.v1.GetGatewayInfoRequest - 52, // 327: openshell.v1.OpenShell.CreateSandbox:input_type -> openshell.v1.CreateSandboxRequest - 60, // 328: openshell.v1.OpenShell.BeginRootfsTarStaging:input_type -> openshell.v1.BeginRootfsTarStagingRequest - 62, // 329: openshell.v1.OpenShell.GetSandbox:input_type -> openshell.v1.GetSandboxRequest - 63, // 330: openshell.v1.OpenShell.ListSandboxes:input_type -> openshell.v1.ListSandboxesRequest - 53, // 331: openshell.v1.OpenShell.CreateSandboxTemplate:input_type -> openshell.v1.CreateSandboxTemplateRequest - 54, // 332: openshell.v1.OpenShell.GetSandboxTemplate:input_type -> openshell.v1.GetSandboxTemplateRequest - 55, // 333: openshell.v1.OpenShell.ListSandboxTemplates:input_type -> openshell.v1.ListSandboxTemplatesRequest - 56, // 334: openshell.v1.OpenShell.DeleteSandboxTemplate:input_type -> openshell.v1.DeleteSandboxTemplateRequest - 64, // 335: openshell.v1.OpenShell.ListSandboxProviders:input_type -> openshell.v1.ListSandboxProvidersRequest - 65, // 336: openshell.v1.OpenShell.AttachSandboxProvider:input_type -> openshell.v1.AttachSandboxProviderRequest - 66, // 337: openshell.v1.OpenShell.DetachSandboxProvider:input_type -> openshell.v1.DetachSandboxProviderRequest - 82, // 338: openshell.v1.OpenShell.GetSandboxProviderStatus:input_type -> openshell.v1.GetSandboxProviderStatusRequest - 67, // 339: openshell.v1.OpenShell.DeleteSandbox:input_type -> openshell.v1.DeleteSandboxRequest - 68, // 340: openshell.v1.OpenShell.StopSandbox:input_type -> openshell.v1.StopSandboxRequest - 69, // 341: openshell.v1.OpenShell.StartSandbox:input_type -> openshell.v1.StartSandboxRequest - 87, // 342: openshell.v1.OpenShell.CreateSshSession:input_type -> openshell.v1.CreateSshSessionRequest - 89, // 343: openshell.v1.OpenShell.ExposeService:input_type -> openshell.v1.ExposeServiceRequest - 90, // 344: openshell.v1.OpenShell.GetService:input_type -> openshell.v1.GetServiceRequest - 91, // 345: openshell.v1.OpenShell.ListServices:input_type -> openshell.v1.ListServicesRequest - 93, // 346: openshell.v1.OpenShell.DeleteService:input_type -> openshell.v1.DeleteServiceRequest - 97, // 347: openshell.v1.OpenShell.RevokeSshSession:input_type -> openshell.v1.RevokeSshSessionRequest - 99, // 348: openshell.v1.OpenShell.ExecSandbox:input_type -> openshell.v1.ExecSandboxRequest - 105, // 349: openshell.v1.OpenShell.ForwardTcp:input_type -> openshell.v1.TcpForwardFrame - 106, // 350: openshell.v1.OpenShell.ExecSandboxInteractive:input_type -> openshell.v1.ExecSandboxInput - 113, // 351: openshell.v1.OpenShell.CreateProvider:input_type -> openshell.v1.CreateProviderRequest - 114, // 352: openshell.v1.OpenShell.GetProvider:input_type -> openshell.v1.GetProviderRequest - 115, // 353: openshell.v1.OpenShell.ListProviders:input_type -> openshell.v1.ListProvidersRequest - 120, // 354: openshell.v1.OpenShell.ListProviderProfiles:input_type -> openshell.v1.ListProviderProfilesRequest - 121, // 355: openshell.v1.OpenShell.GetProviderProfile:input_type -> openshell.v1.GetProviderProfileRequest - 145, // 356: openshell.v1.OpenShell.ImportProviderProfiles:input_type -> openshell.v1.ImportProviderProfilesRequest - 147, // 357: openshell.v1.OpenShell.UpdateProviderProfiles:input_type -> openshell.v1.UpdateProviderProfilesRequest - 149, // 358: openshell.v1.OpenShell.LintProviderProfiles:input_type -> openshell.v1.LintProviderProfilesRequest - 116, // 359: openshell.v1.OpenShell.UpdateProvider:input_type -> openshell.v1.UpdateProviderRequest - 133, // 360: openshell.v1.OpenShell.GetProviderRefreshStatus:input_type -> openshell.v1.GetProviderRefreshStatusRequest - 135, // 361: openshell.v1.OpenShell.ConfigureProviderRefresh:input_type -> openshell.v1.ConfigureProviderRefreshRequest - 137, // 362: openshell.v1.OpenShell.RotateProviderCredential:input_type -> openshell.v1.RotateProviderCredentialRequest - 139, // 363: openshell.v1.OpenShell.DeleteProviderRefresh:input_type -> openshell.v1.DeleteProviderRefreshRequest - 117, // 364: openshell.v1.OpenShell.DeleteProvider:input_type -> openshell.v1.DeleteProviderRequest - 152, // 365: openshell.v1.OpenShell.DeleteProviderProfile:input_type -> openshell.v1.DeleteProviderProfileRequest - 290, // 366: openshell.v1.OpenShell.GetSandboxConfig:input_type -> openshell.sandbox.v1.GetSandboxConfigRequest - 291, // 367: openshell.v1.OpenShell.GetGatewayConfig:input_type -> openshell.sandbox.v1.GetGatewayConfigRequest - 160, // 368: openshell.v1.OpenShell.UpdateConfig:input_type -> openshell.v1.UpdateConfigRequest - 170, // 369: openshell.v1.OpenShell.GetSandboxPolicyStatus:input_type -> openshell.v1.GetSandboxPolicyStatusRequest - 172, // 370: openshell.v1.OpenShell.ListSandboxPolicies:input_type -> openshell.v1.ListSandboxPoliciesRequest - 174, // 371: openshell.v1.OpenShell.ReportPolicyStatus:input_type -> openshell.v1.ReportPolicyStatusRequest - 247, // 372: openshell.v1.OpenShell.ReportEndpointStatus:input_type -> openshell.v1.ReportEndpointStatusRequest - 84, // 373: openshell.v1.OpenShell.ReportProviderReadiness:input_type -> openshell.v1.ReportProviderReadinessRequest - 177, // 374: openshell.v1.OpenShell.ReportSandboxConfiguration:input_type -> openshell.v1.ReportSandboxConfigurationRequest - 154, // 375: openshell.v1.OpenShell.GetSandboxProviderEnvironment:input_type -> openshell.v1.GetSandboxProviderEnvironmentRequest - 158, // 376: openshell.v1.OpenShell.ExchangeProviderSubjectToken:input_type -> openshell.v1.ExchangeProviderSubjectTokenRequest - 180, // 377: openshell.v1.OpenShell.GetSandboxLogs:input_type -> openshell.v1.GetSandboxLogsRequest - 181, // 378: openshell.v1.OpenShell.PushSandboxLogs:input_type -> openshell.v1.PushSandboxLogsRequest - 184, // 379: openshell.v1.OpenShell.ConnectSupervisor:input_type -> openshell.v1.SupervisorMessage - 191, // 380: openshell.v1.OpenShell.ReportMainProcessExit:input_type -> openshell.v1.ReportMainProcessExitRequest - 193, // 381: openshell.v1.OpenShell.FinalizeMainProcessExit:input_type -> openshell.v1.FinalizeMainProcessExitRequest - 199, // 382: openshell.v1.OpenShell.RelayStream:input_type -> openshell.v1.RelayFrame - 201, // 383: openshell.v1.OpenShell.PeerRelay:input_type -> openshell.v1.PeerRelayFrame - 84, // 384: openshell.v1.OpenShell.PeerReportProviderReadiness:input_type -> openshell.v1.ReportProviderReadinessRequest - 247, // 385: openshell.v1.OpenShell.PeerReportEndpointStatus:input_type -> openshell.v1.ReportEndpointStatusRequest - 82, // 386: openshell.v1.OpenShell.PeerGetSandboxProviderStatus:input_type -> openshell.v1.GetSandboxProviderStatusRequest - 109, // 387: openshell.v1.OpenShell.WatchSandbox:input_type -> openshell.v1.WatchSandboxRequest - 210, // 388: openshell.v1.OpenShell.SubmitPolicyAnalysis:input_type -> openshell.v1.SubmitPolicyAnalysisRequest - 212, // 389: openshell.v1.OpenShell.GetDraftPolicy:input_type -> openshell.v1.GetDraftPolicyRequest - 214, // 390: openshell.v1.OpenShell.ApproveDraftChunk:input_type -> openshell.v1.ApproveDraftChunkRequest - 216, // 391: openshell.v1.OpenShell.RejectDraftChunk:input_type -> openshell.v1.RejectDraftChunkRequest - 219, // 392: openshell.v1.OpenShell.ApproveAllDraftChunks:input_type -> openshell.v1.ApproveAllDraftChunksRequest - 221, // 393: openshell.v1.OpenShell.EditDraftChunk:input_type -> openshell.v1.EditDraftChunkRequest - 223, // 394: openshell.v1.OpenShell.UndoDraftChunk:input_type -> openshell.v1.UndoDraftChunkRequest - 225, // 395: openshell.v1.OpenShell.ClearDraftChunks:input_type -> openshell.v1.ClearDraftChunksRequest - 227, // 396: openshell.v1.OpenShell.GetDraftHistory:input_type -> openshell.v1.GetDraftHistoryRequest - 20, // 397: openshell.v1.OpenShell.IssueSandboxToken:input_type -> openshell.v1.IssueSandboxTokenRequest - 22, // 398: openshell.v1.OpenShell.RefreshSandboxToken:input_type -> openshell.v1.RefreshSandboxTokenRequest - 230, // 399: openshell.v1.OpenShell.CreateWorkspace:input_type -> openshell.v1.CreateWorkspaceRequest - 232, // 400: openshell.v1.OpenShell.GetWorkspace:input_type -> openshell.v1.GetWorkspaceRequest - 234, // 401: openshell.v1.OpenShell.ListWorkspaces:input_type -> openshell.v1.ListWorkspacesRequest - 236, // 402: openshell.v1.OpenShell.DeleteWorkspace:input_type -> openshell.v1.DeleteWorkspaceRequest - 239, // 403: openshell.v1.OpenShell.AddWorkspaceMember:input_type -> openshell.v1.AddWorkspaceMemberRequest - 241, // 404: openshell.v1.OpenShell.RemoveWorkspaceMember:input_type -> openshell.v1.RemoveWorkspaceMemberRequest - 243, // 405: openshell.v1.OpenShell.ListWorkspaceMembers:input_type -> openshell.v1.ListWorkspaceMembersRequest - 25, // 406: openshell.v1.OpenShell.Health:output_type -> openshell.v1.HealthResponse - 27, // 407: openshell.v1.OpenShell.GetCurrentUser:output_type -> openshell.v1.GetCurrentUserResponse - 29, // 408: openshell.v1.OpenShell.GetGatewayInfo:output_type -> openshell.v1.GetGatewayInfoResponse - 70, // 409: openshell.v1.OpenShell.CreateSandbox:output_type -> openshell.v1.SandboxResponse - 61, // 410: openshell.v1.OpenShell.BeginRootfsTarStaging:output_type -> openshell.v1.BeginRootfsTarStagingResponse - 70, // 411: openshell.v1.OpenShell.GetSandbox:output_type -> openshell.v1.SandboxResponse - 71, // 412: openshell.v1.OpenShell.ListSandboxes:output_type -> openshell.v1.ListSandboxesResponse - 57, // 413: openshell.v1.OpenShell.CreateSandboxTemplate:output_type -> openshell.v1.SandboxTemplateResponse - 57, // 414: openshell.v1.OpenShell.GetSandboxTemplate:output_type -> openshell.v1.SandboxTemplateResponse - 58, // 415: openshell.v1.OpenShell.ListSandboxTemplates:output_type -> openshell.v1.ListSandboxTemplatesResponse - 59, // 416: openshell.v1.OpenShell.DeleteSandboxTemplate:output_type -> openshell.v1.DeleteSandboxTemplateResponse - 72, // 417: openshell.v1.OpenShell.ListSandboxProviders:output_type -> openshell.v1.ListSandboxProvidersResponse - 73, // 418: openshell.v1.OpenShell.AttachSandboxProvider:output_type -> openshell.v1.AttachSandboxProviderResponse - 74, // 419: openshell.v1.OpenShell.DetachSandboxProvider:output_type -> openshell.v1.DetachSandboxProviderResponse - 83, // 420: openshell.v1.OpenShell.GetSandboxProviderStatus:output_type -> openshell.v1.GetSandboxProviderStatusResponse - 86, // 421: openshell.v1.OpenShell.DeleteSandbox:output_type -> openshell.v1.DeleteSandboxResponse - 70, // 422: openshell.v1.OpenShell.StopSandbox:output_type -> openshell.v1.SandboxResponse - 70, // 423: openshell.v1.OpenShell.StartSandbox:output_type -> openshell.v1.SandboxResponse - 88, // 424: openshell.v1.OpenShell.CreateSshSession:output_type -> openshell.v1.CreateSshSessionResponse - 96, // 425: openshell.v1.OpenShell.ExposeService:output_type -> openshell.v1.ServiceEndpointResponse - 96, // 426: openshell.v1.OpenShell.GetService:output_type -> openshell.v1.ServiceEndpointResponse - 92, // 427: openshell.v1.OpenShell.ListServices:output_type -> openshell.v1.ListServicesResponse - 94, // 428: openshell.v1.OpenShell.DeleteService:output_type -> openshell.v1.DeleteServiceResponse - 98, // 429: openshell.v1.OpenShell.RevokeSshSession:output_type -> openshell.v1.RevokeSshSessionResponse - 103, // 430: openshell.v1.OpenShell.ExecSandbox:output_type -> openshell.v1.ExecSandboxEvent - 105, // 431: openshell.v1.OpenShell.ForwardTcp:output_type -> openshell.v1.TcpForwardFrame - 103, // 432: openshell.v1.OpenShell.ExecSandboxInteractive:output_type -> openshell.v1.ExecSandboxEvent - 118, // 433: openshell.v1.OpenShell.CreateProvider:output_type -> openshell.v1.ProviderResponse - 118, // 434: openshell.v1.OpenShell.GetProvider:output_type -> openshell.v1.ProviderResponse - 119, // 435: openshell.v1.OpenShell.ListProviders:output_type -> openshell.v1.ListProvidersResponse - 144, // 436: openshell.v1.OpenShell.ListProviderProfiles:output_type -> openshell.v1.ListProviderProfilesResponse - 143, // 437: openshell.v1.OpenShell.GetProviderProfile:output_type -> openshell.v1.ProviderProfileResponse - 146, // 438: openshell.v1.OpenShell.ImportProviderProfiles:output_type -> openshell.v1.ImportProviderProfilesResponse - 148, // 439: openshell.v1.OpenShell.UpdateProviderProfiles:output_type -> openshell.v1.UpdateProviderProfilesResponse - 150, // 440: openshell.v1.OpenShell.LintProviderProfiles:output_type -> openshell.v1.LintProviderProfilesResponse - 118, // 441: openshell.v1.OpenShell.UpdateProvider:output_type -> openshell.v1.ProviderResponse - 134, // 442: openshell.v1.OpenShell.GetProviderRefreshStatus:output_type -> openshell.v1.GetProviderRefreshStatusResponse - 136, // 443: openshell.v1.OpenShell.ConfigureProviderRefresh:output_type -> openshell.v1.ConfigureProviderRefreshResponse - 138, // 444: openshell.v1.OpenShell.RotateProviderCredential:output_type -> openshell.v1.RotateProviderCredentialResponse - 140, // 445: openshell.v1.OpenShell.DeleteProviderRefresh:output_type -> openshell.v1.DeleteProviderRefreshResponse - 151, // 446: openshell.v1.OpenShell.DeleteProvider:output_type -> openshell.v1.DeleteProviderResponse - 153, // 447: openshell.v1.OpenShell.DeleteProviderProfile:output_type -> openshell.v1.DeleteProviderProfileResponse - 292, // 448: openshell.v1.OpenShell.GetSandboxConfig:output_type -> openshell.sandbox.v1.GetSandboxConfigResponse - 293, // 449: openshell.v1.OpenShell.GetGatewayConfig:output_type -> openshell.sandbox.v1.GetGatewayConfigResponse - 169, // 450: openshell.v1.OpenShell.UpdateConfig:output_type -> openshell.v1.UpdateConfigResponse - 171, // 451: openshell.v1.OpenShell.GetSandboxPolicyStatus:output_type -> openshell.v1.GetSandboxPolicyStatusResponse - 173, // 452: openshell.v1.OpenShell.ListSandboxPolicies:output_type -> openshell.v1.ListSandboxPoliciesResponse - 175, // 453: openshell.v1.OpenShell.ReportPolicyStatus:output_type -> openshell.v1.ReportPolicyStatusResponse - 248, // 454: openshell.v1.OpenShell.ReportEndpointStatus:output_type -> openshell.v1.ReportEndpointStatusResponse - 85, // 455: openshell.v1.OpenShell.ReportProviderReadiness:output_type -> openshell.v1.ReportProviderReadinessResponse - 178, // 456: openshell.v1.OpenShell.ReportSandboxConfiguration:output_type -> openshell.v1.ReportSandboxConfigurationResponse - 157, // 457: openshell.v1.OpenShell.GetSandboxProviderEnvironment:output_type -> openshell.v1.GetSandboxProviderEnvironmentResponse - 159, // 458: openshell.v1.OpenShell.ExchangeProviderSubjectToken:output_type -> openshell.v1.ExchangeProviderSubjectTokenResponse - 183, // 459: openshell.v1.OpenShell.GetSandboxLogs:output_type -> openshell.v1.GetSandboxLogsResponse - 182, // 460: openshell.v1.OpenShell.PushSandboxLogs:output_type -> openshell.v1.PushSandboxLogsResponse - 185, // 461: openshell.v1.OpenShell.ConnectSupervisor:output_type -> openshell.v1.GatewayMessage - 192, // 462: openshell.v1.OpenShell.ReportMainProcessExit:output_type -> openshell.v1.ReportMainProcessExitResponse - 194, // 463: openshell.v1.OpenShell.FinalizeMainProcessExit:output_type -> openshell.v1.FinalizeMainProcessExitResponse - 199, // 464: openshell.v1.OpenShell.RelayStream:output_type -> openshell.v1.RelayFrame - 201, // 465: openshell.v1.OpenShell.PeerRelay:output_type -> openshell.v1.PeerRelayFrame - 85, // 466: openshell.v1.OpenShell.PeerReportProviderReadiness:output_type -> openshell.v1.ReportProviderReadinessResponse - 248, // 467: openshell.v1.OpenShell.PeerReportEndpointStatus:output_type -> openshell.v1.ReportEndpointStatusResponse - 83, // 468: openshell.v1.OpenShell.PeerGetSandboxProviderStatus:output_type -> openshell.v1.GetSandboxProviderStatusResponse - 110, // 469: openshell.v1.OpenShell.WatchSandbox:output_type -> openshell.v1.SandboxStreamEvent - 211, // 470: openshell.v1.OpenShell.SubmitPolicyAnalysis:output_type -> openshell.v1.SubmitPolicyAnalysisResponse - 213, // 471: openshell.v1.OpenShell.GetDraftPolicy:output_type -> openshell.v1.GetDraftPolicyResponse - 215, // 472: openshell.v1.OpenShell.ApproveDraftChunk:output_type -> openshell.v1.ApproveDraftChunkResponse - 217, // 473: openshell.v1.OpenShell.RejectDraftChunk:output_type -> openshell.v1.RejectDraftChunkResponse - 220, // 474: openshell.v1.OpenShell.ApproveAllDraftChunks:output_type -> openshell.v1.ApproveAllDraftChunksResponse - 222, // 475: openshell.v1.OpenShell.EditDraftChunk:output_type -> openshell.v1.EditDraftChunkResponse - 224, // 476: openshell.v1.OpenShell.UndoDraftChunk:output_type -> openshell.v1.UndoDraftChunkResponse - 226, // 477: openshell.v1.OpenShell.ClearDraftChunks:output_type -> openshell.v1.ClearDraftChunksResponse - 229, // 478: openshell.v1.OpenShell.GetDraftHistory:output_type -> openshell.v1.GetDraftHistoryResponse - 21, // 479: openshell.v1.OpenShell.IssueSandboxToken:output_type -> openshell.v1.IssueSandboxTokenResponse - 23, // 480: openshell.v1.OpenShell.RefreshSandboxToken:output_type -> openshell.v1.RefreshSandboxTokenResponse - 231, // 481: openshell.v1.OpenShell.CreateWorkspace:output_type -> openshell.v1.CreateWorkspaceResponse - 233, // 482: openshell.v1.OpenShell.GetWorkspace:output_type -> openshell.v1.GetWorkspaceResponse - 235, // 483: openshell.v1.OpenShell.ListWorkspaces:output_type -> openshell.v1.ListWorkspacesResponse - 237, // 484: openshell.v1.OpenShell.DeleteWorkspace:output_type -> openshell.v1.DeleteWorkspaceResponse - 240, // 485: openshell.v1.OpenShell.AddWorkspaceMember:output_type -> openshell.v1.AddWorkspaceMemberResponse - 242, // 486: openshell.v1.OpenShell.RemoveWorkspaceMember:output_type -> openshell.v1.RemoveWorkspaceMemberResponse - 244, // 487: openshell.v1.OpenShell.ListWorkspaceMembers:output_type -> openshell.v1.ListWorkspaceMembersResponse - 406, // [406:488] is the sub-list for method output_type - 324, // [324:406] is the sub-list for method input_type - 324, // [324:324] is the sub-list for extension type_name - 324, // [324:324] is the sub-list for extension extendee - 0, // [0:324] is the sub-list for field type_name + 275, // 319: openshell.v1.SandboxProvisioning.preparation_deadline:type_name -> google.protobuf.Timestamp + 275, // 320: openshell.v1.SandboxProvisioning.admission_start_time:type_name -> google.protobuf.Timestamp + 19, // 321: openshell.v1.SandboxServiceExposure.authorization_mode:type_name -> openshell.v1.ServiceAuthorizationMode + 275, // 322: openshell.v1.UpdateProviderRequest.CredentialExpirationTimesEntry.value:type_name -> google.protobuf.Timestamp + 275, // 323: openshell.v1.GetSandboxProviderEnvironmentResponse.CredentialExpirationTimesEntry.value:type_name -> google.protobuf.Timestamp + 127, // 324: openshell.v1.GetSandboxProviderEnvironmentResponse.DynamicCredentialsEntry.value:type_name -> openshell.v1.ProviderProfileCredential + 156, // 325: openshell.v1.GetSandboxProviderEnvironmentResponse.StaticCredentialBindingsEntry.value:type_name -> openshell.v1.StaticCredentialBinding + 24, // 326: openshell.v1.OpenShell.Health:input_type -> openshell.v1.HealthRequest + 26, // 327: openshell.v1.OpenShell.GetCurrentUser:input_type -> openshell.v1.GetCurrentUserRequest + 28, // 328: openshell.v1.OpenShell.GetGatewayInfo:input_type -> openshell.v1.GetGatewayInfoRequest + 52, // 329: openshell.v1.OpenShell.CreateSandbox:input_type -> openshell.v1.CreateSandboxRequest + 60, // 330: openshell.v1.OpenShell.BeginRootfsTarStaging:input_type -> openshell.v1.BeginRootfsTarStagingRequest + 62, // 331: openshell.v1.OpenShell.GetSandbox:input_type -> openshell.v1.GetSandboxRequest + 63, // 332: openshell.v1.OpenShell.ListSandboxes:input_type -> openshell.v1.ListSandboxesRequest + 53, // 333: openshell.v1.OpenShell.CreateSandboxTemplate:input_type -> openshell.v1.CreateSandboxTemplateRequest + 54, // 334: openshell.v1.OpenShell.GetSandboxTemplate:input_type -> openshell.v1.GetSandboxTemplateRequest + 55, // 335: openshell.v1.OpenShell.ListSandboxTemplates:input_type -> openshell.v1.ListSandboxTemplatesRequest + 56, // 336: openshell.v1.OpenShell.DeleteSandboxTemplate:input_type -> openshell.v1.DeleteSandboxTemplateRequest + 64, // 337: openshell.v1.OpenShell.ListSandboxProviders:input_type -> openshell.v1.ListSandboxProvidersRequest + 65, // 338: openshell.v1.OpenShell.AttachSandboxProvider:input_type -> openshell.v1.AttachSandboxProviderRequest + 66, // 339: openshell.v1.OpenShell.DetachSandboxProvider:input_type -> openshell.v1.DetachSandboxProviderRequest + 82, // 340: openshell.v1.OpenShell.GetSandboxProviderStatus:input_type -> openshell.v1.GetSandboxProviderStatusRequest + 67, // 341: openshell.v1.OpenShell.DeleteSandbox:input_type -> openshell.v1.DeleteSandboxRequest + 68, // 342: openshell.v1.OpenShell.StopSandbox:input_type -> openshell.v1.StopSandboxRequest + 69, // 343: openshell.v1.OpenShell.StartSandbox:input_type -> openshell.v1.StartSandboxRequest + 87, // 344: openshell.v1.OpenShell.CreateSshSession:input_type -> openshell.v1.CreateSshSessionRequest + 89, // 345: openshell.v1.OpenShell.ExposeService:input_type -> openshell.v1.ExposeServiceRequest + 90, // 346: openshell.v1.OpenShell.GetService:input_type -> openshell.v1.GetServiceRequest + 91, // 347: openshell.v1.OpenShell.ListServices:input_type -> openshell.v1.ListServicesRequest + 93, // 348: openshell.v1.OpenShell.DeleteService:input_type -> openshell.v1.DeleteServiceRequest + 97, // 349: openshell.v1.OpenShell.RevokeSshSession:input_type -> openshell.v1.RevokeSshSessionRequest + 99, // 350: openshell.v1.OpenShell.ExecSandbox:input_type -> openshell.v1.ExecSandboxRequest + 105, // 351: openshell.v1.OpenShell.ForwardTcp:input_type -> openshell.v1.TcpForwardFrame + 106, // 352: openshell.v1.OpenShell.ExecSandboxInteractive:input_type -> openshell.v1.ExecSandboxInput + 113, // 353: openshell.v1.OpenShell.CreateProvider:input_type -> openshell.v1.CreateProviderRequest + 114, // 354: openshell.v1.OpenShell.GetProvider:input_type -> openshell.v1.GetProviderRequest + 115, // 355: openshell.v1.OpenShell.ListProviders:input_type -> openshell.v1.ListProvidersRequest + 120, // 356: openshell.v1.OpenShell.ListProviderProfiles:input_type -> openshell.v1.ListProviderProfilesRequest + 121, // 357: openshell.v1.OpenShell.GetProviderProfile:input_type -> openshell.v1.GetProviderProfileRequest + 145, // 358: openshell.v1.OpenShell.ImportProviderProfiles:input_type -> openshell.v1.ImportProviderProfilesRequest + 147, // 359: openshell.v1.OpenShell.UpdateProviderProfiles:input_type -> openshell.v1.UpdateProviderProfilesRequest + 149, // 360: openshell.v1.OpenShell.LintProviderProfiles:input_type -> openshell.v1.LintProviderProfilesRequest + 116, // 361: openshell.v1.OpenShell.UpdateProvider:input_type -> openshell.v1.UpdateProviderRequest + 133, // 362: openshell.v1.OpenShell.GetProviderRefreshStatus:input_type -> openshell.v1.GetProviderRefreshStatusRequest + 135, // 363: openshell.v1.OpenShell.ConfigureProviderRefresh:input_type -> openshell.v1.ConfigureProviderRefreshRequest + 137, // 364: openshell.v1.OpenShell.RotateProviderCredential:input_type -> openshell.v1.RotateProviderCredentialRequest + 139, // 365: openshell.v1.OpenShell.DeleteProviderRefresh:input_type -> openshell.v1.DeleteProviderRefreshRequest + 117, // 366: openshell.v1.OpenShell.DeleteProvider:input_type -> openshell.v1.DeleteProviderRequest + 152, // 367: openshell.v1.OpenShell.DeleteProviderProfile:input_type -> openshell.v1.DeleteProviderProfileRequest + 290, // 368: openshell.v1.OpenShell.GetSandboxConfig:input_type -> openshell.sandbox.v1.GetSandboxConfigRequest + 291, // 369: openshell.v1.OpenShell.GetGatewayConfig:input_type -> openshell.sandbox.v1.GetGatewayConfigRequest + 160, // 370: openshell.v1.OpenShell.UpdateConfig:input_type -> openshell.v1.UpdateConfigRequest + 170, // 371: openshell.v1.OpenShell.GetSandboxPolicyStatus:input_type -> openshell.v1.GetSandboxPolicyStatusRequest + 172, // 372: openshell.v1.OpenShell.ListSandboxPolicies:input_type -> openshell.v1.ListSandboxPoliciesRequest + 174, // 373: openshell.v1.OpenShell.ReportPolicyStatus:input_type -> openshell.v1.ReportPolicyStatusRequest + 247, // 374: openshell.v1.OpenShell.ReportEndpointStatus:input_type -> openshell.v1.ReportEndpointStatusRequest + 84, // 375: openshell.v1.OpenShell.ReportProviderReadiness:input_type -> openshell.v1.ReportProviderReadinessRequest + 177, // 376: openshell.v1.OpenShell.ReportSandboxConfiguration:input_type -> openshell.v1.ReportSandboxConfigurationRequest + 154, // 377: openshell.v1.OpenShell.GetSandboxProviderEnvironment:input_type -> openshell.v1.GetSandboxProviderEnvironmentRequest + 158, // 378: openshell.v1.OpenShell.ExchangeProviderSubjectToken:input_type -> openshell.v1.ExchangeProviderSubjectTokenRequest + 180, // 379: openshell.v1.OpenShell.GetSandboxLogs:input_type -> openshell.v1.GetSandboxLogsRequest + 181, // 380: openshell.v1.OpenShell.PushSandboxLogs:input_type -> openshell.v1.PushSandboxLogsRequest + 184, // 381: openshell.v1.OpenShell.ConnectSupervisor:input_type -> openshell.v1.SupervisorMessage + 191, // 382: openshell.v1.OpenShell.ReportMainProcessExit:input_type -> openshell.v1.ReportMainProcessExitRequest + 193, // 383: openshell.v1.OpenShell.FinalizeMainProcessExit:input_type -> openshell.v1.FinalizeMainProcessExitRequest + 199, // 384: openshell.v1.OpenShell.RelayStream:input_type -> openshell.v1.RelayFrame + 201, // 385: openshell.v1.OpenShell.PeerRelay:input_type -> openshell.v1.PeerRelayFrame + 84, // 386: openshell.v1.OpenShell.PeerReportProviderReadiness:input_type -> openshell.v1.ReportProviderReadinessRequest + 247, // 387: openshell.v1.OpenShell.PeerReportEndpointStatus:input_type -> openshell.v1.ReportEndpointStatusRequest + 82, // 388: openshell.v1.OpenShell.PeerGetSandboxProviderStatus:input_type -> openshell.v1.GetSandboxProviderStatusRequest + 109, // 389: openshell.v1.OpenShell.WatchSandbox:input_type -> openshell.v1.WatchSandboxRequest + 210, // 390: openshell.v1.OpenShell.SubmitPolicyAnalysis:input_type -> openshell.v1.SubmitPolicyAnalysisRequest + 212, // 391: openshell.v1.OpenShell.GetDraftPolicy:input_type -> openshell.v1.GetDraftPolicyRequest + 214, // 392: openshell.v1.OpenShell.ApproveDraftChunk:input_type -> openshell.v1.ApproveDraftChunkRequest + 216, // 393: openshell.v1.OpenShell.RejectDraftChunk:input_type -> openshell.v1.RejectDraftChunkRequest + 219, // 394: openshell.v1.OpenShell.ApproveAllDraftChunks:input_type -> openshell.v1.ApproveAllDraftChunksRequest + 221, // 395: openshell.v1.OpenShell.EditDraftChunk:input_type -> openshell.v1.EditDraftChunkRequest + 223, // 396: openshell.v1.OpenShell.UndoDraftChunk:input_type -> openshell.v1.UndoDraftChunkRequest + 225, // 397: openshell.v1.OpenShell.ClearDraftChunks:input_type -> openshell.v1.ClearDraftChunksRequest + 227, // 398: openshell.v1.OpenShell.GetDraftHistory:input_type -> openshell.v1.GetDraftHistoryRequest + 20, // 399: openshell.v1.OpenShell.IssueSandboxToken:input_type -> openshell.v1.IssueSandboxTokenRequest + 22, // 400: openshell.v1.OpenShell.RefreshSandboxToken:input_type -> openshell.v1.RefreshSandboxTokenRequest + 230, // 401: openshell.v1.OpenShell.CreateWorkspace:input_type -> openshell.v1.CreateWorkspaceRequest + 232, // 402: openshell.v1.OpenShell.GetWorkspace:input_type -> openshell.v1.GetWorkspaceRequest + 234, // 403: openshell.v1.OpenShell.ListWorkspaces:input_type -> openshell.v1.ListWorkspacesRequest + 236, // 404: openshell.v1.OpenShell.DeleteWorkspace:input_type -> openshell.v1.DeleteWorkspaceRequest + 239, // 405: openshell.v1.OpenShell.AddWorkspaceMember:input_type -> openshell.v1.AddWorkspaceMemberRequest + 241, // 406: openshell.v1.OpenShell.RemoveWorkspaceMember:input_type -> openshell.v1.RemoveWorkspaceMemberRequest + 243, // 407: openshell.v1.OpenShell.ListWorkspaceMembers:input_type -> openshell.v1.ListWorkspaceMembersRequest + 25, // 408: openshell.v1.OpenShell.Health:output_type -> openshell.v1.HealthResponse + 27, // 409: openshell.v1.OpenShell.GetCurrentUser:output_type -> openshell.v1.GetCurrentUserResponse + 29, // 410: openshell.v1.OpenShell.GetGatewayInfo:output_type -> openshell.v1.GetGatewayInfoResponse + 70, // 411: openshell.v1.OpenShell.CreateSandbox:output_type -> openshell.v1.SandboxResponse + 61, // 412: openshell.v1.OpenShell.BeginRootfsTarStaging:output_type -> openshell.v1.BeginRootfsTarStagingResponse + 70, // 413: openshell.v1.OpenShell.GetSandbox:output_type -> openshell.v1.SandboxResponse + 71, // 414: openshell.v1.OpenShell.ListSandboxes:output_type -> openshell.v1.ListSandboxesResponse + 57, // 415: openshell.v1.OpenShell.CreateSandboxTemplate:output_type -> openshell.v1.SandboxTemplateResponse + 57, // 416: openshell.v1.OpenShell.GetSandboxTemplate:output_type -> openshell.v1.SandboxTemplateResponse + 58, // 417: openshell.v1.OpenShell.ListSandboxTemplates:output_type -> openshell.v1.ListSandboxTemplatesResponse + 59, // 418: openshell.v1.OpenShell.DeleteSandboxTemplate:output_type -> openshell.v1.DeleteSandboxTemplateResponse + 72, // 419: openshell.v1.OpenShell.ListSandboxProviders:output_type -> openshell.v1.ListSandboxProvidersResponse + 73, // 420: openshell.v1.OpenShell.AttachSandboxProvider:output_type -> openshell.v1.AttachSandboxProviderResponse + 74, // 421: openshell.v1.OpenShell.DetachSandboxProvider:output_type -> openshell.v1.DetachSandboxProviderResponse + 83, // 422: openshell.v1.OpenShell.GetSandboxProviderStatus:output_type -> openshell.v1.GetSandboxProviderStatusResponse + 86, // 423: openshell.v1.OpenShell.DeleteSandbox:output_type -> openshell.v1.DeleteSandboxResponse + 70, // 424: openshell.v1.OpenShell.StopSandbox:output_type -> openshell.v1.SandboxResponse + 70, // 425: openshell.v1.OpenShell.StartSandbox:output_type -> openshell.v1.SandboxResponse + 88, // 426: openshell.v1.OpenShell.CreateSshSession:output_type -> openshell.v1.CreateSshSessionResponse + 96, // 427: openshell.v1.OpenShell.ExposeService:output_type -> openshell.v1.ServiceEndpointResponse + 96, // 428: openshell.v1.OpenShell.GetService:output_type -> openshell.v1.ServiceEndpointResponse + 92, // 429: openshell.v1.OpenShell.ListServices:output_type -> openshell.v1.ListServicesResponse + 94, // 430: openshell.v1.OpenShell.DeleteService:output_type -> openshell.v1.DeleteServiceResponse + 98, // 431: openshell.v1.OpenShell.RevokeSshSession:output_type -> openshell.v1.RevokeSshSessionResponse + 103, // 432: openshell.v1.OpenShell.ExecSandbox:output_type -> openshell.v1.ExecSandboxEvent + 105, // 433: openshell.v1.OpenShell.ForwardTcp:output_type -> openshell.v1.TcpForwardFrame + 103, // 434: openshell.v1.OpenShell.ExecSandboxInteractive:output_type -> openshell.v1.ExecSandboxEvent + 118, // 435: openshell.v1.OpenShell.CreateProvider:output_type -> openshell.v1.ProviderResponse + 118, // 436: openshell.v1.OpenShell.GetProvider:output_type -> openshell.v1.ProviderResponse + 119, // 437: openshell.v1.OpenShell.ListProviders:output_type -> openshell.v1.ListProvidersResponse + 144, // 438: openshell.v1.OpenShell.ListProviderProfiles:output_type -> openshell.v1.ListProviderProfilesResponse + 143, // 439: openshell.v1.OpenShell.GetProviderProfile:output_type -> openshell.v1.ProviderProfileResponse + 146, // 440: openshell.v1.OpenShell.ImportProviderProfiles:output_type -> openshell.v1.ImportProviderProfilesResponse + 148, // 441: openshell.v1.OpenShell.UpdateProviderProfiles:output_type -> openshell.v1.UpdateProviderProfilesResponse + 150, // 442: openshell.v1.OpenShell.LintProviderProfiles:output_type -> openshell.v1.LintProviderProfilesResponse + 118, // 443: openshell.v1.OpenShell.UpdateProvider:output_type -> openshell.v1.ProviderResponse + 134, // 444: openshell.v1.OpenShell.GetProviderRefreshStatus:output_type -> openshell.v1.GetProviderRefreshStatusResponse + 136, // 445: openshell.v1.OpenShell.ConfigureProviderRefresh:output_type -> openshell.v1.ConfigureProviderRefreshResponse + 138, // 446: openshell.v1.OpenShell.RotateProviderCredential:output_type -> openshell.v1.RotateProviderCredentialResponse + 140, // 447: openshell.v1.OpenShell.DeleteProviderRefresh:output_type -> openshell.v1.DeleteProviderRefreshResponse + 151, // 448: openshell.v1.OpenShell.DeleteProvider:output_type -> openshell.v1.DeleteProviderResponse + 153, // 449: openshell.v1.OpenShell.DeleteProviderProfile:output_type -> openshell.v1.DeleteProviderProfileResponse + 292, // 450: openshell.v1.OpenShell.GetSandboxConfig:output_type -> openshell.sandbox.v1.GetSandboxConfigResponse + 293, // 451: openshell.v1.OpenShell.GetGatewayConfig:output_type -> openshell.sandbox.v1.GetGatewayConfigResponse + 169, // 452: openshell.v1.OpenShell.UpdateConfig:output_type -> openshell.v1.UpdateConfigResponse + 171, // 453: openshell.v1.OpenShell.GetSandboxPolicyStatus:output_type -> openshell.v1.GetSandboxPolicyStatusResponse + 173, // 454: openshell.v1.OpenShell.ListSandboxPolicies:output_type -> openshell.v1.ListSandboxPoliciesResponse + 175, // 455: openshell.v1.OpenShell.ReportPolicyStatus:output_type -> openshell.v1.ReportPolicyStatusResponse + 248, // 456: openshell.v1.OpenShell.ReportEndpointStatus:output_type -> openshell.v1.ReportEndpointStatusResponse + 85, // 457: openshell.v1.OpenShell.ReportProviderReadiness:output_type -> openshell.v1.ReportProviderReadinessResponse + 178, // 458: openshell.v1.OpenShell.ReportSandboxConfiguration:output_type -> openshell.v1.ReportSandboxConfigurationResponse + 157, // 459: openshell.v1.OpenShell.GetSandboxProviderEnvironment:output_type -> openshell.v1.GetSandboxProviderEnvironmentResponse + 159, // 460: openshell.v1.OpenShell.ExchangeProviderSubjectToken:output_type -> openshell.v1.ExchangeProviderSubjectTokenResponse + 183, // 461: openshell.v1.OpenShell.GetSandboxLogs:output_type -> openshell.v1.GetSandboxLogsResponse + 182, // 462: openshell.v1.OpenShell.PushSandboxLogs:output_type -> openshell.v1.PushSandboxLogsResponse + 185, // 463: openshell.v1.OpenShell.ConnectSupervisor:output_type -> openshell.v1.GatewayMessage + 192, // 464: openshell.v1.OpenShell.ReportMainProcessExit:output_type -> openshell.v1.ReportMainProcessExitResponse + 194, // 465: openshell.v1.OpenShell.FinalizeMainProcessExit:output_type -> openshell.v1.FinalizeMainProcessExitResponse + 199, // 466: openshell.v1.OpenShell.RelayStream:output_type -> openshell.v1.RelayFrame + 201, // 467: openshell.v1.OpenShell.PeerRelay:output_type -> openshell.v1.PeerRelayFrame + 85, // 468: openshell.v1.OpenShell.PeerReportProviderReadiness:output_type -> openshell.v1.ReportProviderReadinessResponse + 248, // 469: openshell.v1.OpenShell.PeerReportEndpointStatus:output_type -> openshell.v1.ReportEndpointStatusResponse + 83, // 470: openshell.v1.OpenShell.PeerGetSandboxProviderStatus:output_type -> openshell.v1.GetSandboxProviderStatusResponse + 110, // 471: openshell.v1.OpenShell.WatchSandbox:output_type -> openshell.v1.SandboxStreamEvent + 211, // 472: openshell.v1.OpenShell.SubmitPolicyAnalysis:output_type -> openshell.v1.SubmitPolicyAnalysisResponse + 213, // 473: openshell.v1.OpenShell.GetDraftPolicy:output_type -> openshell.v1.GetDraftPolicyResponse + 215, // 474: openshell.v1.OpenShell.ApproveDraftChunk:output_type -> openshell.v1.ApproveDraftChunkResponse + 217, // 475: openshell.v1.OpenShell.RejectDraftChunk:output_type -> openshell.v1.RejectDraftChunkResponse + 220, // 476: openshell.v1.OpenShell.ApproveAllDraftChunks:output_type -> openshell.v1.ApproveAllDraftChunksResponse + 222, // 477: openshell.v1.OpenShell.EditDraftChunk:output_type -> openshell.v1.EditDraftChunkResponse + 224, // 478: openshell.v1.OpenShell.UndoDraftChunk:output_type -> openshell.v1.UndoDraftChunkResponse + 226, // 479: openshell.v1.OpenShell.ClearDraftChunks:output_type -> openshell.v1.ClearDraftChunksResponse + 229, // 480: openshell.v1.OpenShell.GetDraftHistory:output_type -> openshell.v1.GetDraftHistoryResponse + 21, // 481: openshell.v1.OpenShell.IssueSandboxToken:output_type -> openshell.v1.IssueSandboxTokenResponse + 23, // 482: openshell.v1.OpenShell.RefreshSandboxToken:output_type -> openshell.v1.RefreshSandboxTokenResponse + 231, // 483: openshell.v1.OpenShell.CreateWorkspace:output_type -> openshell.v1.CreateWorkspaceResponse + 233, // 484: openshell.v1.OpenShell.GetWorkspace:output_type -> openshell.v1.GetWorkspaceResponse + 235, // 485: openshell.v1.OpenShell.ListWorkspaces:output_type -> openshell.v1.ListWorkspacesResponse + 237, // 486: openshell.v1.OpenShell.DeleteWorkspace:output_type -> openshell.v1.DeleteWorkspaceResponse + 240, // 487: openshell.v1.OpenShell.AddWorkspaceMember:output_type -> openshell.v1.AddWorkspaceMemberResponse + 242, // 488: openshell.v1.OpenShell.RemoveWorkspaceMember:output_type -> openshell.v1.RemoveWorkspaceMemberResponse + 244, // 489: openshell.v1.OpenShell.ListWorkspaceMembers:output_type -> openshell.v1.ListWorkspaceMembersResponse + 408, // [408:490] is the sub-list for method output_type + 326, // [326:408] is the sub-list for method input_type + 326, // [326:326] is the sub-list for extension type_name + 326, // [326:326] is the sub-list for extension extendee + 0, // [0:326] is the sub-list for field type_name } func init() { file_openshell_proto_init() } From 71c3cd957abef062eb7f37010056717cd49f2ed3 Mon Sep 17 00:00:00 2001 From: Shiju Date: Sat, 3 Oct 2026 21:22:33 +0000 Subject: [PATCH 13/13] fix(gateway): check launch signing before VM image preparation (#4034) * fix(vm): validate launch credentials before preparation Negotiate a launch-authentication requirement and reject incomplete gateway configuration before driver validation, archive consumption or persistence. Validate replacement VM credentials before stopping active compute and prefer explicit signer configuration over local discovery. Closes #3949. Part of #3955. Signed-off-by: Shiju * fix(vm): validate saved launch credentials before restore side effects Signed-off-by: Shiju * chore(gateway): satisfy launch preflight lint checks Signed-off-by: Shiju * docs(gateway): complete the MicroVM launch-signing example Include the gateway JWT paths in the standalone VM configuration and show the output directory required by local certificate generation. Explain the distinction between configuration preflight and launch-key validation. Refs #3949. Part of #3955. Signed-off-by: Shiju * test(vm): account for preparation state in launch credential fixture Signed-off-by: John Myers <9696606+johntmyers@users.noreply.github.com> --------- Signed-off-by: Shiju Signed-off-by: John Myers <9696606+johntmyers@users.noreply.github.com> Co-authored-by: John Myers <9696606+johntmyers@users.noreply.github.com> --- .../openshell-core/src/extension_protocol.rs | 13 +- crates/openshell-driver-vm/src/driver.rs | 404 ++++++++++++++++-- .../src/auth/launch_signing.rs | 201 +++++++++ crates/openshell-server/src/auth/mod.rs | 1 + crates/openshell-server/src/cli.rs | 67 ++- crates/openshell-server/src/compute/mod.rs | 132 +++++- crates/openshell-server/src/defaults.rs | 6 +- crates/openshell-server/src/grpc/sandbox.rs | 111 +++++ crates/openshell-server/src/lib.rs | 53 +-- crates/openshell-server/src/test_support.rs | 7 + docs/how-it-works/gateways/configuration.mdx | 26 ++ 11 files changed, 919 insertions(+), 102 deletions(-) create mode 100644 crates/openshell-server/src/auth/launch_signing.rs diff --git a/crates/openshell-core/src/extension_protocol.rs b/crates/openshell-core/src/extension_protocol.rs index 9b5a8f2a97..9c25c6cb87 100644 --- a/crates/openshell-core/src/extension_protocol.rs +++ b/crates/openshell-core/src/extension_protocol.rs @@ -12,6 +12,13 @@ use crate::proto::extension::v1::{PeerMetadata, ProtocolVersion}; pub const PROTOCOL_MAJOR: u32 = 1; pub const PROTOCOL_MINOR: u32 = 0; +/// Capability for compute drivers that require gateway-minted launch credentials. +/// +/// Drivers require this capability when every launch needs a +/// [`crate::jwt::SandboxLaunchAuthentication`] bundle. The gateway checks its +/// configured signer before accepting sandbox creation. +pub const COMPUTE_LAUNCH_AUTHENTICATION: &str = "openshell.compute.launch-authentication"; + const MAX_IMPLEMENTATION_NAME_BYTES: usize = 128; const MAX_IMPLEMENTATION_VERSION_BYTES: usize = 128; const MAX_CAPABILITY_BYTES: usize = 128; @@ -103,6 +110,10 @@ pub enum NegotiationError { #[must_use] pub fn gateway_metadata(family: ExtensionFamily) -> PeerMetadata { let contract = family.contract_capability(); + let mut supported_capabilities = vec![contract.clone()]; + if family == ExtensionFamily::Compute { + supported_capabilities.push(COMPUTE_LAUNCH_AUTHENTICATION.to_string()); + } PeerMetadata { protocol_version: Some(ProtocolVersion { major: PROTOCOL_MAJOR, @@ -110,7 +121,7 @@ pub fn gateway_metadata(family: ExtensionFamily) -> PeerMetadata { }), implementation_name: "openshell/gateway".to_string(), implementation_version: crate::VERSION.to_string(), - supported_capabilities: vec![contract.clone()], + supported_capabilities, required_capabilities: vec![contract], } } diff --git a/crates/openshell-driver-vm/src/driver.rs b/crates/openshell-driver-vm/src/driver.rs index f737b6e79e..799cb0b6e8 100644 --- a/crates/openshell-driver-vm/src/driver.rs +++ b/crates/openshell-driver-vm/src/driver.rs @@ -1034,6 +1034,15 @@ impl VmDriver { #[must_use] pub fn capabilities(&self) -> GetCapabilitiesResponse { + let mut extension = openshell_core::extension_protocol::extension_metadata( + openshell_core::extension_protocol::ExtensionFamily::Compute, + "openshell/vm", + openshell_core::VERSION, + [], + ); + extension + .required_capabilities + .push(openshell_core::extension_protocol::COMPUTE_LAUNCH_AUTHENTICATION.to_string()); GetCapabilitiesResponse { resource_admission_policy: openshell_core::resource_admission::DriverAdmissionConfig { allow_driver_config: self.config.allow_driver_config, @@ -1064,12 +1073,7 @@ impl VmDriver { .to_string_lossy() .into_owned(), rootfs_tar_max_bytes: self.config.rootfs_tar_max_bytes(), - extension: Some(openshell_core::extension_protocol::extension_metadata( - openshell_core::extension_protocol::ExtensionFamily::Compute, - "openshell/vm", - openshell_core::VERSION, - [], - )), + extension: Some(extension), } } @@ -1097,6 +1101,12 @@ impl VmDriver { #[allow(clippy::result_large_err)] pub async fn create_sandbox(&self, sandbox: &Sandbox) -> Result { self.validate_sandbox(sandbox)?; + decode_launch_authentication( + sandbox + .spec + .as_ref() + .map_or(&[], |spec| spec.launch_authentication.as_slice()), + )?; info!( sandbox_id = %sandbox.id, sandbox_name = %sandbox.name, @@ -1288,6 +1298,14 @@ impl VmDriver { overlay_preparation: OverlayPreparation, ) -> Result<(), Status> { self.ensure_provisioning_active(&sandbox.id).await?; + // Launch credentials are independent of the image. Validate them before + // resolving registry references, preparing disks, or publishing progress. + let launch_authentication = decode_launch_authentication( + sandbox + .spec + .as_ref() + .map_or(&[], |spec| spec.launch_authentication.as_slice()), + )?; let is_gpu = sandbox .spec .as_ref() @@ -1373,29 +1391,6 @@ impl VmDriver { ))); } }; - let launch_authentication = sandbox - .spec - .as_ref() - .filter(|spec| !spec.launch_authentication.is_empty()) - .ok_or_else(|| { - Status::failed_precondition("VM sandbox launch authentication is required") - }) - .and_then(|spec| { - serde_json::from_slice::( - &spec.launch_authentication, - ) - .map_err(|error| { - Status::failed_precondition(format!( - "decode VM sandbox launch authentication: {error}" - )) - }) - })?; - launch_authentication.validate().map_err(|error| { - Status::failed_precondition(format!( - "validate VM sandbox launch authentication: {error}" - )) - })?; - self.publish_platform_event( sandbox.id.clone(), platform_event( @@ -1933,6 +1928,10 @@ impl VmDriver { return Ok(()); } } + // An invalid replacement must not stop the active VM or remove the + // previous generation's credentials. Preserve the empty idempotent + // replay above, which does not request a new launch. + decode_launch_authentication(&launch_authentication)?; let mut sandbox = read_sandbox_request(&state_dir.join(SANDBOX_REQUEST_FILE)) .await .map_err(|error| { @@ -1960,17 +1959,6 @@ impl VmDriver { ) .await .map_err(|error| Status::internal(format!("persist VM start generation: {error}")))?; - let authentication = serde_json::from_slice::< - openshell_core::jwt::SandboxLaunchAuthentication, - >(&launch_authentication) - .map_err(|error| { - Status::failed_precondition(format!("decode VM sandbox launch authentication: {error}")) - })?; - authentication.validate().map_err(|error| { - Status::failed_precondition(format!( - "validate VM sandbox launch authentication: {error}" - )) - })?; let spec = sandbox .spec .as_mut() @@ -2271,6 +2259,17 @@ impl VmDriver { clear_stop_marker: bool, reconciliation_span: &tracing::Span, ) -> bool { + // Startup recovery bypasses create/start admission. Validate saved + // credentials before host preparation, extension hooks, or state changes. + if let Err(error) = decode_launch_authentication( + sandbox + .spec + .as_ref() + .map_or(&[], |spec| spec.launch_authentication.as_slice()), + ) { + warn!(sandbox_id = %sandbox.id, reason = %error.message(), "VM recovery denied by launch authentication"); + return false; + } if let Err(error) = self.validate_sandbox(&sandbox) { warn!(sandbox_id = %sandbox.id, reason = %error.message(), "VM recovery denied by admission"); return false; @@ -4902,6 +4901,29 @@ fn check_gpu_privileges() -> Result<(), String> { // `tonic::Status` is ~176 bytes; it's the standard error type across the // gRPC API surface, so boxing here would diverge from every other handler. +#[allow(clippy::result_large_err)] +fn decode_launch_authentication( + encoded: &[u8], +) -> Result { + if encoded.is_empty() { + return Err(Status::failed_precondition( + "VM sandbox launch authentication is required; configure the gateway's \ + [openshell.gateway.gateway_jwt] signing bundle. Listener TLS is separate", + )); + } + // Serde errors may quote an unexpected field or value from the secret + // bundle. Return fixed diagnostics rather than forwarding those details. + let authentication = + serde_json::from_slice::(encoded) + .map_err(|_| { + Status::failed_precondition("VM sandbox launch authentication is malformed") + })?; + authentication.validate().map_err(|_| { + Status::failed_precondition("VM sandbox launch authentication has invalid fields") + })?; + Ok(authentication) +} + #[allow(clippy::result_large_err)] fn validate_vm_sandbox(sandbox: &Sandbox, gpu_enabled: bool) -> Result<(), Status> { validate_sandbox_id(&sandbox.id)?; @@ -7790,6 +7812,7 @@ mod tests { id: "sb-spawned-trace".to_string(), name: "spawned-trace".to_string(), spec: Some(SandboxSpec { + launch_authentication: test_launch_authentication("spawned-trace").0, template: Some(SandboxTemplate { image: "invalid image reference".to_string(), ..Default::default() @@ -7839,6 +7862,7 @@ mod tests { id: format!("sb-restored-trace-{suffix}"), name: format!("restored-trace-{suffix}"), spec: Some(SandboxSpec { + launch_authentication: test_launch_authentication(suffix).0, template: Some(SandboxTemplate { image: "invalid image reference".to_string(), ..Default::default() @@ -9919,6 +9943,79 @@ mod tests { ); } + #[tokio::test] + async fn malformed_start_preserves_active_and_stopped_generation_material() { + for active in [false, true] { + let directory = tempfile::tempdir().unwrap(); + let mut driver = test_driver_with_extensions(LifecycleExtensionRegistry::new()); + driver.config.state_dir = directory.path().to_path_buf(); + driver.config.default_image = "test/image:latest".to_string(); + let sandbox = Sandbox { + id: "sb-auth-start".to_string(), + name: "auth-start".to_string(), + spec: Some(SandboxSpec { + launch_authentication: test_launch_authentication("existing").0, + ..Default::default() + }), + ..Default::default() + }; + let state_dir = directory.path().join("sandbox"); + create_private_dir_all(&state_dir).await.unwrap(); + write_sandbox_request(&state_dir, &sandbox).await.unwrap(); + // An empty marker on the stopped fixture avoids invoking debugfs if + // the guard regresses, so the test reaches the destructive host cleanup. + let generation = if active { "g0000000000000001" } else { "" }; + std::fs::write(state_dir.join(HOST_BOUNDARY_GENERATION_FILE), generation).unwrap(); + std::fs::write( + state_dir.join(HOST_AUTH_BUNDLE_FILE), + b"retained authentication", + ) + .unwrap(); + let task = active.then(|| tokio::spawn(std::future::pending())); + driver.registry.lock().await.insert( + sandbox.id.clone(), + SandboxRecord { + snapshot: sandbox.clone(), + state_dir: state_dir.clone(), + process: None, + provisioning_task: task, + preparation: None, + gpu_bdf: None, + deleting: false, + }, + ); + let error = driver + .start_sandbox( + &sandbox.id, + &sandbox.name, + "g0000000000000001", + br#"{"secret-marker-that-must-not-be-logged":true}"#.to_vec(), + ) + .await + .unwrap_err(); + assert_eq!(error.code(), Code::FailedPrecondition); + assert_eq!( + error.message(), + "VM sandbox launch authentication is malformed" + ); + assert_eq!( + std::fs::read_to_string(state_dir.join(HOST_BOUNDARY_GENERATION_FILE)).unwrap(), + generation + ); + assert_eq!( + std::fs::read(state_dir.join(HOST_AUTH_BUNDLE_FILE)).unwrap(), + b"retained authentication" + ); + let record = driver.registry.lock().await.remove(&sandbox.id).unwrap(); + assert_eq!(record.snapshot, sandbox); + assert_eq!(record.provisioning_task.is_some(), active); + if let Some(task) = record.provisioning_task { + assert!(!task.is_finished()); + task.abort(); + } + } + } + fn test_launch_authentication(label: &str) -> (Vec, openshell_core::SandboxSessionId) { use openshell_core::jwt::{ SandboxLaunchAuthentication, SecretJwt, SessionVerificationKey, SupervisorAuthBundle, @@ -10049,6 +10146,228 @@ mod tests { }; assert_eq!(driver.capabilities().default_image, "openshell/sandbox:dev"); + let gateway = openshell_core::extension_protocol::gateway_metadata( + openshell_core::extension_protocol::ExtensionFamily::Compute, + ); + let negotiated = openshell_core::extension_protocol::negotiate( + openshell_core::extension_protocol::ExtensionFamily::Compute, + "vm", + &gateway, + driver.capabilities().extension, + ) + .unwrap(); + assert!(negotiated.required_capabilities.iter().any(|capability| { + capability == openshell_core::extension_protocol::COMPUTE_LAUNCH_AUTHENTICATION + })); + } + + #[tokio::test] + async fn launch_authentication_is_checked_before_create_side_effects() { + let directory = tempfile::tempdir().unwrap(); + let mut driver = test_driver_with_extensions(LifecycleExtensionRegistry::new()); + driver.config.state_dir = directory.path().to_path_buf(); + driver.config.default_image = "invalid image reference".to_string(); + let mut events = driver.events.subscribe(); + let mut invalid_fields: serde_json::Value = + serde_json::from_slice(&test_launch_authentication("invalid-fields").0).unwrap(); + invalid_fields["verification_keys"] = serde_json::json!([]); + let malformed = br#"{"secret-marker-that-must-not-be-logged":true}"#.to_vec(); + + for (authentication, diagnostic) in [ + (Vec::new(), "is required"), + (malformed, "is malformed"), + ( + serde_json::to_vec(&invalid_fields).unwrap(), + "has invalid fields", + ), + ] { + let sandbox = Sandbox { + id: "sb-auth-preflight".to_string(), + name: "auth-preflight".to_string(), + spec: Some(SandboxSpec { + launch_authentication: authentication, + ..Default::default() + }), + ..Default::default() + }; + let error = driver + .create_sandbox(&sandbox) + .await + .expect_err("invalid launch material must fail synchronously"); + assert_eq!(error.code(), Code::FailedPrecondition); + assert!(error.message().contains(diagnostic), "{error}"); + assert!(!error.message().contains("secret-marker")); + assert!(driver.registry.lock().await.is_empty()); + assert!(directory.path().read_dir().unwrap().next().is_none()); + assert!(matches!( + events.try_recv(), + Err(broadcast::error::TryRecvError::Empty) + )); + } + } + + #[tokio::test] + async fn provisioning_rechecks_launch_authentication_before_resolving_images() { + let directory = tempfile::tempdir().unwrap(); + let mut driver = test_driver_with_extensions(LifecycleExtensionRegistry::new()); + driver.config.state_dir = directory.path().to_path_buf(); + driver.config.bootstrap_image = "invalid bootstrap image reference".to_string(); + let sandbox = Sandbox { + id: "sb-auth-recovery".to_string(), + name: "auth-recovery".to_string(), + spec: Some(SandboxSpec::default()), + ..Default::default() + }; + let state_dir = directory.path().join("sandbox"); + driver.registry.lock().await.insert( + sandbox.id.clone(), + SandboxRecord { + snapshot: sandbox.clone(), + state_dir: state_dir.clone(), + process: None, + provisioning_task: None, + preparation: None, + gpu_bdf: None, + deleting: false, + }, + ); + let mut events = driver.events.subscribe(); + let error = driver + .provision_sandbox_inner( + sandbox, + "invalid image reference".to_string(), + state_dir, + None, + OverlayPreparation::PreserveExisting, + ) + .await + .unwrap_err(); + assert!( + error + .message() + .contains("VM sandbox launch authentication is required") + ); + assert!(matches!( + events.try_recv(), + Err(broadcast::error::TryRecvError::Empty) + )); + assert!(directory.path().read_dir().unwrap().next().is_none()); + } + + #[tokio::test] + async fn recovery_rejects_invalid_launch_authentication_before_side_effects() { + #[derive(Debug, Default)] + struct RestoreObserver { + calls: AtomicUsize, + } + + #[tonic::async_trait] + impl LifecycleExtension for RestoreObserver { + fn name(&self) -> &'static str { + "restore-observer" + } + + async fn before_restore(&self, _sandbox: &RestoreContext) -> LifecycleResult<()> { + self.calls.fetch_add(1, Ordering::Relaxed); + Ok(()) + } + } + + for scan_at_startup in [true, false] { + for authentication in [ + Vec::new(), + br#"{"secret-marker-that-must-not-be-logged":true}"#.to_vec(), + ] { + let directory = tempfile::tempdir().unwrap(); + let observer = Arc::new(RestoreObserver::default()); + let mut driver = + test_driver_with_extensions(LifecycleExtensionRegistry::with(vec![ + observer.clone(), + ])); + driver.config.state_dir = directory.path().to_path_buf(); + driver.config.default_image = "invalid image reference".to_string(); + let sandbox = Sandbox { + id: "sb-auth-restore".to_string(), + name: "auth-restore".to_string(), + spec: Some(SandboxSpec { + launch_authentication: authentication, + ..Default::default() + }), + ..Default::default() + }; + let state_dir = sandbox_state_dir(directory.path(), &sandbox.id).unwrap(); + create_private_dir_all(&state_dir).await.unwrap(); + write_sandbox_request(&state_dir, &sandbox).await.unwrap(); + fs::write(state_dir.join("overlay.ext4"), b"retained overlay").unwrap(); + fs::write(state_dir.join(HOST_AUTH_BUNDLE_FILE), b"retained auth").unwrap(); + if !scan_at_startup { + fs::write(state_dir.join(SANDBOX_STOPPED_FILE), b"stopped\n").unwrap(); + } + let mut original_files = fs::read_dir(&state_dir) + .unwrap() + .map(|entry| { + let entry = entry.unwrap(); + (entry.file_name(), fs::read(entry.path()).unwrap()) + }) + .collect::>(); + original_files.sort(); + let mut events = driver.events.subscribe(); + + let accepted = if scan_at_startup { + driver.restore_persisted_sandboxes().await; + false + } else { + driver + .restore_persisted_sandbox( + sandbox.clone(), + state_dir.clone(), + true, + &tracing::Span::current(), + ) + .await + }; + // If validation regresses, join the failed provisioning task so + // its later cleanup cannot race with the state assertions. + let task = driver + .registry + .lock() + .await + .get_mut(&sandbox.id) + .and_then(|record| record.provisioning_task.take()); + if let Some(task) = task { + task.await.unwrap(); + } + + assert!( + !accepted, + "invalid launch credentials must deny restoration" + ); + assert_eq!( + observer.calls.load(Ordering::Relaxed), + 0, + "launch authentication must precede lifecycle extension hooks" + ); + assert!(driver.registry.lock().await.is_empty()); + assert!(matches!( + events.try_recv(), + Err(broadcast::error::TryRecvError::Empty) + )); + assert!( + !extension_state_dir(&state_dir, "restore-observer") + .unwrap() + .exists() + ); + let mut retained_files = fs::read_dir(&state_dir) + .unwrap() + .map(|entry| { + let entry = entry.unwrap(); + (entry.file_name(), fs::read(entry.path()).unwrap()) + }) + .collect::>(); + retained_files.sort(); + assert_eq!(retained_files, original_files); + } + } } #[test] @@ -11025,7 +11344,10 @@ mod tests { .create_sandbox(&Sandbox { id: "sandbox-123".to_string(), name: "sandbox-123".to_string(), - spec: Some(SandboxSpec::default()), + spec: Some(SandboxSpec { + launch_authentication: test_launch_authentication("duplicate").0, + ..Default::default() + }), ..Default::default() }) .await diff --git a/crates/openshell-server/src/auth/launch_signing.rs b/crates/openshell-server/src/auth/launch_signing.rs new file mode 100644 index 0000000000..a1b510bb24 --- /dev/null +++ b/crates/openshell-server/src/auth/launch_signing.rs @@ -0,0 +1,201 @@ +// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +//! Load and validate sandbox launch signing before gateway startup connects drivers. + +use std::sync::Arc; + +use openshell_core::{Error, GatewayJwtConfig}; + +use super::sandbox_jwt::{ExtensionJwtIssuer, SandboxSessionJwtAuthority}; + +pub struct LaunchSigningAuthorities { + pub extension: Arc, + pub sandbox_session: Arc, +} + +pub fn load(config: &GatewayJwtConfig) -> openshell_core::Result { + let signing_pem = std::fs::read(&config.signing_key_path).map_err(|error| { + Error::config(format!( + "cannot read sandbox launch-signing private key from {}: {error}", + config.signing_key_path.display() + )) + })?; + let public_pem = std::fs::read(&config.public_key_path).map_err(|error| { + Error::config(format!( + "cannot read sandbox launch-signing public key from {}: {error}", + config.public_key_path.display() + )) + })?; + let kid = std::fs::read_to_string(&config.kid_path) + .map_err(|error| { + Error::config(format!( + "cannot read sandbox launch-signing key ID from {}: {error}", + config.kid_path.display() + )) + })? + .trim() + .to_string(); + if kid.is_empty() { + return Err(Error::config(format!( + "sandbox launch-signing key ID file {} is empty", + config.kid_path.display() + ))); + } + let extension = ExtensionJwtIssuer::from_pem( + &signing_pem, + &public_pem, + kid.clone(), + &config.gateway_id, + config.token_ttl(), + ) + .map_err(|_| { + Error::config( + "invalid sandbox launch-signing bundle: expected Ed25519 private and public keys", + ) + })?; + let sandbox_session = SandboxSessionJwtAuthority::from_pem( + &signing_pem, + &public_pem, + kid, + &config.gateway_id, + config.sandbox_token_ttl(), + ) + .map_err(|error| { + // This constructor maps core SessionJwtError variants to fixed text; + // preserve their actionable field and lifetime diagnostics. + Error::config(format!("invalid sandbox launch-signing bundle: {error}")) + })?; + + // Parsing two keys does not establish that they are a pair. Sign and verify + // a local probe so mismatched keys fail before a driver prepares an image. + // The probe is never persisted or sent to a driver or supervisor. + let identity = super::sandbox_session::PersistedSandboxIdentity::new() + .map_err(|_| Error::config("cannot initialize sandbox launch-signing validation"))?; + let probe = sandbox_session + .mint_persisted_launch("launch-signing-preflight", &identity) + .map_err(|_| Error::config("sandbox launch-signing key cannot mint session credentials"))?; + sandbox_session + .verify_gateway_token(probe.supervisor.gateway_token.expose_secret()) + .map_err(|_| { + Error::config("sandbox launch-signing private and public keys do not match") + })?; + + Ok(LaunchSigningAuthorities { + extension: Arc::new(extension), + sandbox_session: Arc::new(sandbox_session), + }) +} + +#[cfg(test)] +mod tests { + use super::*; + use openshell_bootstrap::jwt::generate_jwt_key; + + fn bundle(directory: &std::path::Path) -> GatewayJwtConfig { + let material = generate_jwt_key().expect("generate signing material"); + std::fs::create_dir_all(directory).unwrap(); + let config = GatewayJwtConfig { + signing_key_path: directory.join("signing.pem"), + public_key_path: directory.join("public.pem"), + kid_path: directory.join("kid"), + gateway_id: "test-gateway".to_string(), + ttl_secs: None, + }; + std::fs::write(&config.signing_key_path, material.signing_key_pem).unwrap(); + std::fs::write(&config.public_key_path, material.public_key_pem).unwrap(); + std::fs::write(&config.kid_path, material.kid).unwrap(); + config + } + + fn failure(config: &GatewayJwtConfig) -> String { + match load(config) { + Ok(_) => panic!("invalid bundle must fail before driver startup"), + Err(error) => error.to_string(), + } + } + + #[test] + fn explicit_bundle_mints_and_verifies_launch_credentials() { + let directory = tempfile::tempdir().unwrap(); + let config = bundle(directory.path()); + assert!(load(&config).is_ok()); + } + + #[test] + fn discovered_bundle_mints_and_verifies_launch_credentials() { + let directory = tempfile::tempdir().unwrap(); + bundle(&directory.path().join("jwt")); + let config = crate::defaults::local_jwt_config(directory.path()) + .unwrap() + .unwrap(); + assert!(load(&config).is_ok()); + } + + #[test] + fn missing_signing_file_names_the_required_file() { + let directory = tempfile::tempdir().unwrap(); + let config = bundle(directory.path()); + std::fs::remove_file(&config.signing_key_path).unwrap(); + let error = failure(&config); + assert!(error.contains("cannot read sandbox launch-signing private key")); + assert!(error.contains("signing.pem")); + } + + #[test] + fn invalid_signing_metadata_keeps_specific_diagnostics() { + let directory = tempfile::tempdir().unwrap(); + let mut config = bundle(directory.path()); + config.ttl_secs = std::num::NonZeroU64::new(1); + assert!(failure(&config).contains("between 60 and 3600 seconds")); + config.ttl_secs = None; + config.gateway_id = "invalid gateway secret-marker".to_string(); + let error = failure(&config); + assert!(error.contains("gateway ID is invalid")); + assert!(!error.contains("secret-marker")); + config.gateway_id = "valid-gateway".to_string(); + std::fs::write(&config.kid_path, "invalid kid secret-marker").unwrap(); + let error = failure(&config); + assert!(error.contains("key ID is invalid")); + assert!(!error.contains("secret-marker")); + } + + #[test] + fn missing_public_key_and_key_id_name_the_required_file() { + let directory = tempfile::tempdir().unwrap(); + let config = bundle(directory.path()); + std::fs::remove_file(&config.public_key_path).unwrap(); + assert!(failure(&config).contains("cannot read sandbox launch-signing public key")); + let config = bundle(directory.path()); + std::fs::remove_file(&config.kid_path).unwrap(); + assert!(failure(&config).contains("cannot read sandbox launch-signing key ID")); + } + + #[test] + fn malformed_signing_material_is_not_in_diagnostics() { + let directory = tempfile::tempdir().unwrap(); + let config = bundle(directory.path()); + let marker = "secret-marker-that-must-not-be-logged"; + std::fs::write(&config.signing_key_path, marker).unwrap(); + let error = failure(&config); + assert!(error.contains("invalid sandbox launch-signing bundle")); + assert!(!error.contains(marker)); + } + + #[test] + fn empty_key_id_is_reported_before_driver_startup() { + let directory = tempfile::tempdir().unwrap(); + let config = bundle(directory.path()); + std::fs::write(&config.kid_path, " \n").unwrap(); + assert!(failure(&config).contains("is empty")); + } + + #[test] + fn mismatched_keys_are_rejected_before_driver_startup() { + let directory = tempfile::tempdir().unwrap(); + let config = bundle(directory.path()); + let other = generate_jwt_key().unwrap(); + std::fs::write(&config.public_key_path, other.public_key_pem).unwrap(); + assert!(failure(&config).contains("private and public keys do not match")); + } +} diff --git a/crates/openshell-server/src/auth/mod.rs b/crates/openshell-server/src/auth/mod.rs index 3994ee537b..da54287744 100644 --- a/crates/openshell-server/src/auth/mod.rs +++ b/crates/openshell-server/src/auth/mod.rs @@ -16,6 +16,7 @@ pub mod extension_mint_limit; pub mod guard; mod http; pub mod identity; +pub mod launch_signing; pub mod method_authz; pub mod oidc; pub mod peer; diff --git a/crates/openshell-server/src/cli.rs b/crates/openshell-server/src/cli.rs index bdba1e32c0..bfe237af39 100644 --- a/crates/openshell-server/src/cli.rs +++ b/crates/openshell-server/src/cli.rs @@ -368,7 +368,15 @@ fn prepare_server_config_with_drivers( args.disable_tls, ) .map_err(|error| miette::miette!("invalid gateway guest TLS configuration: {error}"))?; - let local_jwt = defaults::complete_local_jwt_config()?; + // Explicit signing configuration must not depend on an unrelated, partial + // local bundle left by a package-managed installation. + let explicit_jwt = file + .as_ref() + .and_then(|file| file.openshell.gateway.gateway_jwt.clone()); + let gateway_jwt = match explicit_jwt { + Some(jwt) => Some(jwt), + None => defaults::complete_local_jwt_config()?, + }; let bind = SocketAddr::new(args.bind_address, args.port); @@ -586,14 +594,7 @@ fn prepare_server_config_with_drivers( // package-managed starts also auto-detect the JWT bundle written next to // the generated TLS bundle so upgrades pick up sandbox auth without a // user-authored config file. - if let Some(jwt) = file - .as_ref() - .and_then(|f| f.openshell.gateway.gateway_jwt.clone()) - { - config.gateway_jwt = Some(jwt); - } else if let Some(jwt) = local_jwt { - config.gateway_jwt = Some(jwt); - } + config.gateway_jwt = gateway_jwt; Ok(ServerStartupConfig { config, @@ -3389,4 +3390,52 @@ mem_mib = "not-a-number" ); } } + + #[test] + fn explicit_launch_signing_config_ignores_partial_local_bundle() { + let _lock = ENV_LOCK + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner); + let directory = tempfile::tempdir().unwrap(); + let _state = EnvVarGuard::set("XDG_STATE_HOME", directory.path().to_str().unwrap()); + let _local = EnvVarGuard::set( + "OPENSHELL_LOCAL_TLS_DIR", + directory.path().to_str().unwrap(), + ); + std::fs::create_dir(directory.path().join("jwt")).unwrap(); + std::fs::write( + directory.path().join("jwt/signing.pem"), + "incomplete local bundle", + ) + .unwrap(); + let config_path = directory.path().join("gateway.toml"); + std::fs::write( + &config_path, + r#" +[openshell] +version = 2 +[openshell.gateway.gateway_jwt] +signing_key_path = "/explicit/signing.pem" +public_key_path = "/explicit/public.pem" +kid_path = "/explicit/kid" +gateway_id = "explicit-gateway" +"#, + ) + .unwrap(); + let (mut args, matches) = parse_with_args(&[ + "openshell-gateway", + "--config", + config_path.to_str().unwrap(), + "--db-url", + "sqlite::memory:", + "--compute-driver", + "podman", + "--disable-tls", + ]); + let prepared = super::prepare_server_config(&mut args, &matches).unwrap(); + assert_eq!( + prepared.config.gateway_jwt.unwrap().gateway_id, + "explicit-gateway" + ); + } } diff --git a/crates/openshell-server/src/compute/mod.rs b/crates/openshell-server/src/compute/mod.rs index 52a15c6300..2c84c147c3 100644 --- a/crates/openshell-server/src/compute/mod.rs +++ b/crates/openshell-server/src/compute/mod.rs @@ -1032,6 +1032,29 @@ impl ComputeRuntime { .await } + /// Check the selected driver's launch contract without contacting the driver. + pub(crate) fn validate_launch_signer_configured(&self, configured: bool) -> Result<(), Status> { + // AuthenticateSandbox handles driver-native bootstrap credentials. It + // does not declare whether launches require a gateway signing bundle. + let required = self + .driver_info + .negotiated_extension + .required_capabilities + .iter() + .any(|capability| { + capability == openshell_core::extension_protocol::COMPUTE_LAUNCH_AUTHENTICATION + }); + if required && !configured { + return Err(Status::failed_precondition( + "the selected compute driver requires sandbox launch signing; configure \ + [openshell.gateway.gateway_jwt] or provide jwt/signing.pem, jwt/public.pem, \ + and jwt/kid under OPENSHELL_LOCAL_TLS_DIR (default: the OpenShell state \ + directory's tls/). Listener TLS does not configure sandbox launch signing", + )); + } + Ok(()) + } + pub async fn create_sandbox_authenticated( &self, sandbox: Sandbox, @@ -1066,6 +1089,13 @@ impl ComputeRuntime { lifecycle_guard: SandboxLifecycleGuard, global_guard: SandboxSyncGuard, ) -> Result { + // Defend the internal create path too, before consuming a staged archive + // or persisting the sandbox. The gRPC handler checks before driver validation. + self.validate_launch_signer_configured( + launch_authentication + .as_ref() + .is_some_and(|auth| !auth.is_empty()), + )?; self.validate_caller_driver_config( sandbox .spec @@ -7561,6 +7591,8 @@ mod tests { current_sandboxes: Vec, workspace_rpcs_unimplemented: bool, omit_protocol_metadata: bool, + requires_launch_authentication: bool, + create_calls: AtomicUsize, } #[tonic::async_trait] @@ -7597,12 +7629,19 @@ mod tests { rootfs_tar_staging_dir: String::new(), rootfs_tar_max_bytes: 0, extension: (!self.omit_protocol_metadata).then(|| { - openshell_core::extension_protocol::extension_metadata( + let mut metadata = openshell_core::extension_protocol::extension_metadata( ExtensionFamily::Compute, "openshell/test-driver", "test", [], - ) + ); + if self.requires_launch_authentication { + metadata.required_capabilities.push( + openshell_core::extension_protocol::COMPUTE_LAUNCH_AUTHENTICATION + .to_string(), + ); + } + metadata }), })) } @@ -7662,6 +7701,7 @@ mod tests { &self, _request: Request, ) -> Result, Status> { + self.create_calls.fetch_add(1, Ordering::SeqCst); Ok(tonic::Response::new(CreateSandboxResponse::default())) } @@ -15424,6 +15464,94 @@ mod tests { ); } + #[tokio::test] + async fn negotiated_launch_requirement_rejects_create_before_driver_or_store() { + let driver = Arc::new(TestDriver { + requires_launch_authentication: true, + ..Default::default() + }); + let store = Arc::new(Store::connect("sqlite::memory:").await.unwrap()); + let runtime = ComputeRuntime::from_driver( + "external-requires-launch".to_string(), + driver.clone(), + None, + store.clone(), + SandboxIndex::new(), + SandboxWatchBus::new(), + TracingLogBus::new(), + Arc::new(SupervisorSessionRegistry::new()), + ) + .await + .unwrap(); + + for authentication in [None, Some(Vec::new())] { + let sandbox = sandbox_record( + "sb-missing-signer", + "missing-signer", + SandboxPhase::Provisioning, + ); + let error = runtime + .create_sandbox_authenticated(sandbox, None, authentication, false) + .await + .expect_err("the negotiated launch requirement must be checked before create"); + assert_eq!(error.code(), Code::FailedPrecondition); + assert!(error.message().contains("[openshell.gateway.gateway_jwt]")); + assert!(error.message().contains("OPENSHELL_LOCAL_TLS_DIR")); + assert!(error.message().contains("Listener TLS")); + assert!( + store + .get_message::("sb-missing-signer") + .await + .unwrap() + .is_none() + ); + assert_eq!(driver.create_calls.load(Ordering::SeqCst), 0); + } + + let sandbox = sandbox_record("sb-with-signer", "with-signer", SandboxPhase::Provisioning); + let identity = crate::auth::sandbox_session::PersistedSandboxIdentity::new().unwrap(); + let authentication = test_session_authority() + .mint_persisted_launch("sb-with-signer", &identity) + .unwrap(); + runtime + .create_sandbox_authenticated( + sandbox, + None, + Some(serde_json::to_vec(&authentication).unwrap()), + false, + ) + .await + .unwrap(); + assert_eq!(driver.create_calls.load(Ordering::SeqCst), 1); + } + + #[tokio::test] + async fn external_driver_without_launch_requirement_can_create_without_signer() { + let driver = Arc::new(TestDriver::default()); + let runtime = ComputeRuntime::from_driver( + "external-other-auth".to_string(), + driver.clone(), + None, + Arc::new(Store::connect("sqlite::memory:").await.unwrap()), + SandboxIndex::new(), + SandboxWatchBus::new(), + TracingLogBus::new(), + Arc::new(SupervisorSessionRegistry::new()), + ) + .await + .unwrap(); + runtime.validate_launch_signer_configured(false).unwrap(); + runtime + .create_sandbox( + sandbox_record("sb-other-auth", "other-auth", SandboxPhase::Provisioning), + None, + false, + ) + .await + .unwrap(); + assert_eq!(driver.create_calls.load(Ordering::SeqCst), 1); + } + #[tokio::test] async fn create_sandbox_returns_resource_version_one() { let runtime = test_runtime(Arc::new(TestDriver::default())).await; diff --git a/crates/openshell-server/src/defaults.rs b/crates/openshell-server/src/defaults.rs index 21e66b02bd..390dad1eff 100644 --- a/crates/openshell-server/src/defaults.rs +++ b/crates/openshell-server/src/defaults.rs @@ -95,7 +95,11 @@ pub fn complete_local_tls_paths() -> Result> { pub fn complete_local_jwt_config() -> Result> { let dir = default_local_tls_dir()?; - let paths = LocalJwtPaths::resolve(&dir); + local_jwt_config(&dir) +} + +pub fn local_jwt_config(dir: &Path) -> Result> { + let paths = LocalJwtPaths::resolve(dir); let present = paths.files().iter().filter(|path| path.is_file()).count(); match present { 0 => Ok(None), diff --git a/crates/openshell-server/src/grpc/sandbox.rs b/crates/openshell-server/src/grpc/sandbox.rs index 5dbae2cc7b..8f2bf9a784 100644 --- a/crates/openshell-server/src/grpc/sandbox.rs +++ b/crates/openshell-server/src/grpc/sandbox.rs @@ -634,6 +634,9 @@ async fn handle_create_sandbox_inner( ) .await?; + state + .compute + .validate_launch_signer_configured(state.sandbox_session_jwt_authority.is_some())?; state .compute .validate_sandbox_create(&sandbox) @@ -3942,6 +3945,114 @@ mod tests { GpuResourceRequirements, SandboxServiceExposure, ServiceAuthorizationMode, ServiceEndpoint, }; + #[tokio::test] + async fn missing_launch_signer_precedes_driver_validation_and_preserves_staged_archive() { + let directory = tempfile::tempdir().unwrap(); + let mut state = test_server_state().await; + let mut metadata = openshell_core::extension_protocol::extension_metadata( + openshell_core::extension_protocol::ExtensionFamily::Compute, + "test/launch-authentication", + "test", + [], + ); + metadata + .required_capabilities + .push(openshell_core::extension_protocol::COMPUTE_LAUNCH_AUTHENTICATION.to_string()); + let admission = openshell_core::resource_admission::DriverAdmissionConfig { + allow_driver_config: true, + ..Default::default() + }; + let driver = Arc::new( + crate::test_support::FakeComputeDriver::new().with_capabilities( + openshell_core::proto::compute::v1::GetCapabilitiesResponse { + driver_name: "test".to_string(), + default_image: "test/image:latest".to_string(), + rootfs_tar_staging_dir: directory.path().to_string_lossy().into_owned(), + rootfs_tar_max_bytes: 1024, + resource_admission_policy: admission.acknowledgement(), + extension: Some(metadata), + ..Default::default() + }, + ), + ); + let compute = crate::compute::ComputeRuntime::from_driver( + "test".to_string(), + driver.clone(), + None, + state.store.clone(), + crate::sandbox_index::SandboxIndex::new(), + crate::sandbox_watch::SandboxWatchBus::new(), + crate::tracing_bus::TracingLogBus::new(), + Arc::new(crate::supervisor_session::SupervisorSessionRegistry::new()), + ) + .await + .unwrap() + .with_admission_policy(admission) + .unwrap(); + Arc::get_mut(&mut state).unwrap().compute = compute; + let staging = state.compute.rootfs_tar_staging(); + let slot = staging + .begin("default", "dev-user", "rootfs.tar", 7) + .unwrap(); + std::fs::write(&slot.upload_path, b"archive").unwrap(); + let driver_config = Struct { + fields: std::iter::once(( + "test".to_string(), + Value { + kind: Some(Kind::StructValue(Struct { + fields: std::iter::once(( + crate::compute::rootfs_tar::STAGING_TOKEN_FIELD.to_string(), + Value { + kind: Some(Kind::StringValue(slot.token.clone())), + }, + )) + .collect(), + })), + }, + )) + .collect(), + }; + driver.clear_calls(); + let error = handle_create_sandbox( + &state, + authed_request(CreateSandboxRequest { + name: "missing-signer".to_string(), + spec: Some(SandboxSpec { + template: Some(SandboxTemplate { + driver_config: Some(driver_config), + ..Default::default() + }), + ..Default::default() + }), + workspace_scope: Some(openshell_core::proto::workspace_selector( + "default".to_string(), + )), + ..Default::default() + }), + ) + .await + .unwrap_err(); + assert_eq!(error.code(), tonic::Code::FailedPrecondition); + assert!(error.message().contains("sandbox launch signing")); + assert!( + driver.calls().is_empty(), + "driver validation must not precede signing preflight" + ); + assert_eq!( + staging.peek(&slot.token).unwrap(), + std::path::PathBuf::from(&slot.upload_path) + ); + assert_eq!(std::fs::read(&slot.upload_path).unwrap(), b"archive"); + assert!( + state + .store + .get_message_by_name::("default", "missing-signer") + .await + .unwrap() + .is_none() + ); + } + // ---- shell_escape ---- #[test] diff --git a/crates/openshell-server/src/lib.rs b/crates/openshell-server/src/lib.rs index d17aa19b33..6baa769f66 100644 --- a/crates/openshell-server/src/lib.rs +++ b/crates/openshell-server/src/lib.rs @@ -536,59 +536,16 @@ pub(crate) async fn run_server( // startup Describe calls can authenticate with gateway-caller tokens. let (extension_jwt_issuer, sandbox_session_jwt_authority) = if let Some(ref jwt) = config.gateway_jwt { - let signing_pem = std::fs::read(&jwt.signing_key_path).map_err(|e| { - Error::config(format!( - "failed to read sandbox JWT signing key from {}: {e}", - jwt.signing_key_path.display() - )) - })?; - let public_pem = std::fs::read(&jwt.public_key_path).map_err(|e| { - Error::config(format!( - "failed to read sandbox JWT public key from {}: {e}", - jwt.public_key_path.display() - )) - })?; - let kid = std::fs::read_to_string(&jwt.kid_path) - .map_err(|e| { - Error::config(format!( - "failed to read sandbox JWT kid from {}: {e}", - jwt.kid_path.display() - )) - })? - .trim() - .to_string(); - if kid.is_empty() { - return Err(Error::config(format!( - "sandbox JWT kid file {} is empty", - jwt.kid_path.display() - ))); - } - let issuer = Arc::new( - auth::sandbox_jwt::ExtensionJwtIssuer::from_pem( - &signing_pem, - &public_pem, - kid.clone(), - &jwt.gateway_id, - jwt.token_ttl(), - ) - .map_err(Error::config)?, - ); - let session_authority = Arc::new( - auth::sandbox_jwt::SandboxSessionJwtAuthority::from_pem( - &signing_pem, - &public_pem, - kid, - &jwt.gateway_id, - jwt.sandbox_token_ttl(), - ) - .map_err(Error::config)?, - ); + let authorities = auth::launch_signing::load(jwt)?; info!( gateway_id = %jwt.gateway_id, ttl_secs = jwt.ttl_secs.map(std::num::NonZeroU64::get), "gateway-minted sandbox JWT enabled" ); - (Some(issuer), Some(session_authority)) + ( + Some(authorities.extension), + Some(authorities.sandbox_session), + ) } else { (None, None) }; diff --git a/crates/openshell-server/src/test_support.rs b/crates/openshell-server/src/test_support.rs index dcbc2e1cf9..0cd4abf2d5 100644 --- a/crates/openshell-server/src/test_support.rs +++ b/crates/openshell-server/src/test_support.rs @@ -140,6 +140,13 @@ impl Default for FakeComputeDriver { } impl FakeComputeDriver { + /// Override the handshake response to exercise gateway capability admission. + #[must_use] + pub fn with_capabilities(self, capabilities: GetCapabilitiesResponse) -> Self { + self.with_state(|state| state.capabilities = capabilities); + self + } + #[must_use] pub fn new() -> Self { Self { diff --git a/docs/how-it-works/gateways/configuration.mdx b/docs/how-it-works/gateways/configuration.mdx index bdcd65a4bb..4d247f476b 100644 --- a/docs/how-it-works/gateways/configuration.mdx +++ b/docs/how-it-works/gateways/configuration.mdx @@ -275,6 +275,12 @@ The client-certificate handshake policy has no `require_client_auth` TOML field. `[openshell.gateway.gateway_jwt] ttl_secs` controls generation-bound gateway-facing and Sandbox Protocol credentials minted for a sandbox session, plus typed extension JWTs. Omit it for non-expiring local sandbox session credentials: both session tokens carry `exp = 0`, supervisors skip periodic session-token renewal, and refresh responses omit their expiration timestamps. Typed extension JWTs retain a 900-second default when the field is omitted. Use omission only for local single-player Docker, Podman, or VM gateways. Explicit `0` is invalid. Kubernetes and other shared deployments should set a positive TTL; Helm renders `3600` seconds by default, and the gateway logs a warning when a Kubernetes gateway omits the field. +Listener TLS and sandbox launch signing use separate keys. A working HTTPS listener or mTLS client certificate does not supply the signing key needed to start a VM sandbox. Configure `signing_key_path`, `public_key_path`, `kid_path`, and `gateway_id` in `[openshell.gateway.gateway_jwt]`, as shown above. For a local installation, `openshell-gateway generate-certs --output-dir ` writes a signing bundle alongside the TLS bundle. Without explicit `gateway_jwt` configuration, the gateway discovers `jwt/signing.pem`, `jwt/public.pem`, and `jwt/kid` under `OPENSHELL_LOCAL_TLS_DIR`, or under the OpenShell state directory's `tls/` directory when that variable is unset. Explicit signing configuration takes precedence over local discovery. + +The gateway rejects partial or invalid signing bundles at startup, including private and public keys that do not match. When the selected driver requires launch signing and no bundle is configured, sandbox creation fails before the driver validates or prepares the sandbox. The error names the signing configuration and discovery location; it does not print keys or tokens. The VM driver also validates launch material before resolving images or creating sandbox state. + +External compute drivers declare this requirement by adding `openshell.compute.launch-authentication` to their extension metadata's `required_capabilities`. This requires a gateway that understands the launch-signing preflight and supplies `SandboxLaunchAuthentication` on creation. Drivers that use another launch mechanism can omit the requirement. The separate `supports_sandbox_authentication` capability describes the `AuthenticateSandbox` RPC and does not require launch signing. + `[openshell.gateway.auth] allow_unauthenticated_users = true` is an unsafe local-development and trusted-proxy escape hatch. It accepts user-facing CLI/API calls without OIDC or mTLS credentials while sandbox supervisors still authenticate with gateway-minted sandbox JWTs. Leave it false for shared and production gateways. ## OCSF JSONL Output @@ -1128,6 +1134,8 @@ runtime-selected profile. Each sandbox runs inside its own libkrun microVM managed by the standalone `openshell-driver-vm` subprocess. Use this driver when you want stronger isolation than container namespaces alone. +The example uses explicit launch-signing paths. Point them at an existing signing bundle, or omit the `gateway_jwt` table to use local discovery. Listener and guest TLS certificates do not supply these signing credentials. + ```toml [openshell] version = 2 @@ -1140,6 +1148,13 @@ compute_driver = "vm" # Gateway-owned CA injected into the selected local driver. guest_tls_ca = "/var/lib/openshell/guest-tls/ca.pem" +# Sandbox launch signing is separate from listener and guest TLS. +[openshell.gateway.gateway_jwt] +signing_key_path = "/etc/openshell/jwt/signing.pem" +public_key_path = "/etc/openshell/jwt/public.pem" +kid_path = "/etc/openshell/jwt/kid" +gateway_id = "openshell" + [openshell.drivers.vm] state_dir = "/var/lib/openshell/vm" # Where the gateway looks for the openshell-driver-vm subprocess binary. @@ -1207,6 +1222,15 @@ overlay_disk_mib = 4096 # rootfs_tar_max_bytes = 10737418240 ``` +To generate a local bundle for discovery, run these commands in fish and retain the same environment when starting the gateway: + +```shell +set -gx OPENSHELL_LOCAL_TLS_DIR ~/.local/state/openshell/tls +openshell-gateway generate-certs --output-dir "$OPENSHELL_LOCAL_TLS_DIR" +``` + +The command also installs local CLI mTLS credentials. For explicit signing configuration, set the example's signing paths to `jwt/signing.pem`, `jwt/public.pem`, and `jwt/kid` under that directory. The gateway loads those paths in preference to local discovery. + Rootfs tar staging requires the gateway and the VM driver to share a filesystem and run as the same user. That holds for the managed VM driver, which the gateway starts as a subprocess. If you point `compute_driver_endpoints` at an @@ -1246,6 +1270,8 @@ openshell-gateway config preflight --path ~/.config/openshell/gateway.toml Without `--path`, the command validates a nonempty `OPENSHELL_GATEWAY_CONFIG`. Otherwise, it validates an existing XDG gateway config when one is discovered. When neither source selects a config, preflight validates the effective daemon arguments. An explicit missing path, a legacy schema-v1 file, invalid TOML, a symlink, or any nonregular file fails. Preflight merges the selected file with the current `OPENSHELL_*` environment and applies the daemon's read-only startup checks. These checks include selector and socket normalization, registered-driver selection and configuration, rate-limit pairs, TLS and mTLS relationships, interceptor registrations, and supervisor middleware registrations. When a selected file omits `compute_driver`, preflight validates each configured table for an auto-detectable driver without running the runtime detection probes, which can connect local sockets or launch discovery commands. It validates configured driver TLS requirements, including the gateway CA, without requiring package-generated certificate files to exist before certificate generation. It does not construct a compute driver or connect to a transport. A failed preflight always preserves the file; it never migrates, replaces, or rewrites configuration. +Config preflight validates configuration without loading launch-signing keys, so a successful result does not confirm that the key files exist or form a usable pair. Gateway startup validates configured or discovered signing bundles. Sandbox creation checks the selected driver's launch-signing requirement before image preparation. + To validate the exact daemon arguments that a wrapper will pass, place them after `--` instead of using `--path`: