From a89253be961000a08ce81fceb9544aa1a1d47a37 Mon Sep 17 00:00:00 2001 From: cabbage Date: Mon, 24 Aug 2026 06:23:17 +0000 Subject: [PATCH] docs: finalize rvbox v1 design and implementation plan --- .gitignore | 8 + docs/README.md | 4 + docs/architecture.md | 226 +++-- docs/configuration.md | 88 ++ docs/control-plane.md | 107 ++- docs/examples/client.toml | 205 +++++ docs/examples/server.toml | 109 +++ docs/implementation-plan.v1.md | 1375 +++++++++++++++++++++++++++++++ docs/platform-and-operations.md | 161 +++- docs/protocol.md | 132 ++- protos/rvbox/v1/agent.proto | 70 +- protos/rvbox/v1/common.proto | 68 +- protos/rvbox/v1/control.proto | 127 ++- 13 files changed, 2490 insertions(+), 190 deletions(-) create mode 100644 .gitignore create mode 100644 docs/configuration.md create mode 100644 docs/examples/client.toml create mode 100644 docs/examples/server.toml create mode 100644 docs/implementation-plan.v1.md diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..87a166e --- /dev/null +++ b/.gitignore @@ -0,0 +1,8 @@ +/bin/ +/dist/ +/coverage/ +*.db +*.db-shm +*.db-wal +*.sock +*.tmp diff --git a/docs/README.md b/docs/README.md index 3a094ea..bd0f851 100644 --- a/docs/README.md +++ b/docs/README.md @@ -11,6 +11,10 @@ Read the documents in this order: query/foreground semantics. 4. [Platform and operations](platform-and-operations.md) — Unix/Windows contracts, recovery, storage safety, telemetry, and defaults. +5. [Configuration contract](configuration.md) — strict TOML loading, shell + resolution, cross-field validation, and annotated server/client examples. +6. [Go implementation plan](implementation-plan.v1.md) — phased build order, + package boundaries, storage/session/client details, tests, and release gates. The wire authority is in [`../protos/rvbox/v1`](../protos/rvbox/v1): `common.proto` contains shared data types, `agent.proto` contains the diff --git a/docs/architecture.md b/docs/architecture.md index 18767d6..3a3d4dc 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -12,6 +12,9 @@ RVBox deliberately executes arbitrary commands with the identity, permissions, and base environment of the client daemon. It is therefore an administrative tool, not a multi-tenant remote-execution service. +Both daemons use strict TOML 1.0 configuration. The normative schema and fully +annotated examples are in the [configuration contract](configuration.md). + ## Non-goals - Mutual TLS, client certificates, enrollment tokens, and a client-ID allowlist @@ -26,10 +29,14 @@ tool, not a multi-tenant remote-execution service. The server accepts a self-reported hostname as `client_id`; it is an opaque 1–128 ASCII-character routing/display key. Unknown IDs are accepted. A newer -registration for an ID replaces its prior live session. Consequently, a peer -able to reach nginx can impersonate or take over a client ID. This is an -accepted v1 limitation and deployments must restrict the endpoint to a trusted -network. +registration from the same durable client instance replaces its prior live +session. A different instance is accepted normally when no session for that +client ID is live. While one is live, a different instance is rejected unless +an operator grants a one-shot override for that exact pending instance. This +prevents accidental hostname collisions but is not authentication: a peer able +to reach nginx and copy or guess the identifiers can still impersonate a +client. This is an accepted v1 limitation and deployments must restrict the +endpoint to a trusted network. The Unix control socket is local-only and mode `0600`, owned by the server account. The optional HTTP JSON-RPC endpoint is intentionally unauthenticated; @@ -57,53 +64,113 @@ its dispatch loops. 1. The client connects over WSS and sends `ClientHello` with its client ID, protocol capability, OS/architecture, daemon version, current daemon CWD, - supported shells, and a fresh reconnect UUID. + supported shells, and a durable random client-instance UUID. 2. The server accepts the current compatible protocol version, fences the - previous connection for that ID, and returns a fresh server-issued - `session_id` and monotonic `session_generation`. + previous connection for that ID and same client instance, and returns a + fresh server-issued `session_id` and monotonic `session_generation`. A live + claim from a different client instance is rejected while the current session + is live unless an operator has explicitly authorized that pending instance. + With no live session, the new instance is accepted normally. 3. Every client-to-server envelope and server dispatch is bound to that token. The server discards traffic from superseded sessions, including late output. 4. The server queues work while a client is offline and dispatches it only when the active session advertises capacity. A replacement session immediately - resumes non-terminal reconciliation. + performs bidirectional reconciliation between the server's non-terminal set + and the client's complete retained-command set. The server returns explicit + local terminate/discard decisions before new dispatch begins. ## Command model -Each user request has a server-generated UUID (`issue_uuid`) and a durable -request record: target client, request/issue timestamps, shell type, command -text or script descriptor, CWD, environment overrides, resource-profile flags, -and lifecycle state. The UUID is the end-to-end idempotency key. Transport is -at-least-once, but the client durably remembers accepted UUIDs and never starts -the same request twice. +Each user request has a UUID (`issue_uuid`) and a durable request record: target +client, request/issue timestamps, shell type, command text or script descriptor, +CWD, environment overrides, resource-profile flags, and lifecycle state. `rvc` +normally supplies this UUID as its optional `request_id`; the server generates +one when it is omitted. Command and mutation identifiers are UUIDv7 values. +Transport is at-least-once, but the client durably +remembers accepted UUIDs and provides an at-most-once execution guarantee: it +never authorizes the same request to execute twice. A crash in the launch window +may interrupt a command before its requested code runs, but must never cause an +automatic retry with an uncertain prior outcome. + +After full terminal history is removed, each client retains a compact FIFO of +the most recent 1,000,000 command tombstones containing binary UUIDv7, +immutable request hash, and completion/acknowledgement time. The server retains +the most recent 1,000,000 command tombstones globally as a first-line duplicate +check. A matching UUID/hash returns structured `ALREADY_EXECUTED`; a matching +UUID with different immutable content is a conflict. These ledgers have separate +count-based budgets and do not retain command payload or output. Replay +protection older than the retained client tombstone horizon is best-effort. The server states are: ```text -queued -> dispatched -> accepted -> running -> succeeded | failed | terminated - \-------------------------------> cancelled +queued -> dispatched | cancelled | expired +dispatched -> queued | accepted | rejected | cancelled +accepted -> running | rejected | cancelled +running -> succeeded | failed | terminated | interrupted ``` -`accepted` means the client has durably accepted the request; `running` means -the process has been launched. `cancelled` is used when it is stopped before -launch. A kill racing launch is resolved by command revision: the client either -acknowledges cancellation before launch or launches then immediately applies -the requested signal, recording the race. +Recovery/corruption handling may also move an affected `dispatched` or +`accepted` command to `interrupted`; those are exceptional reconciliation +transitions, not normal execution outcomes. + +`accepted` means the client has durably admitted the request; `running` means +the requested process has been launched. `cancelled` is used when it is stopped +before launch. A permanent pre-launch validation, script-transfer, or process- +preparation failure is terminal `rejected`; `failed` is reserved for code that +actually launched. A transient +capacity rejection returns to `queued` with backoff and remains subject to its +queue TTL. The server may cancel work that has never been dispatched immediately. +For dispatched or accepted work, it persists a higher command revision and +sends the revisioned signal without prematurely declaring a terminal state. The +client either returns a revisioned `cancelled` lifecycle before launch +authorization or applies the signal after authorization and returns a +revisioned signal result. Cancellation intent remains internal rather than +adding a public lifecycle state. + +Queued work has a configurable acceptance deadline, default 15 minutes; zero +explicitly means no expiry. A command that was never dispatched becomes +terminal `expired` at its deadline. A dispatched command whose acceptance is +uncertain remains non-terminal and is displayed as expired pending +reconciliation. Later client evidence updates the actual lifecycle. Acceptance +or execution observed after the deadline creates an incident and is displayed +as late-after-expiry; the server requests termination but continues recording +the actual client-reported outcome. + +Process launch has internal durable phases `launch_prepared` and +`launch_authorized` between public `accepted` and `running`. The client creates +the process behind an OS-specific execution barrier, durably records its process +identity, durably authorizes launch, and only then releases requested command +code. Once authorization is durable, an uncertain outcome is reconciled as +`interrupted`, never by redispatching that UUID. The client permits 16 concurrent processes and 100 pending commands by default. Those values are configurable and advertised to the server. A full queue causes a structured capacity error rather than creating unbounded work. +The server separately defaults to at most 1,000 queued commands for one target +client and 10,000 queued commands globally, subject to the stricter byte quotas. ### Execution contract - Shell selection is explicit: Unix-like clients support `sh` and `bash`; Windows supports `cmd` and `powershell`. Defaults are `sh` and `powershell`. Unsupported shells are rejected; no fallback occurs. +- The client materializes `command_text` as a private generated wrapper and + executes that file with exactly the selected shell: `sh`/`bash`, `cmd`, or + PowerShell. This avoids transport quoting and Windows command-line limits; + it never performs shell detection or fallback. Wrapper cleanup follows the + same terminal rule as uploaded scripts. - A command receives the daemon account's permissions and startup environment, overlaid with the persisted `env_overrides` map. The effective CWD is the requested existing directory or the registered daemon CWD when omitted. - Each command is isolated into a process tree: a Unix session/process group or a Windows Job Object. A daemon that cannot supervise its children terminates them and reports interruption rather than claiming recovery it cannot make. +- Terminal lifecycle means the supervised tree is empty, not only that its root + shell exited. After root exit, RVBox allows a configurable 5-second descendant + and output-drain grace period, then terminates residual descendants. It drains + capture pipes and durably sequences all retained output (or an explicit + `OutputIncomplete` marker) before emitting the terminal lifecycle event. - Resource profiles are composable flags (`LIGHT`, `CPU_MEDIUM`, `CPU_HEAVY`, `MEM_MEDIUM`, `MEM_HEAVY`, `DISK_MEDIUM`, `DISK_HEAVY`). Profiles are opt-in; the concrete administrator-configured limits are applied with cgroup v2 when @@ -115,18 +182,20 @@ a structured capacity error rather than creating unbounded work. Scripts are content-addressed uploads, not shell-escaped command strings. The server sends a descriptor containing SHA-256 and then ordered chunks (default maximum: 10 MiB). The client verifies the digest, writes an owner-only temporary -file beneath the effective CWD, executes it with the selected shell, and removes -it after the command reaches a terminal state. Script content is not copied into -the audit log; its digest and metadata are. +file beneath the effective CWD, executes it with exactly the selected shell, and +removes it after the command reaches a terminal state. The descriptor filename +is display metadata only and is never used as a path component. Script content +is not copied into the audit log; its digest and metadata are. ## Process control and diagnostics -On Unix, `kill` addresses the command's process group and accepts normal signal -names/numbers supported by that client. On Windows, only `SIGTERM` and `SIGKILL` -are valid. `SIGTERM` makes a best-effort `CTRL_BREAK_EVENT` delivery to the -dedicated console group, waits 10 seconds, then terminates the Job Object if -needed. `SIGKILL` immediately terminates the Job Object. The response reports -the actual escalation outcome. +On Unix, `kill` addresses the command's process group and accepts the portable +v1 set `HUP`, `INT`, `TERM`, `KILL`, `USR1`, and `USR2` (including their +`SIG`-prefixed CLI spellings). Arbitrary native signal numbers are not part of +v1. On Windows, only `TERM` and `KILL` are valid. `TERM` makes a best-effort +`CTRL_BREAK_EVENT` delivery to the dedicated console group, waits 10 seconds, +then terminates the Job Object if needed. `KILL` immediately terminates the Job +Object. The response reports the actual escalation outcome. Lifecycle state never asserts `hung`. A separate `suspected_hung` diagnostic is emitted after the configurable default of 10 minutes without observable @@ -142,31 +211,73 @@ restart and reconciles only these; server-confirmed historical terminal commands are not re-reconciled. The client stores only active command state and output not acknowledged by the server. -Every execution-originated event has a strictly increasing `event_seq` scoped to -one command. This includes lifecycle transitions, stdout/stderr chunks, stdin -acknowledgments, resource snapshots, signals, and terminal events. The server -preserves the sequence and additionally records receipt time. This is the -canonical reconstruction order across interleaved streams and retries. -Truncation is separate range metadata so it can truthfully describe missing -event sequences without consuming one itself. +Every transmitted execution event has a strictly increasing `event_seq` scoped +to one command. The client first stores lifecycle transitions, stdout/stderr, +stdin acknowledgements, resource snapshots, signals, and terminal state in a +durable local order. It durably assigns wire sequences only as entries enter the +bounded send window, after any unsent-output compaction. Once assigned, an event +is pinned until acknowledged and retry content is immutable. A client-side +truncation marker therefore consumes a normal sequence without creating a wire +gap. Server-side retention exposes removed event ranges as query metadata +without allocating client sequences. The server also records receipt time; +`event_seq` remains the canonical transmitted order across streams and retries. -Output chunks are Zstandard-compressed before persistent quota accounting. -Per-command history is a rolling compressed window (10 MiB default), so the -oldest output segments for that command are removed first and a sequence-range -truncation marker remains. This applies to active and terminal commands. When -connected, the client first removes server-acknowledged segments. While offline, -it must still honor both hard caps: it retains the newest tail, removes oldest -unacknowledged compressed chunks when necessary, and records their exact missing -ranges for durable reporting on reconnect. +The terminal lifecycle event is always the final client event for a command. +The control-plane follow wrapper emits any server-created retention metadata +before returning that terminal event. Followers may therefore stop on terminal +without missing subsequently sequenced stdout/stderr or known truncation data. -Each client also has a 50 MiB aggregate compressed spool cap for active, -unacknowledged work. The server's matching per-client compressed-history cap is -50 MiB; it evicts that client's oldest terminal command records as needed. The -server-wide cap is 1 GiB; it evicts whole oldest terminal command records -(metadata and output), never arbitrary stdout/stderr rows. Active commands are -protected. If active commands alone consume a per-client budget, their oldest -acknowledged output rotates by the per-command rule; pipes continue draining so -a child cannot deadlock on output. +All command-owned stored data is quota-accounted: execution metadata, script +body, pending stdin, events, and output. Stored message/blob payloads are +Zstandard-compressed; necessary SQLite index/state columns are charged by their +encoded lengths plus a conservative versioned per-row/index overhead rather +than pretending they are free. This logical accounting is deterministic across +SQLite compaction. A separate filesystem free-space floor protects WAL, +temporary files, tombstones, and accounting variance. +The default total is 32 MiB per command, 256 MiB per client on both client and +server, and 4 GiB server-wide. A separate 10 MiB rolling output window remains +per command, and raw script input remains limited to 10 MiB. Client accounting +also charges the raw generated execution wrapper/script file while it exists, +even though the compressed durable source was already charged. + +Each accepted active command reserves 64 KiB of its quota for bounded closeout +metadata. Essential state for an accepted stdin/signal mutation is additionally +reserved before that mutation succeeds. Essential active state is never silently +rolled. Server output is evictable: the oldest retained chunks are removed first +and sequence-range query metadata remains. The client removes acknowledged data, +pins its bounded assigned send window, and may replace older unsequenced output +with normally sequenced byte-loss markers. While offline or under sustained +overload, it retains the newest tail and records exact known lost bytes. At +client/server aggregate limits, terminal commands are evicted as whole UUID +records in oldest server issue-time/UUIDv7 order first. If active data +alone reaches a limit, output rotates or enters loss mode and new essential +allocations are rejected with `CAPACITY_EXHAUSTED`; pipes continue draining. + +Transient raw output also has bounded high/low watermarks before compression: +1 MiB/256 KiB per command, 8 MiB/4 MiB per client daemon, and 64 MiB/32 MiB +server-wide by default. Crossing a client high watermark enters loss mode; +still-unsequenced output bytes may be discarded before compression and replaced +in durable local order by a later `OutputTruncation`. Loss mode ends only below +the corresponding low watermark. The server never drops an already sequenced +client event: at its ingress high watermark it withholds acknowledgement and +closes an overproducing session if bounded admission cannot continue, letting +the durable client retry after the server backlog falls below its low watermark. +Lifecycle, +stdin acknowledgements, signal results, and truncation/incomplete markers use +reserved capacity and are never treated as droppable output. + +Independently of byte pressure, the server reclaims each whole terminal command +and all command-owned data 30 days after its terminal time by default. Zero +explicitly disables age rotation. Compact replay tombstones, audit records, and +storage incidents remain under their separate retention policies. + +Storage health is derived from durable incident records. Safe repairs resolve +an incident automatically; known data loss remains dirty until an operator +explicitly acknowledges it. Resolution clears the dirty health flag but does +not erase records as part of that action. Unresolved compact records are +non-evictable; resolved summaries and audit entries follow the separate 100 MiB +audit/incident history rotation. Recovery and incident management are described +in the platform contract and exposed through the control plane. ## Audit and timestamps @@ -175,8 +286,13 @@ events: source/transport identity where available, target client, UUID, action, request time, result, and error. It records command text, environment-override names (not values), and script metadata/digest, but not duplicated stdin/stdout/stderr payloads. The persisted execution request necessarily keeps -override values for dispatch/retry and must be access-controlled as sensitive -data. Audit retention is configured independently of output retention. +override values for dispatch/retry. In v1, command text, scripts, environment +values, stdin, output, and other command-owned payloads are stored in plaintext; +application-level encryption and key management are deferred to a future +version. Audit storage uses compressed segments under a separate 100 MiB default +quota and rotates complete oldest segments. Audit retention is configured +independently of command retention; its default zero age limit means quota-only +rotation. All protocol timestamps are UTC `google.protobuf.Timestamp` values. Client observed timestamps and server receipt timestamps are distinct; the latter is diff --git a/docs/configuration.md b/docs/configuration.md new file mode 100644 index 0000000..eabc8cf --- /dev/null +++ b/docs/configuration.md @@ -0,0 +1,88 @@ +# RVBox v1 configuration contract + +RVBox v1 uses TOML 1.0 for daemon configuration. The normative annotated +examples are [`examples/server.toml`](examples/server.toml) and +[`examples/client.toml`](examples/client.toml). They list every supported v1 +knob, with the routing, persistence, and safety limits first in each section. + +## Loading and precedence + +- `rvbox-server --config PATH` and `rvbox --config PATH` load one UTF-8 TOML + file. There is no implicit merge of multiple files and no hot reload in v1. +- Precedence is compiled default, then TOML, then an explicitly supplied CLI + flag. Flags exist for operationally important scalar keys; they use the same + validation as TOML. RVBox does not implicitly import configuration from + environment variables. +- Unknown keys, duplicate keys/tables, type mismatches, invalid UTF-8, and + values outside documented ranges are startup errors. Parsing never silently + substitutes a default for a present invalid value. +- Durations are quoted Go-style duration strings such as `"250ms"`, `"15m"`, + and `"720h"`. Byte sizes and counts are base-10 TOML integers whose values are + bytes; comments show the equivalent binary unit. URLs and paths are strings. +- Relative paths are rejected for state, socket, CA, shell-executable, and + allowed-CWD-root fields. `client.daemon_cwd` is resolved once at startup and + then stored and advertised as an absolute path. The annotated client example + uses Unix paths; a Windows deployment replaces `state_dir`, `daemon_cwd`, and + relevant shell paths with absolute Windows paths. Shell fields for the other + platform are syntax-checked but not resolved or advertised. +- The daemon prints its effective configuration after validation, with no + command data or TLS material. Since v1 stores command/environment payloads in + plaintext, configuration output is hygiene rather than a secrecy guarantee. + +## Limits and cross-field validation + +Configuration may lower protocol and storage limits, but may not raise a hard +wire ceiling above the v1 values in the examples. For every high/low watermark, +`0 < low < high`; send windows must fit below their corresponding durable quota. +The 64 KiB command closeout reserve must fit within the command quota, and the +command quota must fit within the client and server tiers. Queue and byte counts +must be positive except where a comment explicitly gives zero a disabling or +indefinite meaning. `queue.max_per_client` may not exceed `queue.max_server`. + +The server must reject external JSON-RPC binds unless `json_rpc.enabled=true`. +Any enabled non-loopback bind produces a conspicuous warning but is permitted by +the accepted v1 debugging contract. The control Unix socket always uses mode +`0600`; it is not a configurable relaxation. + +## Shell executable resolution + +Each `ShellType` maps to one startup-validated absolute executable path from +`[shells]`. The client canonicalizes the path, verifies that it names an +executable regular file appropriate to the platform, and advertises only shells +that passed validation. The configured platform default must be one of those +shells. There is no fallback. + +The resolved executable is independent of a command's `PATH` override. For +example, a request for `SHELL_BASH` still launches the validated `/bin/bash` +even when the request contains `PATH=/tmp/untrusted`; it never searches that +directory for another `bash`. An administrator may intentionally select another +implementation, such as an absolute `pwsh.exe` path for `SHELL_POWERSHELL`, but +the selection remains fixed until daemon restart. + +Command text and uploaded scripts are written to generated wrapper paths and +passed to exactly this executable. The user-supplied script filename is display +metadata only. + +## Resource-profile composition + +`LIGHT` is exclusive. Otherwise, a request may combine at most one CPU tier, +one memory tier, and one disk tier. Thus `CPU_HEAVY + MEM_MEDIUM` is valid, while +`CPU_MEDIUM + CPU_HEAVY` and `LIGHT + MEM_HEAVY` are invalid. A configured +profile declares which controls are required. If the platform cannot apply a +required control atomically before launch, the command is terminal `REJECTED`. + +Profile names describe administrator-defined allowance classes: a `HEAVY` tier +normally permits more resources than `MEDIUM`; RVBox does not invent numeric +values. Zero for an individual numeric limit means that control is not requested +by that profile, but every name in `required_controls` must have a nonzero, +platform-applicable value. Linux disk limits use configured cgroup device +major/minor keys. Windows applies the equivalent whole-Job rate control and +ignores Linux device maps only when disk control is not declared required. + +## Validation ownership + +`internal/config` owns TOML DTOs, strict decoding, default application, flag +overrides, canonicalization, and cross-field validation. It converts the parsed +form into immutable domain configuration before listeners or child processes +start. Network, storage, and supervisor packages receive only their relevant +validated sub-configuration and never parse TOML themselves. diff --git a/docs/control-plane.md b/docs/control-plane.md index 74fcc82..e3882d8 100644 --- a/docs/control-plane.md +++ b/docs/control-plane.md @@ -12,32 +12,83 @@ an optional JSON-RPC 2.0 HTTP adapter for local debugging and batch automation. It has no authentication by design. Binding it beyond loopback is an explicit deployment choice and requires external protection. -gRPC can stream `RunCommandAndFollow` and `FollowCommand`. JSON-RPC remains -simple: callers issue work, query command state, poll event/output pages after -an event sequence, append stdin, close stdin, or signal a command. It does not -invent a separate event-stream protocol. +gRPC streams command history and live events through `FollowCommand`. JSON-RPC +remains simple: callers issue work, query command state, poll event/output pages +after an event sequence, append stdin, close stdin, or signal a command. It does +not invent a separate event-stream protocol. The JSON-RPC method names are the lower-camel protobuf operation names: `listClients`, `getClient`, `listCommands`, `getCommand`, `runCommand`, -`appendStdin`, `closeStdin`, `signalCommand`, and `getOutput`. Parameters and -results use protobuf JSON mapping (including base64 strings for `bytes` and UTC -RFC 3339 strings for timestamps); JSON-RPC errors carry the corresponding -`ControlError` code/data. `getOutput` and `getCommand` are the polling path for -what gRPC exposes as follow streams. +`appendStdin`, `closeStdin`, `signalCommand`, `getOutput`, and the three storage +incident methods documented below. Parameters and results use protobuf JSON +mapping (including base64 strings for `bytes` and UTC RFC 3339 strings for +timestamps). Control response messages contain successful results only. gRPC +failures use canonical non-OK status codes with structured RVBox details where +needed; the JSON-RPC adapter maps the same domain errors to standard JSON-RPC +error objects. `getOutput` and `getCommand` are the polling path for what gRPC +exposes as follow streams. + +`rvc stat CLIENT` also shows the active durable client-instance ID and the most +recent different instance rejected while that client is live. An operator may +run `rvc client takeover CLIENT INSTANCE-ID`; this creates a one-shot 5-minute +authorization for that exact pending claim. Its matching reconnect consumes the +authorization and fences the old session. If the old session is no longer live, +the replacement connects normally without this command. The JSON-RPC method is +`authorizeClientTakeover`. + +Both transports enforce the same decoded field limits. A control gRPC request +may be at most 16 MiB; a JSON-RPC HTTP body may be at most 24 MiB to accommodate +base64 expansion of the 10 MiB script maximum. `ExecutionSpec` itself may be at +most 768 KiB, which also keeps its agent dispatch below the 1 MiB envelope cap. ## CLI semantics `rvc stat` maps to `ListClients`, `GetClient`, `ListCommands`, and `GetCommand`. -History pages default to 20 commands and may request at most 100. Output pages -default to 100 lines; a line is a display operation over ordered chunks, not a -protocol boundary. Output can be filtered by stream and timestamped with the -server's recorded client-observed timestamp plus stream name. +History pages default to 20 commands and may request at most 100. Historical +output uses opaque, byte-bounded cursors and may resume within an output event; +it is not numbered or paginated by lines. The server returns uncompressed output +slices with event sequence, byte offset, stream, observed timestamp, and server +receipt timestamp. `rvc` may render line-oriented human output, but line +boundaries are not storage or pagination boundaries. -`rvc run` creates a command. Foreground mode runs `RunCommandAndFollow`, which -streams output and stops on a terminal event. `--background` uses `RunCommand` -and returns the UUID immediately. Interrupting the CLI, timing out its local -wait, or losing the local control connection never cancels remote work. The -explicit `rvc kill` operation is the only termination path. +`rvc run` always creates a durable command through unary `RunCommand`, which +returns the UUID. Foreground mode then calls `FollowCommand` with that UUID, +`after_event_seq=0`, and `include_existing=true`, streaming output until a +terminal event. `--background` returns immediately after `RunCommand`. +Interrupting the CLI, timing out its local wait, or losing the local control +connection never cancels remote work. The explicit `rvc kill` operation is the +only termination path. A foreground caller can resume `FollowCommand` after its +last received event sequence without missing durable history. + +`FollowCommandResponse` wraps either a client-sequenced `CommandEvent` or a +server-created retention marker and exposes server receipt/recording time +separately from client observation time. A retention marker does not advance the +resume cursor. When retained history is incomplete, the server emits the relevant +marker before any terminal event on that stream, so a follower never stops on +terminal while believing truncated history was complete. + +Mutating requests accept an optional `request_id`. `rvc` generates one per +mutation and reuses it for transport retries; `--request-id` lets automation +reuse it across CLI invocations. The CLI accepts that option globally and drops +it for reads. For `RunCommand`, the supplied request ID is the command's +`issue_uuid`. For stdin, close, signal, repair, and acknowledgement operations +it deduplicates that action while `issue_uuid` or `incident_id` continues to +identify the target. The server generates a request ID when omitted, preserving +simple JSON-RPC use. + +The server stores the mutation kind, target, immutable request hash, and result +under that ID. An identical retry returns the original result; reuse with any +different method, target, or content returns `CONFLICT`. The record is owned by +the affected command or incident for quota and retention. A retry after that +owner has been reclaimed cannot repeat the action: it returns retained +tombstone information where available or `NOT_FOUND`. + +`rvc run --queue-ttl` controls how long work may wait for server-confirmed +acceptance and defaults to 15 minutes; zero means indefinite. `rvc stat` +distinguishes terminal `Expired`, `Expired (awaiting reconciliation)`, and +actual lifecycle with a late-after-expiry warning. It renders terminal +`Rejected` with the client's structured validation/platform reason; `Failed` +means the requested code actually launched. `rvc append` turns a string into `StdinWrite` with `append_newline=true` unless the caller selects raw mode; `--file` supplies raw bytes; `--attach` streams @@ -53,6 +104,20 @@ must not be shown as a terminal state. Pagination response cursors are stable within their declared ordering (newest issue time for command lists; increasing `event_seq` for events/output). -The service returns `ControlError` codes for not found, offline, capacity, -invalid request, unsupported platform feature, conflict, truncation, and -internal/transient errors. It never encodes errors only as CLI text. +Control failures are never embedded in otherwise-successful response messages. +They use canonical gRPC status codes with structured RVBox details; JSON-RPC +returns the corresponding JSON-RPC error object, and `rvc` maps the same domain +error to a stable exit code rather than parsing text. + +## Storage incidents + +`rvc storage incidents` lists unresolved storage incidents by default and can +include resolved history. `rvc storage repair INCIDENT` attempts only a known +safe repair. `rvc storage acknowledge INCIDENT --note ...` accepts documented +irrecoverable loss and clears that incident from dirty health. Both mutations +use the same optional `--request-id` behavior as other mutations. An +acknowledgement note is required and bounded to 4 KiB. Repair or acknowledgement +changes incident state; it does not delete history as part of that action; +resolved history later follows the independent audit/incident quota. The +JSON-RPC equivalents are `listStorageIncidents`, +`repairStorageIncident`, and `acknowledgeStorageIncident`. diff --git a/docs/examples/client.toml b/docs/examples/client.toml new file mode 100644 index 0000000..0606525 --- /dev/null +++ b/docs/examples/client.toml @@ -0,0 +1,205 @@ +# RVBox v1 client example. Integer sizes are bytes; durations are quoted strings. + +[client] +# Reverse WebSocket endpoint exposed by nginx. +server_url = "wss://rvbox.example.test/v1/agent" +# Private durable accepted-command, event-spool, and tombstone root. +state_dir = "/var/lib/rvbox" +# Empty selects the local hostname; otherwise use an opaque 1-128 ASCII ID. +client_id = "" +# Default absolute CWD when a request omits cwd. +daemon_cwd = "/" +# Maximum simultaneously running supervised process trees. +max_running_commands = 16 +# Maximum durably accepted commands waiting to start. +max_queued_commands = 100 +# Grace for reserved terminal cleanup during orderly daemon shutdown. +shutdown_grace = "30s" + +[tls] +# Optional PEM CA bundle; empty uses the operating-system trust store. +ca_file = "" +# Optional certificate name override; empty derives it from server_url. +server_name = "" + +[shells] +# Default shell enum on Unix; the matching path must validate at startup. +default_unix = "sh" +# Default shell enum on Windows; the matching path must validate at startup. +default_windows = "powershell" +# Absolute executable used for SHELL_SH; empty marks it unsupported. +sh = "/bin/sh" +# Absolute executable used for SHELL_BASH; empty marks it unsupported. +bash = "/bin/bash" +# Absolute executable used for SHELL_CMD on Windows; empty marks it unsupported. +cmd = "C:\\Windows\\System32\\cmd.exe" +# Absolute executable used for SHELL_POWERSHELL; may instead point to pwsh.exe. +powershell = "C:\\Windows\\System32\\WindowsPowerShell\\v1.0\\powershell.exe" +# Absolute CWD roots permitted by local policy; empty allows any accessible path. +allowed_cwd_roots = [] + +[network] +# Send WebSocket Ping after this period without inbound activity. +heartbeat_idle = "10s" +# Close and reconnect after this total period without inbound activity. +liveness_timeout = "30s" +# Initial full-jitter reconnect backoff. +reconnect_initial = "1s" +# Maximum full-jitter reconnect backoff. +reconnect_max = "60s" +# Continuous session duration that resets reconnect backoff. +stable_session_reset = "60s" +# Timeout for DNS/TCP/TLS/WebSocket establishment. +connect_timeout = "15s" +# Deadline for an individual WebSocket data-frame write. +write_deadline = "10s" + +[storage] +# Maximum rolling compressed output retained for one command (10 MiB). +command_output_limit_bytes = 10485760 +# Maximum charged data, including raw execution wrapper, per command (32 MiB). +command_total_limit_bytes = 33554432 +# Maximum charged command data across this client daemon (256 MiB). +client_total_limit_bytes = 268435456 +# Compact completed-command replay ledger entry cap. +tombstone_max_entries = 1000000 +# Per-active-command quota held for terminal and loss metadata (64 KiB). +command_closeout_reserve_bytes = 65536 +# Reject new unreserved writes below this filesystem free space (64 MiB). +free_space_floor_bytes = 67108864 +# Target size for a sealed append-only spool segment (256 KiB). +segment_target_bytes = 262144 +# Maximum time a group-commit waits before fsync; acknowledgements wait too. +durability_interval = "100ms" + +[flow] +# Enter per-command raw-output loss mode at this backlog (1 MiB). +raw_output_command_high_bytes = 1048576 +# Leave per-command raw-output loss mode below this backlog (256 KiB). +raw_output_command_low_bytes = 262144 +# Enter client-wide raw-output loss mode at this backlog (8 MiB). +raw_output_client_high_bytes = 8388608 +# Leave client-wide raw-output loss mode below this backlog (4 MiB). +raw_output_client_low_bytes = 4194304 +# Maximum assigned but unacknowledged encoded bytes per command (1 MiB). +unacknowledged_per_command_bytes = 1048576 +# Maximum assigned but unacknowledged encoded bytes for the session (8 MiB). +unacknowledged_per_session_bytes = 8388608 + +[execution] +# Wait after root exit for descendants and capture EOF before forced cleanup. +descendant_drain_grace = "5s" +# Wait after Windows CTRL_BREAK before terminating the complete Job Object. +windows_term_grace = "10s" +# Mark suspected_hung after no observable progress for this duration. +hung_threshold = "10m" +# Interval for best-effort process/resource diagnostic snapshots. +diagnostic_interval = "30s" +# Maximum raw uploaded script or generated script body (10 MiB). +max_script_bytes = 10485760 +# Maximum serialized command ExecutionSpec accepted from the server (768 KiB). +max_execution_spec_bytes = 786432 +# Maximum decoded AgentEnvelope accepted from the server (1 MiB). +max_agent_envelope_bytes = 1048576 +# Maximum uncompressed stdout/stderr chunk emitted to the protocol (64 KiB). +max_raw_chunk_bytes = 65536 +# Maximum protocol detail/reason text encoded as UTF-8 (4 KiB). +protocol_detail_max_bytes = 4096 + +[observability] +# Loopback HTTP listener for local liveness, storage, and supervisor health. +listen = "127.0.0.1:6902" +# Liveness route. +liveness_path = "/livez" +# Readiness route; false while reconciliation or essential recovery is pending. +readiness_path = "/readyz" +# Prometheus metrics route. +metrics_path = "/metrics" +# Structured logging threshold: debug, info, warn, or error. +log_level = "info" +# Structured log encoding: json or text. +log_format = "json" + +# Resource profiles are administrator policy. These illustrative values are not +# protocol guarantees. LIGHT is exclusive; otherwise combine at most one tier +# from each cpu_*, mem_*, and disk_* family. + +[profiles.light] +# Advertise this profile when all required controls validate. +enabled = true +# Controls that must be applied atomically or the request is rejected. +required_controls = ["cpu", "memory", "pids"] +# CPU allowance as percent of one logical CPU; zero omits CPU control. +cpu_percent = 50 +# Hard resident/commit memory allowance (512 MiB); zero omits memory control. +memory_max_bytes = 536870912 +# Maximum processes in the supervised tree; zero omits process-count control. +pids_max = 64 +# Windows whole-Job read rate; zero omits it. +windows_io_read_bps = 0 +# Windows whole-Job write rate; zero omits it. +windows_io_write_bps = 0 +# Linux cgroup read rates keyed by device major:minor. +linux_io_read_bps = {} +# Linux cgroup write rates keyed by device major:minor. +linux_io_write_bps = {} + +[profiles.cpu_medium] +# Advertise the CPU_MEDIUM allowance class. +enabled = true +# Controls that must be applied atomically or the request is rejected. +required_controls = ["cpu"] +# CPU allowance as percent of one logical CPU. +cpu_percent = 200 + +[profiles.cpu_heavy] +# Advertise the CPU_HEAVY allowance class. +enabled = true +# Controls that must be applied atomically or the request is rejected. +required_controls = ["cpu"] +# CPU allowance as percent of one logical CPU. +cpu_percent = 800 + +[profiles.mem_medium] +# Advertise the MEM_MEDIUM allowance class. +enabled = true +# Controls that must be applied atomically or the request is rejected. +required_controls = ["memory"] +# Hard memory allowance (2 GiB). +memory_max_bytes = 2147483648 + +[profiles.mem_heavy] +# Advertise the MEM_HEAVY allowance class. +enabled = true +# Controls that must be applied atomically or the request is rejected. +required_controls = ["memory"] +# Hard memory allowance (8 GiB). +memory_max_bytes = 8589934592 + +[profiles.disk_medium] +# Advertise DISK_MEDIUM only after device/rate controls validate. +enabled = false +# Controls that must be applied atomically or the request is rejected. +required_controls = ["io"] +# Windows whole-Job read bandwidth allowance (100 MiB/s). +windows_io_read_bps = 104857600 +# Windows whole-Job write bandwidth allowance (50 MiB/s). +windows_io_write_bps = 52428800 +# Linux cgroup read allowances; replace 8:0 with an actual delegated device. +linux_io_read_bps = { "8:0" = 104857600 } +# Linux cgroup write allowances; replace 8:0 with an actual delegated device. +linux_io_write_bps = { "8:0" = 52428800 } + +[profiles.disk_heavy] +# Advertise DISK_HEAVY only after device/rate controls validate. +enabled = false +# Controls that must be applied atomically or the request is rejected. +required_controls = ["io"] +# Windows whole-Job read bandwidth allowance (500 MiB/s). +windows_io_read_bps = 524288000 +# Windows whole-Job write bandwidth allowance (250 MiB/s). +windows_io_write_bps = 262144000 +# Linux cgroup read allowances; replace 8:0 with an actual delegated device. +linux_io_read_bps = { "8:0" = 524288000 } +# Linux cgroup write allowances; replace 8:0 with an actual delegated device. +linux_io_write_bps = { "8:0" = 262144000 } diff --git a/docs/examples/server.toml b/docs/examples/server.toml new file mode 100644 index 0000000..3b92688 --- /dev/null +++ b/docs/examples/server.toml @@ -0,0 +1,109 @@ +# RVBox v1 server example. Integer sizes are bytes; durations are quoted strings. + +[server] +# Durable SQLite, command payload, output, audit, and incident root. +data_dir = "/var/lib/rvbox-server" +# HTTP listener receiving WebSocket upgrades from nginx. +agent_listen = "127.0.0.1:6899" +# Exact WebSocket request path accepted on agent_listen. +agent_path = "/v1/agent" +# Local gRPC control socket used by rvc; RVBox forces mode 0600. +control_socket = "/run/rvbox/server.sock" +# Grace allowed for daemon workers to finish reserved closeout on shutdown. +shutdown_grace = "30s" + +[json_rpc] +# Enable the intentionally unauthenticated JSON-RPC debugging adapter. +enabled = false +# HTTP bind for JSON-RPC; non-loopback use emits a prominent warning. +listen = "127.0.0.1:6900" + +[queue] +# Default wait for server-confirmed client acceptance; "0s" means indefinite. +default_ttl = "15m" +# Maximum server-side queued commands for one client before rejection. +max_per_client = 1000 +# Maximum server-side queued commands across all clients before rejection. +max_server = 10000 +# Initial delay after a transient client rejection before redispatch. +retry_initial = "1s" +# Maximum full-jitter delay after repeated transient client rejection. +retry_max = "30s" + +[storage] +# Maximum rolling compressed output retained for one command (10 MiB). +command_output_limit_bytes = 10485760 +# Maximum charged data, including metadata/script/stdin/output, per command (32 MiB). +command_total_limit_bytes = 33554432 +# Maximum charged command data for one target client on the server (256 MiB). +client_total_limit_bytes = 268435456 +# Maximum charged command data across the server, excluding audit/tombstones (4 GiB). +server_total_limit_bytes = 4294967296 +# Reclaim whole terminal commands after this age; "0s" disables age rotation. +terminal_retention = "720h" +# Separate compressed audit plus resolved-incident history budget (100 MiB). +audit_limit_bytes = 104857600 +# Optional audit age rotation; "0s" keeps entries until the byte budget rolls them. +audit_retention = "0s" +# Compact global completed-command replay ledger entry cap. +tombstone_max_entries = 1000000 +# Per-active-command quota held for terminal and loss metadata (64 KiB). +command_closeout_reserve_bytes = 65536 +# Reject new unreserved writes below this filesystem free space (256 MiB). +free_space_floor_bytes = 268435456 +# Target size for a sealed append-only payload/output segment (256 KiB). +segment_target_bytes = 262144 +# Maximum time a group-commit waits before fsync; acknowledgements wait too. +durability_interval = "100ms" +# SQLite busy timeout before an operation returns a transient error. +sqlite_busy_timeout = "5s" +# Maximum operator incident acknowledgement note encoded as UTF-8 (4 KiB). +incident_note_max_bytes = 4096 +# Maximum protocol error/detail/reason text encoded as UTF-8 (4 KiB). +protocol_detail_max_bytes = 4096 + +[flow] +# Stop admitting raw server validation/persistence work at this backlog (64 MiB). +raw_output_high_bytes = 67108864 +# Leave raw-output loss mode only below this backlog (32 MiB). +raw_output_low_bytes = 33554432 +# Maximum assigned but unacknowledged encoded bytes per command (1 MiB). +unacknowledged_per_command_bytes = 1048576 +# Maximum assigned but unacknowledged encoded bytes per client session (8 MiB). +unacknowledged_per_session_bytes = 8388608 +# Deadline for an individual WebSocket data-frame write. +write_deadline = "10s" + +[protocol] +# Send WebSocket Ping after this period without inbound activity. +heartbeat_idle = "10s" +# Close a session after this total period without inbound activity. +liveness_timeout = "30s" +# One-shot live client-instance collision override lifetime. +takeover_ttl = "5m" +# Hard decoded AgentEnvelope ceiling (1 MiB). +max_agent_envelope_bytes = 1048576 +# Hard serialized ExecutionSpec ceiling (768 KiB). +max_execution_spec_bytes = 786432 +# Hard uncompressed stdout/stderr chunk ceiling (64 KiB). +max_raw_chunk_bytes = 65536 +# Hard raw uploaded-script ceiling (10 MiB). +max_script_bytes = 10485760 +# Hard decoded local gRPC request ceiling (16 MiB). +max_control_request_bytes = 16777216 +# Hard JSON-RPC HTTP body ceiling including base64 expansion (24 MiB). +max_json_rpc_body_bytes = 25165824 + +[observability] +# Loopback HTTP listener for liveness, readiness, and Prometheus metrics. +listen = "127.0.0.1:6901" +# Liveness route, available before asynchronous storage recovery completes. +liveness_path = "/livez" +# Readiness route; false while required storage scopes are unavailable. +readiness_path = "/readyz" +# Prometheus metrics route. +metrics_path = "/metrics" +# Structured logging threshold: debug, info, warn, or error. +log_level = "info" +# Structured log encoding: json or text. +log_format = "json" diff --git a/docs/implementation-plan.v1.md b/docs/implementation-plan.v1.md new file mode 100644 index 0000000..bbd492f --- /dev/null +++ b/docs/implementation-plan.v1.md @@ -0,0 +1,1375 @@ +# RVBox v1 Go implementation plan + +## 1. Objective and implementation boundary + +Implement the v1 contracts in the existing design documents as three Go +binaries: + +- `rvbox-server`: durable server, WebSocket agent endpoint, local gRPC control + endpoint, optional JSON-RPC adapter, storage, and audit owner. +- `rvbox`: reverse-connecting client daemon, command supervisor, durable active + spool, and reconnect/reconciliation owner. +- `rvc`: local CLI over the server's Unix-domain gRPC socket. + +The primary release target is a Linux server and Linux client. The Go client is +structured behind OS interfaces from the outset; Windows implementations of +process management, diagnostics, and resource controls are completed before a +Windows client is declared supported. Windows v1 requires Windows 10 or Windows +Server 2016 or newer. Do not claim Windows feature parity while those +implementations are absent. + +This plan implements the already agreed v1 contract. In particular, it does +not add authentication, client enrollment, mutual TLS, or a client allowlist. +The deployment trust boundary remains nginx TLS termination and restricted +network access. The unauthenticated JSON-RPC listener remains disabled by +default and loopback-bound when enabled. + +## 2. Development rules and definition of done + +### 2.1 Local-environment rules + +`../ENV.md` is binding for this repository: + +- Do not install Go, Buf, `protoc`, SQLite tooling, or other heavy development + dependencies on the host. +- Put the toolchain, code generation, unit/integration testing, linting, and + local services in Docker/Docker Compose. +- Run container commands as UID/GID `1001:1001` so generated Go code, module + caches mounted into the project, and test artefacts remain host-owned. +- Do not delete Docker volumes, generated artifacts, or state directories as a + convenience cleanup action. Tests use dedicated temporary volumes/directories + and remove only the exact resources they created. + +### 2.2 Completion standard for every phase + +A phase is complete only when all of the following are true: + +1. The public behavior and failure modes are covered by focused tests. +2. The package has context cancellation, bounded channels/queues, and no + network or disk operation on the WebSocket receive loop. +3. Structured logs and metrics exist for the new state transitions/failures. +4. Configuration defaults, validation, and errors are documented. +5. `docker compose run --rm toolchain make fmt lint test` passes, with no host + installation required. The lint target explicitly snapshots the accepted + first-draft Buf enum-prefix/service-suffix findings and fails on any other or + newly added finding; application/code lint has no such exception. +6. Any recovery or retention code is tested with a real temporary SQLite + database and filesystem rather than mocks alone. + +## 3. Repository and build bootstrap (Phase 0) + +### 3.1 Establish the repository layout + +Create the following layout. Generated code is kept separate from handwritten +logic and is never manually edited. + +```text +cmd/ + rvbox-server/main.go + rvbox/main.go + rvc/main.go +internal/ + agentproto/ # envelope validation, transport-neutral helpers + config/ # file/flag parsing and cross-field validation + domain/ # command state machine and typed errors + server/ + control/ # gRPC implementation and JSON-RPC adapter + session/ # client registry, fencing, dispatch + store/ # SQLite metadata, segments, audit, migrations + client/ + runtime/ # reconnect loop, transport, dispatcher + spool/ # active command/event/output durable spool + supervisor/ # OS-neutral interface + supervisor/unix/ # process groups, /proc, cgroup v2 + supervisor/windows/# Job Objects and Windows diagnostics + observability/ # logging, metrics, health/readiness + testkit/ # clocks, fake transport, fault helpers +gen/go/rvbox/v1/ # generated protobuf/grpc code +protos/rvbox/v1/ # existing wire authority +docs/examples/ # fully annotated server/client TOML +deploy/ + Dockerfile.toolchain + Dockerfile.runtime + compose.yaml + nginx/ + systemd/ +``` + +Use a single Go module rooted at the repository. Select and pin exact Go module +versions in `go.mod`; record reasons for non-standard dependencies in +`docs/dependency-decisions.md`. Prefer a pure-Go SQLite driver to avoid a C +toolchain for normal Linux/Windows client builds. Use a maintained, +context-aware WebSocket implementation and a Zstandard implementation that +supports bounded decompression. Do not rely on an archived WebSocket package. + +### 3.2 Containerized developer tooling + +Add: + +- `deploy/Dockerfile.toolchain`: pinned Go base image plus Buf, `protoc`, + `protoc-gen-go`, `protoc-gen-go-grpc`, format/lint tools, and `make`. +- `deploy/Dockerfile.runtime`: minimal non-root image for server/client smoke + tests; it contains only built binaries and runtime CA/config assets. +- `deploy/compose.yaml`: a `toolchain` service with the repository mounted at + `/workspace`, run as `1001:1001`; optional `server`, `client`, `nginx`, and + fault-injection test services use isolated named volumes. +- `Makefile`: `generate`, `fmt`, `lint`, `test`, `test-race`, `test-integration`, + `build`, and `verify`. Each target invokes the container workflow; it must not + silently fall back to host tools. +- `buf.gen.yaml`: Go protobuf and gRPC generation paths under `gen/go`. + +Generation is deterministic: `make generate` followed by `git diff --exit-code` +must be clean in CI. Update `buf.yaml`/`buf.gen.yaml` only with a matching +generation run. + +### 3.3 Initial quality gates + +Add CI (or a repository script ready for CI) that runs, in order: + +1. `buf format --diff` and `buf lint`. Snapshot only the already accepted + enum-prefix/service-suffix findings so CI still fails if their set changes or + another warning appears. This pre-implementation cleanup defines the first + v1 compatibility baseline; enable `buf breaking` against that baseline + immediately after it merges, not against the obsolete draft. +2. Generation freshness check. +3. `go fmt`, `go vet`, static analysis, and unit tests. +4. Race tests for server/client concurrency packages. +5. Linux integration tests in Compose. +6. Cross-compilation/build verification for Windows client packages; Windows + runtime tests run on a Windows runner once available. + +**Exit criteria:** all three empty `main` packages build in the toolchain +container; generated code is checked in; `make verify` works from a fresh clone. + +## 4. Shared contracts, configuration, and state machine (Phase 1) + +### 4.1 Preserve proto authority + +Generate `common.proto`, `agent.proto`, and `control.proto` before writing +transport code. Do not fork request/response structs by hand. These schemas are +an unshipped first draft, so remove obsolete fields and compact their tags now +rather than preserving compatibility holes. Once Phase 0 generated APIs merge, +freeze field meanings: later additions use new tags, enum zero remains +unspecified, and removed tags are then reserved normally. + +At the application boundary, validate what protobuf cannot express: + +- `client_id` is configured hostname text, 1–128 ASCII characters. +- command and mutation IDs are canonical RFC 9562 UUIDv7 strings; all public + UUID parsing rejects empty, non-canonical, or wrong-version input. +- decoded envelopes are <= 1 MiB; output/stdin/script data validates every raw, + compressed, and declared size before allocation or decompression. +- a serialized `ExecutionSpec` is <= 768 KiB, decoded control gRPC requests are + <= 16 MiB, JSON-RPC HTTP bodies are <= 24 MiB, and decoded field limits remain + identical across the two control transports. +- shell type is explicit or assigned only to the documented platform default. +- CWD exists, is a directory, and is allowed by local daemon policy. +- an execution request has exactly one source: command text or script descriptor. +- script descriptors and payloads agree on SHA-256 and <= 10 MiB size. +- user-supplied script filenames are display-only basenames and never path + components; command text and scripts use generated private wrapper paths. +- profile flags are deduplicated and checked for configured supported limits; + `LIGHT` is exclusive, otherwise accept at most one CPU, one memory, and one + disk tier. + +### 4.2 Domain package + +Implement a pure-Go `internal/domain` package with no database/network imports. +It contains: + +- the allowed command transition table and terminal-state predicate; +- `CommandRevision` compare-and-transition helpers for launch/cancel/signal + races; +- typed domain errors mapped to gRPC status/structured details, JSON-RPC error + objects, or agent-protocol `ControlError` as appropriate; +- event-sequence validation, duplicate equivalence checks, and declared gap + validation through `OutputTruncation` metadata; +- default values and hard-limit validation; +- stable ordering/pagination token encoders. + +Model `launch_prepared` and `launch_authorized` as durable internal execution +phases without exposing them as additional public lifecycle enum values. No +recovery path may redispatch a UUID after `launch_authorized`; an uncertain +outcome becomes `interrupted`. + +The normal legal state transitions are: + +```text +queued -> dispatched -> accepted -> running -> succeeded|failed|terminated|interrupted +queued -> expired +dispatched -> rejected (permanent pre-acceptance failure) +dispatched -> queued (transient rejection with backoff) +accepted -> rejected (permanent upload/preparation failure before launch) +queued|dispatched|accepted -> cancelled (subject to revisioned race resolution) +dispatched|accepted -> interrupted (reconciliation/recovery invariant failure) +``` + +An attempted repeat event is harmless only if it has identical immutable +content. A conflicting duplicate UUID/sequence is a protocol error and fences +the bad session rather than rewriting history. + +### 4.3 Configuration model + +Use strict TOML 1.0 and implement the normative contract in +[`configuration.md`](configuration.md). Keep the annotated +[`examples/server.toml`](examples/server.toml) and +[`examples/client.toml`](examples/client.toml) synchronized with the Go config +structs and compiled defaults. + +Implement configuration in this order: + +1. Define separate decode-only TOML structs and immutable validated domain + structs. Every public TOML key has an explicit tag and documentation comment; + packages outside `internal/config` receive only a validated subsection. +2. Decode exactly one UTF-8 file with unknown-field and duplicate-key rejection. + Do not merge include files, expand environment variables, coerce strings to + numbers, or accept bare numeric durations. +3. Apply precedence deterministically: compiled defaults, file values, then + flags that were explicitly present. Generate flag bindings from one key + registry so omitted flags cannot accidentally overwrite TOML values. +4. Parse duration strings with overflow and non-negative checks. Parse byte/count + integers without floating point. Enforce v1 hard ceilings even when TOML asks + for more; operators may lower them but not create an incompatible wire peer. +5. Canonicalize data/state/socket/CA/shell/CWD-root paths to absolute paths. + Reject aliases that collapse distinct server data subdirectories and a + control socket outside its intended runtime-directory policy. Check existing + path ownership/type without claiming that private modes encrypt stored data. +6. Resolve and validate current-platform shells at client startup. Each + nonempty applicable path must be absolute and identify an executable regular + file; other-platform strings are syntax-checked but never advertised. + Advertise only validated entries and require the current platform default to + be one of them. Cache the canonical path and never re-resolve through a + command's overridden `PATH`. Revalidate identity before launch and reject + rather than fall back if the executable changed incompatibly. +7. Validate resource profiles by dimension. `LIGHT` is exclusive; otherwise + allow at most one `CPU_*`, one `MEM_*`, and one `DISK_*`. Each enabled profile + must give nonzero values for its `required_controls`; validate Linux device + keys as canonical `major:minor` values and Windows Job rate support before + advertising the profile. +8. Enforce cross-field invariants: low watermark below high, send window below + durable tier, closeout reserve below command quota, command quota below both + aggregate tiers, reconnect initial <= cap, heartbeat idle < liveness timeout, + positive page/queue limits, per-client server queue <= global server queue, + and zero only where the schema documents disable or indefinite behavior. +9. Print the normalized effective non-command configuration once after + validation. Log a conspicuous warning for enabled non-loopback JSON-RPC, but + retain the accepted behavior. Do not log TLS file contents or profile device + discovery details that reveal unrelated host paths. +10. Bind no command listener and launch no child until static configuration is + valid. Storage recovery remains asynchronous after this static gate as + specified in Phase 2. + +Server TOML covers routing/listeners, queue/retry bounds, unified storage and +audit limits, group-commit behavior, flow windows/watermarks, protocol ceilings, +takeover TTL, and observability. Client TOML covers WSS/TLS routing, durable +identity/state, exact shell paths, allowed CWD roots, queue/concurrency, +reconnect/liveness, spool/flow limits, execution diagnostics/grace periods, +resource profiles, and observability. + +Do not add or claim application-level at-rest encryption in v1. Command-owned +payloads are stored in plaintext under the private state directories; document +the equivalent sensitivity of backups and defer encryption/key management. + +Hard defaults are: 10-second heartbeat-idle interval, 30-second liveness +timeout, 1–60-second full-jitter reconnect, reset after 60 stable seconds, 16 +running/100 queued client commands, 1,000 server-queued commands per client and +10,000 server-wide, 5-minute one-shot live-conflict takeover grant, 15-minute +queue TTL, 64 KiB raw stream +chunk, 1 MiB decoded agent envelope, 768 KiB serialized execution spec, 16 MiB +decoded control request, 24 MiB JSON-RPC HTTP body, 10 MiB output window/raw +script cap, 32 MiB total per command, 256 MiB per client/client daemon, 4 GiB +server-wide command storage, 30-day terminal retention, 100 MiB audit storage +with quota-only rotation by default (zero age retention), +one million compact command tombstones, raw-output high/low watermarks of +1 MiB/256 KiB per command, 8 MiB/4 MiB per client, and 64 MiB/32 MiB server-wide, +plus unacknowledged send windows of 1 MiB per command and 8 MiB per session. +Reserve 64 KiB within every accepted command's quota for terminal/loss closeout +metadata and bound protocol detail/reason plus incident-note text to 4 KiB. +Use emergency filesystem free-space floors of 256 MiB server-side and 64 MiB +client-side by default. + +Test both annotated example files through the production decoder. Add table tests +for every unknown key, invalid duration/size, high/low inversion, tier inversion, +zero semantic, noncanonical path/device, unsupported default shell, conflicting +profile combination, and flag-precedence case. Add a test proving a request-level +`PATH` override cannot change the chosen shell executable. Golden-test the +redacted effective configuration and keep its key order stable enough for +operators to compare deployments. + +**Exit criteria:** state-machine and configuration tests cover all legal/illegal +transitions and boundary values; both examples parse to the documented defaults; +generated protos compile and are the only DTOs crossing process boundaries. + +## 5. Server persistence and retention (Phase 2) + +### 5.1 Storage ownership and directory safety + +Implement `internal/server/store` as the sole writer for server state. On first +boot, create a private data directory (mode `0700`), SQLite database, segment +directory, and audit directory. Validate that paths resolve below the configured +data directory; never construct a path directly from unvalidated client IDs or +filenames. Use encoded UUID/client-ID path components. + +Open SQLite with WAL enabled, foreign keys enabled, a busy timeout, and an +explicit durable-write policy. Run migrations transactionally. On startup: + +1. acquire a single-instance lock; +2. bind liveness and diagnostic/control endpoints with readiness false; +3. run integrity/schema checks and committed-range recovery asynchronously; +4. safely truncate file bytes beyond SQLite's `committed_end_offset`, while a + missing/corrupt committed range creates a scoped durable incident; +5. mark prior live sessions disconnected and retain trustworthy non-terminal + commands for reconciliation; interrupt only affected commands whose + essential state cannot be trusted; +6. enable each healthy scope independently. Mutations against an unrecovered or + dirty scope return `UNAVAILABLE`; liveness, incident inspection, and + unaffected scopes remain available throughout best-effort recovery. + +Represent recovery with explicit atomic scope states: `recovering`, `ready`, +`dirty_readable`, and `unavailable`. Liveness depends only on the process/event +loop; global readiness is true only when every required scope is ready, while an +RPC checks the narrower scopes it will touch. The WebSocket endpoint may accept +and close with a retryable storage error during recovery, but it must not issue a +durable `ServerWelcome`, accept events, or dispatch commands until the relevant +client/session scope is ready. + +Run `PRAGMA quick_check`, migration checksum validation, and segment reconciliation +on a bounded recovery pool. Persist a startup incident in the database when it +is writable; otherwise expose an in-memory bootstrap incident through health and +require the offline repair tool. Never loop forever on one corrupt client or +segment. Record progress counters and the last failing path/offset without +putting payload bytes in logs. + +### 5.2 Metadata schema and indexes + +Use migrations to create at least the following tables; names may vary but their +invariants must not. + +| Data | Required fields/invariants | +| --- | --- | +| `clients` | client ID, most-recent platform/capabilities/CWD/version, durable instance ID, current generation, connection/last-seen timestamps, latest rejected live-conflict instance/time, unified charged command-storage total | +| `sessions` | opaque session ID, client ID, durable client-instance UUID, generation, opened/fenced/closed times, close reason | +| `commands` | UUID, client ID, indexed issue/queue-expiry/terminal times, lifecycle, revision, exit result, last event sequence, retention status, and a Zstandard-compressed immutable execution-spec payload including plaintext environment values | +| `command_payloads` | command UUID, payload kind (script or other command-owned blob), Zstandard compression, raw/stored sizes, digest, inline bytes or validated segment reference | +| `command_events` | command UUID + event sequence unique key, observed and server receipt times, event type, payload metadata, immutable duplicate checksum | +| `output_segments` | command UUID, segment ordinal/path, `committed_end_offset`, min/max event sequence, stream mix, compressed/raw byte totals, checksum, created time | +| `output_truncations` | command UUID, removed event range/byte totals, source, reason, recorded time; non-overlapping ranges | +| `stdin_writes` | command UUID + write sequence unique key, compressed payload/raw/stored sizes/checksum, newline/close intent, acknowledged state | +| `control_mutations` | request UUIDv7 unique key, owner command/incident/client, method/target, immutable request hash, assigned write sequence/revision and stable result; conflicting reuse is rejected | +| `takeover_authorizations` | client ID + exact pending instance ID, creation/expiry/consumption times and authorizing request ID; one live, one-shot grant per client | +| `command_tombstones` | binary UUIDv7 unique key, immutable request hash, client ID, completion/acknowledgement time; compact global FIFO capped at 1,000,000 rows | +| `audit_events` | timestamp, source transport/principal where available, client/command IDs, action, outcome, error code, command text/script digest/env key names | +| `storage_incidents` | incident UUID, detected/resolved times, state, affected scope/client/command, summary, repairability and data-loss flags; unresolved rows derive dirty health | +| `schema_migrations` | applied version/checksum/time | + +Index client command lists by `(client_id, issue_time DESC)`, active command +dispatch by `(client_id, lifecycle, issue_time)`, and output reads by +`(command_uuid, event_seq)`. Use cursor tokens based on the sorted key plus a +signature/version, not SQL offsets, so page results remain stable under writes. + +Add foreign keys with explicit delete behavior and database constraints for +terminal timestamps, nonnegative byte counters, unique `(command,event_seq)` and +`(command,write_seq)`, one active session per client, and one unconsumed takeover +grant per client. Store UUIDs and SHA-256 values as fixed-size blobs internally; +canonical strings exist only at protocol/log boundaries. Canonical request hashes +use deterministic protobuf encoding of immutable fields after defaults and map +ordering are normalized; mutable lifecycle/receipt fields are excluded. + +Every mutation transaction updates its owner record, all three applicable quota +counters, the idempotency row, and audit intent/result consistently. Identical +`request_id` plus hash returns the stored result without re-running side effects; +same ID with a different method, owner, or hash returns `CONFLICT`. Add migration +invariant queries that recompute charged totals and fail readiness if persisted +counters disagree until repaired. + +### 5.3 Segment format and atomic append + +Store large command, output, and audit payloads outside SQLite. An active segment +is append-only; seal it at the configured 256 KiB target and never append again. +Use an encoded UUID/ordinal filename created with exclusive create. Each record +contains fixed magic/version/header length/record length, command and event +identity, observed/receipt timestamps, payload kind/stream/compression, raw and +stored lengths, SHA-256 payload digest, header CRC, payload, and trailing record +CRC. All lengths are unsigned and validated against configured maxima before +allocation or seeking. + +The append path is: + +1. validate compressed bytes and bounded decompression; +2. choose/append the active command segment; +3. append and `fsync` (possibly as a configured group commit); +4. in one SQLite transaction, insert event/segment metadata, advance that + segment's `committed_end_offset`, update command sequence/counts, and record + receipt time; +5. only after durable success, return the cumulative `EventAck`. + +The writer serializes appends per active segment but group-commits independent +commands together. Creating/renaming a segment also syncs its containing +directory before metadata can commit. A group-commit timer is a maximum batching +delay, never permission to acknowledge before file sync. SQLite uses a +documented synchronous mode sufficient for the selected durability promise; +tests assert the actual PRAGMA values after opening every connection. + +On recovery, file bytes beyond the committed offset are unacknowledged tail and +are truncated. A short file or checksum failure inside the committed range is +never “repaired” by silently moving the database offset backward: mark the +affected output unavailable/truncated, preserve evidence, and create an +incident. Essential-state corruption interrupts only the affected command and +gates unsafe mutations. + +Apply this committed-offset protocol to command blob and audit segments too. +For a missing committed output range, preserve all still-valid records, create +query-visible truncation/incomplete metadata, and never fabricate byte totals. +For a missing immutable execution spec, request hash, revision, or active stdin +state, mark the command essential state untrustworthy and interrupt it. Recovery +must be idempotent after a crash at every repair step. + +If database or segment persistence fails, do not acknowledge the event. Surface +the server health failure, stop dispatching work as appropriate, and keep the +session/control loops responsive. + +### 5.4 Retention transaction + +Implement retention in a single serialized maintenance worker, never on the +WebSocket receive loop. + +Define a versioned deterministic charge formula: compressed/encoded payload and +segment lengths plus a conservative fixed charge for each SQLite row/index +entry. Do not attempt to attribute shared SQLite pages after the fact. Reconcile +the logical counters transactionally and also reject allocations below the +configured 256 MiB server filesystem free-space floor, regardless of logical +quota headroom. + +Implement one server `Reserve(owner, kind, chargedBytes)` path used by command +creation, script append, stdin, event append, and audit. Reuse the same contract +in the client spool, where it also covers raw execution-file preparation. Under +the store writer lock/transaction it checks, in order: hard field maximum, +per-command total and closeout reserve, per-client total, server/client-daemon +total, and filesystem floor. It either reserves all applicable tiers or none. +Release uses the same stored charge version; a future estimator change requires +a migration, not a silent reinterpretation of old rows. + +1. Account all stored command-owned data, including request/script payload, + pending stdin, metadata/events, and output. The client additionally charges + generated raw execution files while they exist. + Reserve non-evictable active state + before accepting it, including 64 KiB per-command closeout headroom and the + bounded result for each accepted stdin/signal mutation; reject an allocation + that cannot fit. +2. Enforce the command's 10 MiB output window and 32 MiB total cap by rotating + oldest output segments and recording `OutputTruncation`; never roll essential + active state. +3. Enforce the 256 MiB per-client and 4 GiB server-wide caps by evicting terminal + commands as whole UUID records in oldest server issue-time/UUIDv7 order. If active data alone creates + pressure, rotate/drop output with markers and reject new essential state. +4. Reclaim every whole terminal command 30 days after authoritative terminal + time by default, independently of byte pressure; zero explicitly disables + age rotation. Preserve compact tombstones and separate audit/incidents. +5. Maintain audit events in compressed segments under their independent + 100 MiB default, rotating complete oldest segments. Apply an independently + configurable audit age as a second trigger; zero, the default, disables age + rotation but not byte-budget rotation. +6. Keep unresolved storage incidents non-evictable. Resolved incident summaries + may rotate with their audit history under the separate 100 MiB budget, but + resolving dirty health never immediately rewrites or disguises recorded loss. + +Retention selects candidates deterministically in `(server_issue_time, +issue_uuid)` order and rechecks terminal state plus generation inside the delete +transaction. Before deleting a command, close follower snapshots over its +retained ranges, ensure the compact tombstone exists, mark `evicting`, and move +all owned files to a same-filesystem deletion directory using collision-proof +names. A crash-safe sweeper rolls forward rows/files in either order. Never use +a glob or client-provided path for deletion. + +For active pressure, first evict server-retained output and record exact +`RetentionTruncation`; then reject new unreserved mutations. Do not evict the +assigned client send window or essential lifecycle/revision/dedupe state. Age +rotation and byte-pressure eviction call the same whole-command primitive so +their crash behavior cannot diverge. + +Deletion is recoverable: mark rows `evicting` transactionally, move segment +files to a same-filesystem tombstone location, commit metadata deletion, then +remove tombstones asynchronously. Startup completes or rolls forward stale +evictions deterministically. + +Replay tombstones must never have a delete/insert gap. The server inserts the +command tombstone transactionally with authoritative terminal state before it +can acknowledge or later evict command history. On the client, receipt of an +acknowledgement through the terminal event atomically inserts or retains the +UUID/request-hash tombstone and removes the full command record. Enforce each +one-million-entry FIFO cap in that same transaction. + +### 5.5 Incident resolution + +Run safe, deterministic repairs automatically and through the control API. +`RepairStorageIncident` must refuse any action that would knowingly discard +committed data. `AcknowledgeStorageIncident` requires an operator note and is +the explicit path for accepting irrecoverable loss. Both are idempotent by +`request_id`. A repaired or acknowledged incident no longer contributes to the +derived dirty flag; resolution itself does not delete history. Bound notes to +4 KiB and retain resolved summaries/audit entries under their separate 100 MiB +rotation. + +Provide the same store library behind offline `rvbox-server repair --data-dir` +list/repair/acknowledge modes. Provide equivalent local client-spool handling +through `rvbox repair --state-dir` and its local health diagnostics. Never open +a live store from the offline tool; acquisition of the same single-instance lock +is required. + +Define stable incident kinds for uncommitted tail, missing committed bytes, +checksum mismatch, counter mismatch, SQLite integrity failure, failed eviction, +permission error, and disk exhaustion. An incident has immutable evidence plus +mutable state `OPEN -> REPAIRED|ACKNOWLEDGED`; terminal states never reopen, so a +recurrence creates a new incident ID. Scope keys identify global store, client, +command, segment, or audit without embedding raw filesystem paths in the public +API. + +Safe repair may truncate only bytes beyond `committed_end_offset`, rebuild a +derivable counter/index, or roll forward a marked eviction. Anything that would +discard acknowledged bytes or immutable essential state requires acknowledgement +instead. Online repair acquires the same per-scope maintenance lock as recovery +and retention; offline repair additionally requires the process-wide instance +lock. Persist the repair/acknowledgement mutation and audit record before clearing +derived dirty health. + +### 5.6 Store tests + +Test real SQLite/filesystem cases: duplicate event idempotence; conflicting +duplicate rejection; crash between segment write and metadata transaction; +crash after metadata commit; corrupted final record; command-window rotation; +per-client/global whole-command eviction; active-command protection; cursor +stability; uncommitted-tail truncation; short/corrupt committed ranges; dirty +derivation and repair/acknowledgement; and disk-full/permission failure +simulation. Verify recovery endpoints become live promptly while unsafe scoped +mutations remain gated. + +**Exit criteria:** a store restart preserves acknowledged data and never +acknowledges uncommitted data; all rolling, tiered unified-quota, 30-day +terminal-age, audit, and tombstone rules match the documented semantics. + +## 6. Server sessions, WSS endpoint, and dispatch (Phase 3) + +### 6.1 Agent endpoint + +Expose an HTTP endpoint suitable for nginx WebSocket proxying. TLS may terminate +at nginx, but the server must verify WebSocket upgrade origin/path/configuration +and impose read limits itself. It accepts one binary protobuf envelope per +message and refuses text messages, fragmentation misuse, oversized frames, +invalid compressed payloads, and malformed/unknown mandatory messages. + +Use four independently bounded activities per session: + +- a read loop that validates/fragments/decode-dispatches only; +- a prioritized write loop with reserved Ping/Pong/Close/fencing/error capacity, + bounded data frames, and write deadlines; +- a dispatch worker that reads persisted command work; +- a persistence/event worker that writes through `store`. + +No goroutine may hold a session-registry lock while doing database, filesystem, +compression, or WebSocket I/O. Every goroutine receives a session context and +must terminate on fencing/close. + +Keep at most 1 MiB unacknowledged per command and 8 MiB per client session; +additional data waits in durable storage. If an essential ingress queue fills, +close without acknowledging rather than blocking the sole reader; the durable +sender retries after jittered reconnect. Ping/Pong remains sendable even while +the data window is full. + +Implement the write scheduler as two bounded lanes owned by one socket writer: +the reserved control lane contains Ping/Pong/Close, welcome/fencing, protocol +errors, acknowledgements, and reconciliation control; the data lane contains +dispatch, script, stdin/signal, and events. Always drain a bounded burst of +control before one data frame so data still progresses. Impose write deadlines +and split data below the frame ceiling. A full control lane is a session-fatal +internal invariant; a full data lane leaves work durable and wakes the writer +later rather than allocating another goroutine. + +The reader checks WebSocket type/size before protobuf decode, validates envelope +session fields before payload allocation, and routes only lightweight immutable +work descriptors. Persistence workers preserve per-command ordering while a fair +round-robin/deficit scheduler prevents a verbose command from monopolizing +decompression or disk. Unit-test queue capacities and goroutine shutdown with a +fake connection that never completes writes. + +Apply the server 64 MiB/32 MiB ingress backlog as admission hysteresis, not as a +license to discard client-sequenced events. Above high, stop acknowledging new +output and close the producing session if the bounded descriptor queue cannot +accept its next frame; the client's durable spool retries it. Resume normal +admission only below low. Keep lifecycle/control reserve independent so another +client's heartbeats and command closeout remain responsive. + +### 6.2 Registration and fencing + +On `ClientHello`, validate client ID/capabilities/version and the durable random +client-instance UUID; choose the highest shared minor under major v1; persist +the registration; allocate a cryptographic random opaque session ID and +increment the client's generation transactionally. Fence the prior session for +the same client ID and instance before making the new one active. Accept a +different instance normally if no session for that client ID is live. Otherwise +record it as the pending claim and reject it unless an unexpired one-shot grant +matches that exact instance; consume the grant transactionally when accepting +the reconnect and fencing the old session. The default grant TTL is 5 minutes. +Send `ServerWelcome` with the outer envelope's canonical session fields. + +Persist a rejected live-conflict claim before closing the candidate socket and +expose only its exact `client_instance_id` and observation time through +`GetClient`. `AuthorizeClientTakeover` verifies that this ID still matches the +pending claim, stores an idempotent UUIDv7 mutation and five-minute expiry, and +does not fence immediately. On matching reconnect, one transaction increments +generation, consumes the grant, closes the old session record, and installs the +new session; only then may the writer send welcome. A mismatched/expired grant is +never broadened to “the next instance.” If no session is live, accept a different +instance without creating or consuming a grant, as agreed. + +All post-hello envelopes must match both session ID and generation. Stale or +unknown session traffic is discarded/audited and its WebSocket is closed. A +dispatch includes the intended generation, and the client must reject a mismatch. +Unknown client IDs are accepted by design; hostname is display/routing identity, +not an authentication assertion. + +### 6.3 Heartbeat and liveness + +Implement the identical policy at both ends: any received valid WebSocket frame +updates inbound activity; after 10 seconds silent send Ping; after 30 seconds +without inbound frame close the session. Pings/Pongs use WebSocket controls, not +protobuf envelopes. Make timers injectable/fakeable for deterministic tests. + +Use a monotonic clock for elapsed activity/backoff and wall-clock UTC only for +persisted observation. Pong/any valid inbound frame updates activity in the read +loop without waiting for a database worker. Reset reconnect backoff only after a +continuous 60-second stable session. Add boundary tests at exactly 10 and 30 +seconds, delayed writer tests proving Ping bypasses a full data lane, and clock +jump tests proving wall-clock adjustment cannot spuriously kill a session. + +### 6.4 Queueing, dispatch, and reconciliation + +`RunCommand` persists a `queued` command before returning. The dispatcher sends +to the active session while either an immediate running slot or a client queue +slot is available: `running < max_running || queued < max_queued`. Dispatch and +capacity updates are serialized per session so concurrent sends cannot +oversubscribe the advertised slots. It updates to `dispatched` before write and +retries after reconnect until a matching `CommandAccepted` arrives. + +Maintain per-session shadow reservations for in-flight dispatches: reserve a +running slot when immediately available, otherwise a queued slot, before +enqueuing the frame. Reconcile the shadow with each `ClientCapacity` and +`CommandAccepted`; release it on definite rejection/session close. The +dispatcher uses one serialized worker per client plus a fair global ready queue, +so the `OR` capacity condition cannot oversubscribe and an offline/noisy client +cannot starve others. + +Apply the configurable queue TTL (15 minutes by default; zero means indefinite) +until server-confirmed acceptance. A never-dispatched command becomes terminal +`expired`. A dispatched command with uncertain acceptance remains protected and +is rendered expired pending reconciliation rather than being prematurely made +terminal. Late client evidence advances the actual lifecycle, creates a +late-after-expiry incident, and triggers best-effort termination while retaining +subsequent actual outcome events. + +Persist an absolute server-clock `queue_expiry_time` when creating the command; +never recompute it after configuration changes or retries. Drive expiry from a +database-backed ordered index and recheck lifecycle/revision transactionally. +Queued commands become `EXPIRED`; dispatched commands set a derived +late-pending-reconciliation indicator without writing a false terminal event. +On late acceptance, persist the incident and higher termination revision before +sending best-effort TERM, then retain every actual subsequent event. + +Classify a negative `CommandAccepted` before changing lifecycle. Explicitly +transient capacity/storage pressure returns `dispatched -> queued` with bounded +jittered backoff and remains under the original TTL. Invalid execution data or +unsupported required platform controls become terminal `rejected` with the +structured reason persisted on `CommandRecord`. Reserve `failed` for a process +that reached `running`. + +On replacement/reconnect, query non-terminal commands for that client and send +a `ReconcileRequest` with last durable event sequence/revision plus immutable +request hash. Require one complete client snapshot of all locally retained +commands, including +terminal-but-unacknowledged records and matching requested tombstones, and +compare the sets idempotently before enabling dispatch. With healthy client +storage, an absent `queued`/`dispatched` command returns to `queued`; an absent +`accepted`/`running` command is interrupted as an invariant failure. A matching +UUID/request-hash tombstone suppresses replay. The client terminates a local +active command absent from the server target set only after receiving the +server's durable `ReconcileResult`; either peer records a contradiction with +server-confirmed terminal state as a storage/recovery incident, and the server +does not invent missing state. The result also lists terminal local records safe +to discard when the server has stored or deliberately tombstoned them, covering +a lost final `EventAck` followed by server retention. Write this idempotent +result before dispatch. A client with unresolved essential-store corruption +cannot complete reconciliation or accept work. + +Implement reconciliation as this explicit matrix: + +| Server state | Client evidence | Durable result | +| --- | --- | --- | +| queued/dispatched | absent from healthy complete snapshot | return/remain queued; redelivery permitted | +| dispatched/accepted/running | matching retained record | keep highest valid revision and resume event/script delivery | +| any non-terminal | matching tombstone | interrupt/suppress replay and record stale-server incident | +| accepted/running | absent | interrupt and record client-state-loss incident | +| missing or contradictory terminal | client active | `ReconcileResult.terminate_local_issue_uuids` plus incident; invent no server command | +| missing/tombstoned or fully stored terminal | client terminal | `discard_local_terminal_issue_uuids` | + +Validate matching UUID, immutable request hash, revision monotonicity, terminal +immutability, and event-sequence bounds for every row before applying any row. +Persist all server state/incident decisions in one reconciliation transaction, +then send one deterministic sorted `ReconcileResult` through the control lane. +The client durably records terminate/discard intent before reporting capacity. +If the result frame is lost, the next complete snapshot yields the same result; +dispatch is disabled until the current result has been written on the session. + +Handle pre-start `kill` locally only for work that has never been dispatched. +For dispatched or accepted work, persist a higher revision and send the +revisioned signal while retaining the existing visible lifecycle until the +client resolves the race. Accept a revisioned `cancelled` lifecycle if launch +was not authorized; otherwise retain the revisioned signal result and actual +terminal outcome. The server maps control `request_id` to this revision/result +and never sends that request ID across the agent protocol. + +### 6.5 Server session tests + +Cover version negotiation, same-instance replacement, different-instance +pending claim, exact one-shot takeover consumption/expiry, stale-generation +late output, Ping/Pong inactivity thresholds, full client queue, dispatch retry +after lost acceptance, complete active/terminal-unacknowledged reconciliation, +lost/repeated `ReconcileResult`, permanent/transient dispatch rejection, +cancellation before launch, launch/cancel race, malformed envelope close, and a +slow client that cannot block another client or local control RPC. + +**Exit criteria:** multiple simulated clients can register, replace each other, +receive bounded dispatches, and recover connection faults without duplicate +execution or server deadlock. + +## 7. Client runtime, spool, and Unix supervisor (Phase 4) + +### 7.1 Client state and reconnect runtime + +Keep client state private (mode `0700`): durable accepted-command records, +active process metadata, command event journal, stdout/stderr spool segments, +stdin write acknowledgements, and outstanding script upload state. It contains +no terminal command history after server acknowledgement and local cleanup. +It retains a separate compact FIFO of the most recent 1,000,000 command +tombstones (binary UUIDv7, immutable request hash, and acknowledgement time) so +stale server state cannot replay recently completed work after full history is +removed. + +The runtime has a long-lived supervisor/dispatcher and replaceable network +session. Generate the random client-instance UUID once in the private state +directory and retain it across daemon restarts and reconnects. Network loss must +not stop running commands or pipe readers. The +connection loop uses full-jitter exponential backoff from 1 second to 60 +seconds, resetting only after 60 stable seconds. On each connection it sends a +hello with the durable client-instance UUID, waits for welcome, replays retained +sequenced loss events and other unacknowledged events in order, answers +reconciliation, and resumes dispatch. + +All client queues are bounded. The disk spool, not an in-memory channel, is the +source of truth. Give stdout, stderr, status, signal results, and stdin +acknowledgements one durable command-local order. Assign and persist wire +`event_seq` only when an entry enters the bounded send window; an assigned entry +is pinned until acknowledgement and immutable on retry. + +Create `client_instance_id` with a cryptographic UUID generator, write and fsync +a temporary file, atomically rename it, and sync the state directory before +first registration. Never regenerate it merely because the network/server +rejects a session. If the identity file is corrupt, enter dirty local health and +require the repair path rather than silently presenting a new installation. + +Give the client store the same committed-offset segment discipline as the +server. Persist, at minimum, command phase/revision/hash, platform launch +identity, local event ordinal, optional assigned wire sequence, payload digest, +stdin write state, script received ranges, and last server acknowledgement. +Maintain separate monotonic `local_ordinal` and `event_seq` columns: insertion +assigns only local order; admission to the send window transactionally assigns +the next contiguous wire sequence and pins the row. Cumulative `EventAck` +atomically unpins/deletes acknowledged payload and updates the command cursor. + +Structure reconnect as explicit states: `backoff -> connecting -> hello -> +reconciling -> active -> closing`. Only pipe/supervisor workers outlive the +replaceable network session. After welcome, send the complete snapshot, apply +`ReconcileResult` durably, replay assigned rows first, then assign/send local +rows in order. Advertise capacity and accept new dispatch only in `active`. +Cancellation of one session context must join all its readers/writers before a +new session can use their queues. + +### 7.2 Output capture and offline caps + +Create non-blocking readers for stdout and stderr immediately after process +start. Read raw bytes irrespective of newline boundaries, divide them into <= +64 KiB raw chunks, add observed UTC time/local order, Zstandard-compress, append +to the command spool, and enqueue only a durable reference to the sender. + +Use one reader per pipe and a bounded shared raw-chunk scheduler. A reader never +waits for network I/O. Under normal load it transfers ownership of a bounded +buffer to a fair compressor, which returns buffers to a pool only after durable +append. Validate Zstandard frame size/checksum on readback before replay. Do not +log raw chunks on compression, checksum, or storage failure. + +Bound work before compression as well as on disk. Defaults are high/low raw +backlogs of 1 MiB/256 KiB per command and 8 MiB/4 MiB for the client daemon; the +server compression/persistence tier independently uses 64 MiB/32 MiB globally +with the non-dropping admission behavior in Phase 3. Crossing a client high +watermark enters loss mode until the matching low watermark. Aggregate +discarded, still-unsequenced chunks into local-order/raw-byte loss records and +replace them with a `CLIENT_OVERLOAD` truncation marker; event-range and +compressed-byte counts are absent when sequencing/compression never occurred. +Use fair worker scheduling and reserved metadata capacity so lifecycle, +stdin/signal +acknowledgements, gap markers, and terminal state remain lossless. + +Loss mode is hysteretic and scoped: crossing either command or daemon high +watermark marks affected raw chunks as dropped until both relevant backlogs fall +below their lows. Coalesce only adjacent dropped local ordinals with identical +cause/stream set; close a loss run before inserting any essential event. Persist +known raw-byte totals and observation interval in local metadata, then emit one +bounded `OutputTruncation`; omit event-range/compressed-byte fields if they were +never assigned/compressed. If even reserved marker persistence fails, mark the +command/store dirty and interrupt rather than silently claiming complete output. + +Respect the local hard limits at all times: + +- 10 MiB compressed rolling output window and 32 MiB total per command; +- 256 MiB aggregate command-owned storage per client daemon. + +When connected, discard server-acknowledged oldest segments first. Assigned but +unacknowledged entries are pinned and bounded by the 1 MiB per-command send +window. When offline or an acknowledgement lags, rotate only still-unsequenced +output as needed and replace each removed run in durable local order with +`OutputTruncation` using `CLIENT_SPOOL`. The marker later receives a normal +wire sequence and is retained until cumulative `EventAck`; no wire sequence is +fabricated or skipped. The server's accepted truncation event is the auditably +visible explanation for the lost bytes. + +Continue draining child pipes after caps are reached. A verbose child may lose +old or newly produced output, but it must not block the client daemon or +deadlock itself. + +Test a continuously verbose process, alternating stdout/stderr, essential events +interleaved with dropped output, reconnect during loss mode, ack loss at every +send-window boundary, and terminal exit while above the high watermark. Assert +contiguous assigned sequences, exact known raw-byte totals, bounded heap/queue +size, and a terminal event after every marker/incomplete event. + +### 7.3 Command acceptance and idempotency + +Persist an accepted dispatch record keyed by `issue_uuid` before responding +`CommandAccepted`. Repeat dispatch with identical immutable fields returns the +same acceptance/revision and does not start another child; a conflicting repeat +is a protocol error. Enforce 16 running/100 queued defaults before acceptance. +After full terminal cleanup, also check the compact tombstone ledger: the same +UUID/hash is rejected as already executed, while changed immutable content is a +conflict. + +Make acceptance a single durable state transition: + +1. Validate the session generation, UUIDv7 syntax, immutable request hash, + command revision, shell/profile/CWD policy, script descriptor, and expiry. +2. Look up both the active-command table and tombstone ledger before reserving + capacity. An exact active duplicate returns its existing result; an exact + tombstone returns `CODE_ALREADY_EXECUTED`; either hash mismatch is a + permanent conflict. The server treats the former as stale-state + reconciliation/interruption, not as a newly rejected execution. +3. Reserve the running/queued slot and worst-case command metadata allocation in + the same local-store transaction that inserts the accepted command. Do not + count retransmission of an existing record twice. +4. Commit and fsync the accepted record, immutable hash, revision, script + cursors, and capacity counters before sending positive acceptance. A failure + before commit returns a transient rejection and leaves no partial command; + loss of the response is resolved by exact replay. +5. Classify policy/schema/hash/unsupported-profile failures as permanent and + local I/O/temporary-capacity failures as transient. Include a bounded, + operator-readable reason without command or environment contents. + +For script dispatch, persist descriptor and chunks at validated offsets, verify +each chunk checksum and final SHA-256/length before launch, and keep the upload +restart-safe until terminal cleanup. Create the final temporary file with +owner-only permissions under the effective CWD using an atomic write/rename; +delete it after the terminal event is durable locally. Emit cumulative durable +`ScriptUploadStatus.received_bytes` progress often enough to advance the 1 MiB +server send window; the server never streams the full 10 MiB without progress +acknowledgement. + +Accept a chunk only when its offset equals the durable contiguous prefix, or +when the entire byte range is an exact replay of already persisted bytes. +Reject holes, changed overlaps, arithmetic overflow, data past the declared +length, and chunks received after commit. Append and fsync bytes before +advancing `received_bytes`; after the final prefix, verify length and SHA-256, +reserve the raw-file quota, atomically materialize the wrapper, sync its parent +directory, and only then make the command launch-eligible. A permanent upload +failure durably ends the command as `REJECTED` before launch authorization and +releases every associated reservation. + +Materialize ordinary `command_text` through the same private generated-file +machinery, while retaining its separate command-text metadata. Execute the file +with exactly `sh`/`bash`, `cmd.exe /D /S /C`, or `powershell.exe` with +`-NoLogo -NoProfile -NonInteractive -File` according to `shell_type`; never infer or +fallback to another shell. Build Windows application name/arguments with the +platform quoting routine rather than shell string concatenation. Treat `$0`, +`%0`, or equivalent exposing the generated wrapper name as documented behavior. + +Resolve each shell setting once during configuration validation to a canonical +absolute executable path. Launch that exact path with an explicit argument +vector/application name; never search the daemon's `PATH`, honor a command's +environment overrides during resolution, follow a wrapper shebang, or select by +file extension. Re-stat the configured executable immediately before launch and +reject if it is no longer the validated regular executable. Add tests with an +attacker-controlled `PATH` entry and same-named fake shell to prove it cannot be +selected. + +### 7.4 Unix process supervisor + +Define a narrow interface used by the runtime: + +```go +type Supervisor interface { + Start(context.Context, StartSpec) (Process, error) + Signal(context.Context, Process, SignalKind) (SignalOutcome, error) + Snapshot(context.Context, Process) (ResourceSnapshot, error) + StopAll(context.Context) error +} +``` + +The Unix implementation starts an RVBox launcher as leader of a new +session/process group; after authorization that launcher creates the selected +`sh`/`bash` child inside the same group and remains as its watchdog. It rejects +unsupported shell values. It applies daemon environment plus persisted +overrides, validates the CWD, connects stdin/stdout/stderr pipes, and records +launcher/root/group identities. Signal the full process group. +On orderly shutdown and unclean-start recovery, terminate surviving managed +groups and emit `interrupted` rather than pretending pipe monitoring survived. + +Treat root exit as the start of a configurable 5-second tree/output drain grace +period. Wait for the cgroup/process group and capture readers; terminate residual +descendants after the grace period, drain to EOF, and emit incomplete-output +metadata as a sequenced `OutputIncomplete` event if handles still cannot be +drained. The terminal lifecycle event must be sequenced and spooled only after +all retained output/truncation events and is +always the final client event. + +Implement launch through an internal blocked-launcher mode with a private +release/watchdog channel. The launcher remains alive as a non-user-code +supervisor for the process-tree lifetime. After the launcher reports ready, +persist and flush +its PID/process-group/platform birth identity as `launch_prepared`; persist and +flush `launch_authorized` before releasing requested command code. EOF before +release aborts without executing it. On Linux create a per-command cgroup v2 +for supervision regardless of resource-profile selection, place the blocked +launcher into it before release (`clone3` with `CLONE_INTO_CGROUP` and +`CLONE_PIDFD` where available), and persist the cgroup path plus +`/proc//stat` start time. Recovery prefers `cgroup.kill`; a process-group +fallback is allowed only after positive birth-identity verification. Other Unix +platforms use the watchdog/process-group fallback and document that descendants +which deliberately create a new session may escape it. + +Encode launch as `accepted -> launch_prepared -> launch_authorized -> running` +and permit no shortcut: + +1. Create the private cgroup/process group, pipes, wrapper, and close-on-exec + release/watchdog channel without executing user code. +2. Start the blocked launcher and obtain a positive ready message containing + the platform process identity; place and verify it in its command cgroup. +3. Commit and fsync `launch_prepared` with that identity. Recheck the latest + revision and pending cancellation while the launcher remains blocked. +4. If still executable, commit and fsync `launch_authorized`, then send the + one-byte release and keep the watchdog channel open until tree cleanup. + Record `running` only after the launcher reports successful requested-shell + creation/`exec`. +5. A crash before durable preparation cleans an untrusted orphan and may retry. + A prepared-but-unauthorized launcher must exit on channel EOF and may retry + only after positive death verification. After release, channel EOF makes the + launcher terminate its process group; an authorized but uncertain launch is + killed as a tree and ends `interrupted`, never launched again. + +Make launch-journal and launcher-control records checksum-framed and bounded. +Fault-inject process death before and after every fsync, ready/release, and exec +acknowledgement; assert that user-visible side effects occur at most once and +that recovery cannot confuse a reused PID. + +Implement the portable Unix set `HUP`, `INT`, `TERM`, `KILL`, `USR1`, and +`USR2`, accepting optional `SIG` prefixes in the CLI and mapping only through +the `SignalKind` enum. Reject arbitrary native numbers and unsupported names. +Never allow signal zero or arbitrary PID targeting. The process's group ID +comes only from durable supervisor metadata, never a request field. + +### 7.5 Linux diagnostics and resource profiles + +Poll active processes at a configurable interval outside pipe/network loops. +Read readable `/proc` values for CPU time, RSS, I/O, state, CWD, and wait reason; +aggregate only clearly associated group/child data. Missing/unreadable values +remain absent. Set `suspected_hung` only after default 10 minutes without +observable progress and label it diagnostic, not lifecycle. + +Make requested resource-profile handling explicit. On Linux the supervisor +uses a delegated writable cgroup v2 for tree supervision whenever available, +even without a profile. Profiles add administrator-defined CPU/memory/disk/ +process controls to that cgroup. Without delegation, no-profile execution uses +the watchdog/process-group fallback; a profile whose essential control cannot +be applied is rejected as unsupported. No profile means no resource restriction, +even when a supervisory cgroup exists. + +Compile TOML resource profiles into immutable validated launch policies during +startup. Enforce `LIGHT` as exclusive; otherwise allow at most one CPU, one +memory, and one disk profile. On Linux, translate configured policy into the +per-command cgroup's `cpu.max`/`cpu.weight`, `memory.max`/`memory.high`, +`pids.max`, and per-device `io.max` controls as applicable. Write and read back +all essential controls while the launcher is blocked. If any essential control +is unsupported or cannot be applied, destroy the empty command cgroup and +permanently reject the dispatch; never run partially constrained. + +Define diagnostic progress as a change in process-tree CPU ticks, cumulative +I/O counters, retained output/input activity, or lifecycle state. Persist only +the latest sample and no-progress start time, clear `suspected_hung` on the next +observed progress, and tolerate counter reset/process exit. Diagnostics must not +keep a command alive, change terminal status, or block the supervisor. + +### 7.6 Client tests + +Use fake transport/clock plus real subprocess tests for shell defaults, CWD/env +overlays, at-most-once duplicate dispatch, concurrent output with no newlines, +stdin ordering/close, process-group termination, reconnect/replay, offline +rolling truncation before sequence assignment, assigned-window pinning, script +progress/checksum failure/cleanup, atomic tombstone replacement, queue limits, +shutdown interruption, and `/proc` absence. Run race tests with multiple +commands and forced network churn. + +Add a table-driven crash suite for every acceptance, script-upload, launch, +event-spool, terminal-acknowledgement, and tombstone-rotation commit point. Run +the same duplicate dispatch after each restart and assert exactly one of: +durable rejection before authorization, one supervised process, or an +`interrupted` uncertain launch—never a second execution. + +**Exit criteria:** a Linux client can stay alive through server loss/restart, +execute up to its capacity at most once, preserve/replay bounded history, and +cleanly manage full command process trees. + +## 8. End-to-end agent protocol (Phase 5) + +Wire the server session layer and client runtime together before adding the CLI. +Use real protobuf bytes through an in-memory WebSocket test server first, then +through Docker Compose with nginx proxying a WSS endpoint. + +Implement flows in this order: + +1. hello/welcome/version/fencing and heartbeat; +2. persisted background command dispatch and lifecycle events; +3. stdout/stderr event persistence and cumulative acknowledgements; +4. reconnect replay, duplicate delivery, and non-terminal reconciliation; +5. queued cancellation and revisioned signal delivery; +6. ordered stdin/close acknowledgement; +7. verified chunked scripts; +8. capacity advertisements, diagnostics, and profile results. + +At each step, add a failure-injection test that drops one frame at every +acknowledgement boundary. Required scenarios include lost `CommandAccepted`, +lost `EventAck`, stale old-session output after takeover, connection loss during +script transfer, server restart before/after event transaction, client restart +with managed process cleanup, and offline spool overflow. Verify command UUIDs +never execute twice and output gaps always appear as truncation metadata. + +Build a deterministic scenario harness with fake wall/monotonic clocks, +controllable frame loss/reordering, daemon kill points, and bounded virtual disk +capacity. For each flow, run disconnect before delivery, after delivery/before +durable commit, after commit/before acknowledgement, and after acknowledgement. +Assert database invariants and quota counters after every restart, not merely +the visible CLI result. Include these cross-cutting cases: + +- a queued command expires at the default 15-minute TTL while disconnected, + then a late client reports a real terminal result and `stat` displays both + expiry history and the updated terminal state; +- server per-command, per-client, and global quotas cross high and low + watermarks while assigned client events remain pinned; +- terminal age reclamation and command-level quota eviction create tombstones, + retention metadata, and audit entries without removing active work; +- client identity collision, exact-instance takeover expiry/consumption, and + an old connection attempting output after the new generation is committed; +- storage corruption found during asynchronous recovery, safe repair, explicit + acknowledgement of unavoidable loss, and continued service in healthy scopes; +- heartbeat control traffic remains serviceable while both data lanes are full + over a slow, intermittently writable WebSocket. + +**Exit criteria:** Compose tests demonstrate a complete background command, +foreground follow, stdin interaction, signal, reconnect, and restart recovery +through the nginx WebSocket path. + +## 9. Server control plane and `rvc` (Phase 6) + +### 9.1 gRPC over Unix socket + +Start a Unix socket only after removing an existing socket only when it is +proven to be a stale socket owned by this server account; do not blindly unlink +arbitrary paths. Set mode `0600` immediately. Implement the generated `Control` +service against domain/store/session interfaces, never directly against socket +state. + +Control response messages represent success only and contain no embedded error +alternative. Return canonical non-OK gRPC statuses with structured RVBox +details; the JSON-RPC adapter maps the same domain error rather than inspecting +response payloads. Agent-protocol rejection envelopes continue using +`ControlError`. + +Implement and test: + +- `ListClients`, `GetClient`, `ListCommands`, `GetCommand`; +- `AuthorizeClientTakeover` for an exact displayed pending instance, with a + one-shot 5-minute grant and normal mutation idempotency; +- `RunCommand` for durable creation and immediate UUID return, followed by + `FollowCommand` for foreground ordered history plus live events; +- `FollowCommand` beginning at a supplied event sequence and returning a wrapper + around either a client event or non-sequenced server retention metadata; +- `AppendStdin`, `CloseStdin`, and `SignalCommand` with persisted idempotency; +- `GetOutput` with stream selection, byte-bounded opaque cursor pagination that + can resume within an event, distinct client-observed/server-receipt times, and + truncation/incomplete metadata; +- `ListStorageIncidents`, `RepairStorageIncident`, and + `AcknowledgeStorageIncident`; dirty health is derived from unresolved rows, + safe repair resolves automatically, and acknowledging loss preserves history. + +Foreground context cancellation/timeouts stop only the local RPC stream. They +must never send a remote kill. The server uses domain/store authorization of +state transitions and an explicit command UUID for every mutation. + +Foreground `rvc run` first completes idempotent unary `RunCommand`, then starts +`FollowCommand` from sequence zero with existing history included. Reconnects +resume after the last received client sequence; retention markers do not advance +that cursor and are emitted before a relevant terminal event. Do not add a +combined create-and-follow RPC whose stream can fail before returning the newly +created UUID. + +For every mutation, canonicalize the validated protobuf request excluding +transport metadata, compute its immutable hash, and transact an idempotency row +keyed globally by `request_id`, with method included in the hashed/stored +identity. An exact replay returns the stored domain result +or stored error; a different hash returns `ALREADY_EXISTS`/request-conflict and +performs no work. Commit the idempotency row atomically with the mutation, retain +it at least as long as the referenced command/takeover/incident record, and +count its compressed storage against the owning command or audit scope. Reads +accept the field at the transport edge but deliberately drop it before domain +execution. + +Define opaque output/page tokens as versioned, authenticated encodings of query +filters plus stable position (command, event sequence, byte offset, snapshot +boundary). Reject a token reused with changed filters. Pagination reads a +consistent upper boundary so concurrent appends do not duplicate/skip retained +bytes; if retention removes the resume point, return the next retained position +with explicit server-retention metadata. + +### 9.2 CLI + +Implement `rvc` with no direct server database access. Required commands are: + +- `rvc stat [client-id] [issue-uuid] [--all --page --per-page]`; +- `rvc run [--background] [--cwd DIR] [--shell TYPE] [--env K=V]... + [--profile FLAG]... [--request-id UUID] [--queue-ttl DURATION] CLIENT COMMAND`; +- `rvc run --script PATH [same execution options] CLIENT`; +- `rvc append CLIENT UUID TEXT`, `--file PATH`, raw/no-newline mode, and + `--attach` stdin streaming; +- `rvc close-stdin CLIENT UUID`; +- `rvc kill [HUP|INT|TERM|KILL|USR1|USR2] CLIENT UUID` (optional `SIG` prefix; + default `TERM`); +- `rvc storage incidents`, `rvc storage repair INCIDENT`, and + `rvc storage acknowledge INCIDENT --note TEXT`; +- `rvc client takeover CLIENT INSTANCE-ID`; +- output selection, `--timestamped`, pagination, and `--follow` semantics. + +Accept `--request-id` globally. Generate a UUIDv7 for every control mutation +unless supplied, reuse it across transport retries, and discard it for +read-only operations. + +Create the request ID before dialing the Unix socket and retain it for all +automatic retries of that invocation. Validate user-supplied IDs as canonical +UUIDv7 strings. For `run`, map the same value directly to `issue_uuid`; for +other mutations keep it only as `request_id`. Do not generate a second +operation identifier. Print the issue UUID immediately after durable creation, +including before entering foreground follow mode. + +Make `stat` render lifecycle and history independently: an expired queued +command is visibly marked `expired` with its expiry time/reason; if a late, +previously accepted client result later arrives, the actual terminal lifecycle +becomes current while the expiry contradiction remains in metadata/audit. Also +show dirty storage scope/incident ID and pending takeover instance, +output retained range, and whether terminal output is incomplete. + +Render output without inventing line boundaries in stored data. Historical +pagination uses opaque cursors and byte limits; the display layer may buffer +partial lines for presentation and clearly prints retained range/truncation +notices. Map structured control errors to stable non-zero exit +codes while retaining machine-readable JSON output as a later optional CLI mode. + +### 9.3 JSON-RPC adapter + +Keep this adapter small and disabled by default. Use standard JSON-RPC 2.0 over +HTTP and protobuf JSON mapping for unary control request/response bodies. +Implement the documented lower-camel method names only, including storage +incident list/repair/acknowledge and `authorizeClientTakeover`. Bind default +`127.0.0.1:6900`; configuration may bind elsewhere but startup logs a conspicuous +unauthenticated-exposure warning. Do not add streaming or a second event model: +callers poll `getOutput`/`getCommand` by cursor. + +Audit every control request with transport/source metadata where available, +action, target, result, and error. Audit logs include environment key names but +not override values or raw stdin/stdout/stderr. + +Use the JSON-RPC envelope `id` only for JSON-RPC response correlation. Accept +the same optional protobuf `requestId` member as gRPC for domain idempotency; +when it is absent, generate a server UUIDv7. Consequently raw JSON-RPC remains +easy to use but only callers that persist/reuse `requestId` receive retry +idempotency. Map parse/invalid-request/method-not-found to standard JSON-RPC +codes and domain failures to one stable server-error code carrying the same +structured RVBox detail as gRPC. + +**Exit criteria:** `rvc` can drive each documented example against the Compose +stack; the Unix socket has mode `0600`; JSON-RPC behavior matches gRPC unary +semantics and is off unless explicitly enabled. + +## 10. Windows client implementation (Phase 7) + +Keep Windows code in platform-specific files/build tags so Linux builds never +import Windows APIs. Implement the same supervisor interface and event/spool +runtime; only OS execution/diagnostics/resource enforcement differ. + +1. Create a non-inheritable per-command Job Object, enable + `JOB_OBJECT_LIMIT_KILL_ON_JOB_CLOSE`, and do not enable breakaway. +2. Start an RVBox launcher suspended with `CREATE_NEW_CONSOLE`, + `CREATE_UNICODE_ENVIRONMENT`, and `EXTENDED_STARTUPINFO_PRESENT`. Assign it + to the Job at creation through `PROC_THREAD_ATTRIBUTE_JOB_LIST`; inherit only + an explicit standard-I/O/control handle list and never the Job handle. +3. Persist and flush `launch_prepared` with launcher PID, `GetProcessTimes` + creation `FILETIME`, and launch generation; persist and flush + `launch_authorized` before `ResumeThread`. The launcher then creates the + requested shell suspended in its console with `CREATE_NEW_PROCESS_GROUP`, + reports its PID/group, connects the allowlisted pipes, and resumes it. A + failure after authorization terminates the Job and becomes interrupted. +4. Keep the launcher as an in-console signal proxy. This separation is + mandatory: Windows ignores `CREATE_NEW_PROCESS_GROUP` when combined with + `CREATE_NEW_CONSOLE`, and `GenerateConsoleCtrlEvent` reaches only groups + sharing the caller's console. On daemon crash, closing the sole Job handle + terminates the launcher and complete tree; restart never signals by PID alone. +5. Expose Job Object accounting snapshots. Children normally remain in the Job. +6. Accept only `SIGTERM` and `SIGKILL`. For TERM, have the launcher call + `GenerateConsoleCtrlEvent(CTRL_BREAK_EVENT, shell_group_id)`, wait 10 + seconds, then terminate the Job. For KILL, terminate the Job immediately. + Report graceful-attempt/escalation outcome in the event. +7. Apply requested profile limits through Job Object limits. If the requested + control cannot be applied, reject the dispatch with structured `UNSUPPORTED`. +8. Use available process/Job Object telemetry for CPU/RSS/I/O; do not emit a + Linux-style wait reason. + +Implement the Windows launcher as a small mode of the same RVBox binary, not a +searchable external helper. Pass it an inherited, random per-launch control +pipe and fixed-size launch-generation token. Apply an explicit +`PROC_THREAD_ATTRIBUTE_HANDLE_LIST` so only stdin/stdout/stderr and that control +pipe cross creation; make every database, log, listener, Job, and unrelated +pipe handle non-inheritable. Frame ready/release/exec/error messages with length, +type, generation, and checksum, and reject any mismatched generation. + +Open the configured shell executable by canonical absolute path and pass it as +the explicit `lpApplicationName`. Produce the command line with one reviewed +Windows argument-quoting routine and construct an explicit Unicode environment +block. Do not invoke `%COMSPEC%`, search `PATH`/file associations, or let command +environment overrides influence executable selection. + +Associate the Job with an I/O completion port and use +`JOB_OBJECT_MSG_ACTIVE_PROCESS_ZERO` plus pipe EOF for tree/drain completion. +Persist the launcher and shell creation `FILETIME` identities before trusting +any PID. On recovery, reopen a process only to compare creation time and +generation evidence; if ownership cannot be proven, record a dirty incident +rather than terminating an unrelated reused PID. + +Fault-inject daemon/launcher death around Job creation, attribute-list process +creation, prepared/authorized fsync, resume, requested-shell creation, and +terminal drain. Test both supported shells, paths with spaces/non-ASCII, +malicious `PATH`/`COMSPEC`, nested descendants, CTRL_BREAK refusal/escalation, +Job limit rejection, and inherited-handle leaks on supported Windows versions. + +Run build checks on every platform and dedicated Windows integration tests for +shell selection, process-tree kill, forced termination, profile rejection, +reconnect, and output/spool behavior. No Windows release is supported until +these tests run on Windows CI. + +## 11. Reliability, observability, and operational delivery (Phase 8) + +### 11.1 Health, metrics, and logs + +Expose separate liveness/readiness for the server. Liveness and incident +inspection start before asynchronous recovery. Global readiness requires all +required scopes recovered and writable, while scoped operations can proceed on +healthy scopes; no health state requires a client connection. Client health +reports reconnect state, spool health, unresolved incidents, and supervisor +health without leaking command output. + +Publish counters/histograms/gauges for registrations/takeovers, stale messages, +heartbeat timeouts, reconnect duration, dispatch latency, command transitions, +queue depth, spool bytes, output compression/rotation/loss, segment eviction, +SQLite write latency/failure, script verification failure, and protocol errors. +Use stable labels with bounded cardinality—never UUID/client ID as a metric +label. Use structured logs and audit records for those identifiers instead. + +### 11.2 Deployment assets + +Provide: + +- nginx configuration showing WebSocket upgrade proxying and TLS termination; +- systemd units for server/client with private state directories, restart + policy, working directory, file descriptor limits, and least privilege; +- example server/client configuration files with every default and an explicit + JSON-RPC exposure warning; +- a backup/restore procedure for SQLite plus output/audit segment directories; +- an upgrade procedure that stops dispatch safely, snapshots data, migrates, + and verifies recovery; +- a troubleshooting runbook for no client, stale session, spool full, output + truncation, storage full, and daemon restart. + +Install the annotated TOML examples from `docs/examples/` and keep the most +operationally important identity/listener/state/TLS/quota keys first in each +section. Add `--check-config` to both daemons: it must run the production strict +decoder, defaulting, cross-field/profile/path/shell validation, print a redacted +normalized summary, and exit without opening stores/listeners or changing +state. CI parses both examples with this path on Linux; Windows CI additionally +validates the documented Windows path/shell variant. + +Use `Delegate=yes` in the Linux client systemd unit when cgroup supervision is +enabled and create a private writable cgroup subtree for the service. Refuse a +configured mandatory resource profile when delegation/controllers are absent, +while still allowing unrestricted execution through the documented fallback. +Package upgrades must retain the previous TOML, run `--check-config` before +restart, and never silently ignore a newly unknown/removed key. + +### 11.3 Security and regression review + +Before v1 release, review path traversal, script temp-file permissions, command +logging, environment override redaction, malformed compression, decompression +bombs, oversized frames, Unicode/ASCII ID validation, Unix socket ownership, +JSON-RPC external bind warnings, SQL injection (all parameterized), segment +record corruption, and process-group PID reuse. Add fuzz tests for envelope +decode, compressed output validation, segment-tail recovery, pagination tokens, +and JSON-RPC parsing. + +**Exit criteria:** release Compose/systemd/nginx assets exist; load/fault tests +show no unbounded memory/goroutine growth; the operations runbook reproduces +recovery and retention behavior. + +## 12. Release gates and implementation order + +The recommended merge order is deliberately vertical: + +1. Phase 0: Docker-first repository/toolchain/proto generation. +2. Phase 1: domain/config validation and state machine. +3. Phase 2: SQLite/segments/audit/retention with recovery tests. +4. Phase 3: server registration, fencing, heartbeat, and persisted dispatch. +5. Phase 4: Linux client spool/supervisor and at-most-once execution. +6. Phase 5: full WSS protocol, fault injection, and nginx end-to-end tests. +7. Phase 6: gRPC Unix socket, `rvc`, and optional JSON-RPC. +8. Phase 7: Windows supervisor and Windows CI. +9. Phase 8: operational assets, stress/fuzz/recovery testing, release review. + +Do not merge a later vertical slice by stubbing a durability/safety invariant. +For example: foreground mode may wait on a durable background command, but must +not bypass persistence; client output may be truncated under the documented +caps, but must never block a child pipe; and a reconnection may replay work, +but may never re-execute an already accepted UUID. + +The v1 release is ready only after all release-gate tests pass on clean Docker +environments, Linux server/client end-to-end behavior matches the design docs, +Windows support is either fully tested or explicitly not shipped, and every +accepted limitation (self-reported identity and unauthenticated optional +JSON-RPC) is conspicuous in deployment documentation. diff --git a/docs/platform-and-operations.md b/docs/platform-and-operations.md index 0f2011d..42529af 100644 --- a/docs/platform-and-operations.md +++ b/docs/platform-and-operations.md @@ -1,5 +1,9 @@ # RVBox v1 platform and operations contract +Daemon configuration uses strict TOML as specified in +[`configuration.md`](configuration.md); the annotated examples contain every +v1 knob and default. + ## Unix-like clients The client starts `sh` or `bash` in a new session/process group. Unix signals @@ -8,20 +12,99 @@ orderly shutdown, or recovery after an unclean daemon failure, managed command groups are terminated and marked interrupted because pipe capture cannot be safely resumed. +Both command text and uploaded scripts execute from generated private files +beneath the effective CWD using exactly the selected executable (`sh FILE` or +`bash FILE`). No user-supplied filename becomes a filesystem path. The wrapper +file is removed during terminal cleanup. + +Root-process exit begins a configurable 5-second drain grace period. RVBox waits +for the supervised tree and capture pipes, then terminates residual group/cgroup +members, drains to EOF, and only afterward emits the terminal lifecycle event. +If capture still cannot reach EOF, it closes the handles and emits explicit +incomplete-output metadata first. Shell-level detachment is not a supported way +to leave descendants running; callers use RVBox background mode instead. + +Launch uses an internal blocked launcher rather than starting requested command +code directly. The launcher establishes its session/process group, reports its +identity, and waits on a private release/watchdog channel. The client durably +records `launch_prepared`, then durably records `launch_authorized`, and only +then sends the release token. The launcher creates the requested shell inside +that group and remains as a non-user-code watchdog until the tree exits. The +daemon keeps the channel open for that lifetime: EOF before authorization exits +without execution, while EOF after release terminates the group. Once +`launch_authorized` is durable, recovery never retries that UUID; an uncertain +launch is marked interrupted. + +On Linux, create a per-command cgroup v2 for supervision even when no resource +profile was requested, whenever the daemon has a delegated writable cgroup. +Put the blocked launcher into that cgroup before release; use +`clone3(CLONE_INTO_CGROUP | CLONE_PIDFD)` where available, otherwise migrate the +still-blocked launcher through `cgroup.procs`. Persist the cgroup path, PID, +process group, `/proc//stat` start time, and launch generation. A live +daemon uses the pidfd where available. Recovery uses `cgroup.kill` as the primary +tree-cleanup operation and verifies the recorded birth identity before any +PID/process-group fallback. Without cgroup delegation it uses the generic +watchdog/process-group fallback unless a requested profile requires cgroup +controls, in which case acceptance fails as unsupported. It never signals a +process based only on a persisted numeric PID or PGID. + +Other Unix-like systems use the same launch barrier plus a watchdog control +channel whose EOF triggers process-group termination. Recovery validates the +platform's process-birth identity before signaling. Descendants that deliberately +create a new session may escape this generic fallback, so complete tree cleanup +outside Linux cgroup supervision is best-effort; the at-most-once launch +guarantee still applies. + Linux diagnostics sample `/proc/` and relevant children for state, CPU, resident memory, I/O counters, CWD, and wait-channel information when readable. These values may be unavailable due to permissions, kernel configuration, or a short-lived process; absence is represented explicitly rather than fabricated. -Cgroup v2 is used for requested resource profiles only when available. +Cgroup v2 profile limits are applied only when requested; a no-profile +supervisory cgroup imposes no resource limit. ## Windows clients -The client launches `cmd` or `powershell` in an appropriate dedicated console -process group and assigns the root process to a per-command Job Object. Child -processes normally join the Job Object. Job Object limits enforce requested -profiles and `KILL_ON_JOB_CLOSE` protects against lost supervision. +The minimum supported v1 Windows versions are Windows 10 and Windows Server +2016. For every command, create a non-inheritable Job Object, set +`JOB_OBJECT_LIMIT_KILL_ON_JOB_CLOSE`, and do not enable breakaway. The daemon +starts an RVBox per-command launcher suspended with `CREATE_NEW_CONSOLE`, +`CREATE_UNICODE_ENVIRONMENT`, and `EXTENDED_STARTUPINFO_PRESENT`; it assigns the +launcher atomically through `PROC_THREAD_ATTRIBUTE_JOB_LIST`. Only an explicit +standard-I/O and launcher-control handle list is inherited, and the Job handle +is never inherited. -Only `SIGTERM` and `SIGKILL` are accepted. `SIGTERM` attempts `CTRL_BREAK_EVENT` +The launcher invokes exactly the selected shell against the generated wrapper: +`cmd.exe /D /S /C` for a `.cmd` wrapper, or `powershell.exe` with `-NoLogo`, +`-NoProfile`, `-NonInteractive`, and `-File` for a `.ps1` wrapper. Application +paths and argument quoting are constructed by the Windows launcher, never by +concatenating an untrusted command line. There is no fallback between shells. + +Persist and flush `launch_prepared` with the launcher PID, +`GetProcessTimes` creation `FILETIME`, and launch generation. Persist and flush +`launch_authorized` before calling `ResumeThread`. The launcher then starts the +requested `cmd` or `powershell` suspended in its console with +`CREATE_NEW_PROCESS_GROUP`, reports the shell PID/group through the private +control channel, connects the allowlisted pipes, and resumes it. This two-step +shape is required because `CREATE_NEW_PROCESS_GROUP` is ignored when combined +with `CREATE_NEW_CONSOLE`, and console control events reach only groups sharing +the caller's console. Failure after authorization terminates the Job and is +reported as interrupted; it never redispatches the UUID. + +The launcher remains the in-console signal proxy and calls +`GenerateConsoleCtrlEvent(CTRL_BREAK_EVENT, shell_group_id)` on request. The +daemon retains the sole Job handle, so an unclean daemon exit closes the last +handle and terminates the launcher, shell, and descendants. Recovery never +kills by persisted PID alone; the PID/creation-time tuple is diagnostic evidence +for PID reuse or cleanup anomalies. Child processes normally join the Job. +Job Object limits enforce requested profiles and `KILL_ON_JOB_CLOSE` protects +against lost supervision. + +Root-process exit begins the same drain grace period. Completion waits for the +Job Object to reach zero active processes; after the grace period RVBox +terminates the Job, drains its capture handles, records any incomplete-output +marker, and emits the terminal lifecycle event last. + +Only `TERM`/`SIGTERM` and `KILL`/`SIGKILL` are accepted. `TERM` attempts `CTRL_BREAK_EVENT` and waits 10 seconds, then calls Job Object termination if the job persists; `SIGKILL` calls Job Object termination immediately. A console signal is best-effort, so callers receive an explicit escalation result. Windows status @@ -30,12 +113,36 @@ diagnostics such as an I/O wait channel. ## Storage and recovery -SQLite runs in WAL mode with integrity checking on startup. Output segments are -written atomically, fsynced according to the configured durability interval, and -indexed only after successful durable append. Startup scans/repairs incomplete -tail records before accepting control requests. Segment compression is Zstandard; -limits always measure stored compressed bytes, while clients expose raw byte -counts separately. +SQLite runs in WAL mode. Every append-only segment has a SQLite-owned +`committed_end_offset`. The writer validates and appends records, syncs the file +(grouped by the configured durability interval), and only then commits event +metadata plus the new offset in SQLite. An acknowledgement waits for both +steps. Therefore a crash can leave an uncommitted file tail, but cannot validly +acknowledge metadata whose bytes were not durable. + +Startup acquires the instance lock and binds liveness/diagnostic endpoints, then +runs integrity and segment recovery asynchronously. A file longer than its +committed offset is safely truncated to that offset. A file shorter than the +offset, a checksum failure inside the committed range, or corrupt essential +metadata creates a durable scoped storage incident; affected output is marked +truncated/incomplete and affected active commands are interrupted when their +essential state cannot be trusted. Healthy scopes remain usable. Readiness is +false and mutations requiring an unrecovered or dirty scope return `UNAVAILABLE`, +but process startup, liveness, incident inspection, and unaffected work do not +wait for a full-store scan. + +Safe repairs are attempted automatically and can also be requested online with +`rvc storage repair`. Irrecoverable loss stays dirty until explicitly accepted +with `rvc storage acknowledge`; an offline server has equivalent +`rvbox-server repair --data-dir ...` repair/list/acknowledge operations. Client +spool recovery follows the same committed-offset rule and exposes equivalent +offline `rvbox repair --state-dir ...` operations and local health diagnostics. +Resolving an incident clears derived dirty health but never erases the incident +or audit history as part of resolution. Unresolved compact incident records are +non-evictable; resolved incident/audit history follows the separate 100 MiB +rotation. Segment compression is Zstandard; +limits measure stored compressed bytes, while raw byte counts are reported +separately. The system must reserve headroom before writes and use transactional metadata updates. Storage-full, permission, and corruption failures are surfaced as @@ -43,6 +150,12 @@ structured server/client health states and audit events. They must isolate the affected command/session, reject work when needed, and keep the daemon's heartbeat/control loops alive. +V1 state is plaintext at rest, including command text, scripts, environment +override values, stdin, and output. Private directory/file modes and dedicated +daemon accounts are deployment hygiene, not an application-level encryption +guarantee. Backups copy the same plaintext sensitivity. Encryption and external +key management are future-version work. + ## Metrics, logging, and safe defaults Both daemons should emit structured logs and metrics for session transitions, @@ -52,10 +165,26 @@ and protocol violations. Never emit stdin or raw output in normal daemon logs. Recommended configuration defaults are: 10-second heartbeat idle period, 30-second liveness timeout, 1–60-second full-jitter reconnect backoff, -60-second stable-session reset, 16 running/100 queued commands per client, -10 MiB per-command compressed window, 50 MiB per-client active spool and server -history, 1 GiB server history, 64 KiB uncompressed stream chunk, 1 MiB decoded -envelope, and 10 MiB script maximum. +60-second stable-session reset, 5-minute one-shot live-conflict takeover grant, +16 running/100 queued commands per client, 1,000 server-queued commands per +target client and 10,000 server-wide, +15-minute queue TTL, 10 MiB per-command output window, 32 MiB total per command, +256 MiB per client/client daemon, 4 GiB server-wide command storage, 30-day +terminal retention, 100 MiB audit storage with quota-only rotation by default, +one million compact command tombstones, 64 KiB uncompressed stream chunk, 1 MiB +decoded agent envelope, +1 MiB/256 KiB per-command raw-output high/low watermarks, 8 MiB/4 MiB per-client +watermarks, 64 MiB/32 MiB server-wide watermarks, 1 MiB per-command and 8 MiB +per-session unacknowledged send windows, and 10 MiB raw script maximum. Control +gRPC accepts at most 16 MiB decoded requests; the JSON-RPC adapter accepts at +most 24 MiB HTTP bodies to allow protobuf JSON's base64 expansion while +retaining the same decoded field limits. Each active command reserves 64 KiB +within its quota for terminal/loss closeout metadata; protocol detail/reason and +incident-note text fields are individually limited to 4 KiB. Default emergency +filesystem free-space floors are 256 MiB on the server and 64 MiB on a client; +crossing one rejects new unreserved allocations even if the logical quota has +headroom. Already-reserved terminal/loss closeout remains writable while bytes +physically remain. These bounds protect RVBox's own loops; they cannot make arbitrary child commands harmless when no resource profile is requested. Operators should diff --git a/docs/protocol.md b/docs/protocol.md index 09a3a3a..d3169f0 100644 --- a/docs/protocol.md +++ b/docs/protocol.md @@ -14,8 +14,10 @@ sizes before allocation/decompression. The protobuf package is `rvbox.v1`. Registration negotiates a major/minor protocol range: incompatible majors are rejected; the highest shared minor is -chosen. New fields are append-only. A peer ignores unknown optional fields but -must respond with `PROTOCOL_ERROR` to an unknown required envelope feature. +chosen. New fields are append-only. A peer ignores unknown optional fields. A +new required behavior requires a negotiated minor-version change; an envelope +with no recognized payload is a `PROTOCOL_ERROR` rather than an implicit +“required unknown field” mechanism that protobuf cannot represent. ## Heartbeat and reconnect @@ -33,11 +35,19 @@ reconciles only commands the server still considers non-terminal. ## Session fencing -`ClientHello` starts registration. `ServerWelcome` gives the selected version, -random `session_id`, and `session_generation`. Except `ClientHello`, all -envelopes carry those values. A new accepted registration fences and disconnects -the prior one for the same client ID. The server accepts messages only from the -current generation; command dispatches also identify their intended generation. +`ClientHello` starts registration and includes a random `client_instance_id` +generated once in the client state directory and retained across daemon restarts +and reconnects. `ServerWelcome` gives the selected version, random `session_id`, +and `session_generation`. Except `ClientHello`, all envelopes carry those +values. A new accepted registration from the same instance fences and +disconnects its prior session. A different instance claiming the same live +client ID is rejected and recorded as the current pending claim unless an +operator explicitly authorized that exact instance. The one-shot authorization +expires after 5 minutes by default and is consumed by the matching reconnect. +When no session for that client ID is live, a new instance is accepted normally. +This is collision protection, not peer authentication. The +server accepts messages only from the current generation; command dispatches +also identify their intended generation. ## Reliable work flows @@ -45,53 +55,113 @@ current generation; command dispatches also identify their intended generation. The server persistently creates a command before dispatching `CommandDispatch`. It redelivers until it receives `CommandAccepted`. Clients durably deduplicate -on `issue_uuid`. A client whose queue is full sends a capacity rejection. +on `issue_uuid`. A client whose queue is full sends a transient capacity +rejection, which the server requeues with backoff. A permanent validation or +unsupported-platform rejection makes the server command terminal `rejected`; +it is not retried or mislabeled as a launched-process failure. +An exact UUID/request-hash hit in the compact tombstone ledger returns +`CODE_ALREADY_EXECUTED`; the server suppresses dispatch and reconciles its stale +state instead of representing the prior execution as a new rejection. -Following registration the server sends `ReconcileRequest` only for its -non-terminal commands for that client. The client replies with a -`ReconcileSnapshot` per requested known command, or `unknown_to_client`. It -continues normal event retransmission from the server's last acknowledged event -sequence. Terminal history already confirmed by the server is deliberately -excluded. +Following registration the server sends `ReconcileRequest` containing its +non-terminal commands for that client. The client replies once with a complete, +bounded `ReconcileSnapshot` of every command it still retains, including +terminal-but-unacknowledged records and matching requested tombstones, then +continues normal retransmission from the server's last acknowledged event +sequence. The server does not dispatch new work until this snapshot is complete. +For healthy client storage, absence means the client never durably accepted the +UUID: server `queued`/`dispatched` work returns to `queued`, while absence of an +`accepted`/`running` record is an invariant failure and becomes interrupted. +A matching UUID/request-hash tombstone always suppresses replay. A client-known +server-missing active command, or a contradiction with server-confirmed +terminal history, is terminated locally after the server returns a durable +`ReconcileResult` and creates a recovery incident rather than inventing server +state. That result also identifies client-retained terminal records the server +has already stored or deliberately tombstoned, allowing the client to discard +them even when its last `EventAck` was lost. The result is idempotent and is +written before any new dispatch on that session. A client +with an unresolved essential-store incident must not claim a complete snapshot +or accept work until repaired/acknowledged. ### Output and events -Client execution events use an increasing `event_seq`; retries reuse the same -sequence and content. The server durably writes an event before sending -`EventAck`. `EventAck` is cumulative through a sequence number. Output is -Zstandard compressed with an explicit original-size field. Server storage can -reuse the validated compressed bytes. +Client execution entries first have a durable local order. They receive an +increasing wire `event_seq` durably when admitted to the bounded send window; +once assigned, the sequence/content is pinned until acknowledged and retries +reuse it exactly. The server durably writes an event before sending `EventAck`. +`EventAck` is cumulative through a sequence number. Output is Zstandard +compressed with an explicit original-size field. Server storage can reuse the +validated compressed bytes. When rolling output removes old retained segments, the server writes `OutputTruncation` metadata containing the removed event/byte ranges. Queries must show that marker rather than silently presenting an apparently complete -stream. An offline client that reaches a cap sends `ClientOutputTruncated` -before replaying its retained tail on reconnect. The server records it, permits -the named event-sequence gap, and exposes it in output queries. Truncation -metadata is not an execution event and does not consume an `event_seq`. +stream. Assigned but unacknowledged events occupy the pinned 1 MiB per-command +send window and are not evicted. An offline client that reaches a cap replaces +one or more still-unsequenced output runs in its durable local order with +`OutputTruncation`; when admitted to the send window the marker receives the +next normal `event_seq`. Its event-range fields are absent because the discarded +bytes never had wire sequences. The normal cumulative `EventAck` acknowledges +the marker. Server-created retention markers include their removed event range +and remain query metadata because the server cannot allocate client sequences. + +If a capture pipe cannot be drained to a provably complete EOF, the client emits +a sequenced `OutputIncomplete` event before the terminal lifecycle event. It +does not invent a missing sequence or byte count for bytes it never observed. ### Stdin and signals `StdinWrite` is binary-safe and ordered by `write_seq`; the client durably -deduplicates it and returns `StdinAck`. `append_newline` is true by default in -the CLI but explicit on the wire. `CloseStdin` is a separate idempotent action. -Signal and cancellation requests carry a command revision to settle start/kill -races. +deduplicates it and returns a sequenced stdin-acknowledgement command event. +`append_newline` is true by default in the CLI but explicit on the wire. +`CloseStdin` is a separate idempotent action. Signal and cancellation requests +carry a command revision to settle start/kill races; the resulting lifecycle or +signal-result event echoes that revision. Control-plane `request_id` values stay +at the server and map to the assigned revision rather than crossing the agent +protocol. ### Scripts For a script command, dispatch first contains a `ScriptDescriptor`; then the server sends `ScriptChunk` messages and a commit. The client checks offset, chunk order, full length, and SHA-256 before it reports upload complete or -launches the process. A retransmitted chunk is idempotent by offset/content. +launches the process. `ScriptUploadStatus.received_bytes` is a cumulative +durable progress acknowledgement: the server sends at most the command's +unacknowledged window, waits for progress, and resumes from that offset. A +retransmitted chunk is idempotent by offset/content. + +Acceptance is durable queue admission, not proof that every later preparation +step will succeed. A permanent script checksum/write error or failure to prepare +the requested executable emits terminal `rejected` before any launch +authorization. User cancellation in that interval emits `cancelled`; an +uncertain crash window emits `interrupted`. None is mislabeled as process +`failed`. ## Flow control and failure containment No receive loop runs an executor, database write, decompressor, or slow socket operation inline. Each side has bounded staging queues. Durable spools are the -source of truth and are charged to the 10 MiB/50 MiB/1 GiB compressed retention -budgets described in the architecture document. A full staging queue pauses the -related read/dispatch path and drains from disk; it never grows without bound. +source of truth and are charged to the tiered unified command-storage budgets +described in the architecture document. Senders keep a bounded unacknowledged +window of encoded wire bytes (defaults: 1 MiB per command and 8 MiB per client +session); unsent data +waits durably. WebSocket Ping/Pong/Close plus fencing and protocol errors use a +reserved priority lane and a dedicated writer with bounded data-frame size and +write deadlines. A full essential ingress queue does not stall the sole +WebSocket reader indefinitely: +the receiver closes without acknowledgement and the durable sender retries +after jittered reconnect. Droppable output uses the sequenced overload-loss +path instead. + +Before compression, raw output uses the architecture's high/low watermarks. +Once a client high watermark is crossed, still-unsequenced droppable chunks are +summarized in durable local order instead of consuming unbounded compression +work. A server at its ingress high watermark does not acknowledge or discard an +already sequenced event; it closes without acknowledgement if its bounded queue +cannot admit the frame, and the client retries from durable spool. Fair +scheduling prevents one verbose command from monopolizing workers. Reserved +metadata capacity remains +available to close each gap and emit the final lifecycle event. Malformed protobuf, over-size payload, invalid compressed data, impossible sequence, bad session token, or protocol-version violation yields a structured diff --git a/protos/rvbox/v1/agent.proto b/protos/rvbox/v1/agent.proto index 6868513..ad764db 100644 --- a/protos/rvbox/v1/agent.proto +++ b/protos/rvbox/v1/agent.proto @@ -2,35 +2,34 @@ syntax = "proto3"; package rvbox.v1; -option go_package = "github.com/rvbox/rvbox/gen/go/rvbox/v1;rvboxv1"; - import "google/protobuf/timestamp.proto"; import "rvbox/v1/common.proto"; +option go_package = "github.com/rvbox/rvbox/gen/go/rvbox/v1;rvboxv1"; + // One AgentEnvelope is carried in one binary WebSocket message. message AgentEnvelope { // Empty only for ClientHello. All other envelopes are fenced to a session. string session_id = 1; uint64 session_generation = 2; - string message_id = 3; oneof payload { - ClientHello client_hello = 10; - ServerWelcome server_welcome = 11; - CommandDispatch command_dispatch = 12; - CommandAccepted command_accepted = 13; - CommandEvent command_event = 14; - EventAck event_ack = 15; - StdinWrite stdin_write = 16; - CloseStdin close_stdin = 17; - SignalCommand signal_command = 18; - ScriptChunk script_chunk = 19; - ScriptCommit script_commit = 20; - ClientCapacity client_capacity = 21; - ReconcileRequest reconcile_request = 22; - ReconcileSnapshot reconcile_snapshot = 23; - AgentError error = 24; - ClientOutputTruncated client_output_truncated = 25; + ClientHello client_hello = 3; + ServerWelcome server_welcome = 4; + CommandDispatch command_dispatch = 5; + CommandAccepted command_accepted = 6; + CommandEvent command_event = 7; + EventAck event_ack = 8; + StdinWrite stdin_write = 9; + CloseStdin close_stdin = 10; + SignalCommand signal_command = 11; + ScriptChunk script_chunk = 12; + ScriptCommit script_commit = 13; + ClientCapacity client_capacity = 14; + ReconcileRequest reconcile_request = 15; + ReconcileSnapshot reconcile_snapshot = 16; + AgentError error = 17; + ReconcileResult reconcile_result = 18; } } @@ -43,7 +42,8 @@ message ClientHello { string architecture = 5; string daemon_cwd = 6; repeated ShellType supported_shells = 7; - string reconnect_uuid = 8; + // Generated once and persisted in the client state directory. + string client_instance_id = 8; uint32 max_running_commands = 9; uint32 max_queued_commands = 10; google.protobuf.Timestamp sent_at = 11; @@ -61,6 +61,7 @@ message CommandDispatch { uint64 target_session_generation = 3; google.protobuf.Timestamp issue_time = 4; ExecutionSpec spec = 5; + google.protobuf.Timestamp queue_expiry_time = 6; } message CommandAccepted { @@ -122,23 +123,30 @@ message ReconcileTarget { string issue_uuid = 1; uint64 last_server_event_seq = 2; uint64 command_revision = 3; + bytes immutable_request_sha256 = 4; } -// Sent only for requests in ReconcileRequest; server-confirmed terminal history -// is intentionally not requested. message ReconcileSnapshot { - string issue_uuid = 1; - bool known_to_client = 2; - CommandLifecycle lifecycle = 3; - uint64 last_client_event_seq = 4; - uint64 command_revision = 5; + // The complete set still retained by the client, including non-terminal and + // terminal-but-unacknowledged commands. + repeated ReconcileCommandState retained_commands = 1; } -// Sent before retained replay data when an offline client had to rotate -// unacknowledged output to remain within its hard spool limits. -message ClientOutputTruncated { +message ReconcileCommandState { string issue_uuid = 1; - OutputTruncation truncation = 2; + CommandLifecycle lifecycle = 2; + uint64 last_client_event_seq = 3; + uint64 command_revision = 4; + // True only for a matching UUID/request hash in the compact tombstone ledger. + bool tombstoned = 5; + bytes immutable_request_sha256 = 6; +} + +// Sent after the server durably compares a complete snapshot and before it +// dispatches new work on the session. Repetition is idempotent. +message ReconcileResult { + repeated string terminate_local_issue_uuids = 1; + repeated string discard_local_terminal_issue_uuids = 2; } message AgentError { diff --git a/protos/rvbox/v1/common.proto b/protos/rvbox/v1/common.proto index bbbc3d8..7b1d548 100644 --- a/protos/rvbox/v1/common.proto +++ b/protos/rvbox/v1/common.proto @@ -2,11 +2,11 @@ syntax = "proto3"; package rvbox.v1; -option go_package = "github.com/rvbox/rvbox/gen/go/rvbox/v1;rvboxv1"; - import "google/protobuf/duration.proto"; import "google/protobuf/timestamp.proto"; +option go_package = "github.com/rvbox/rvbox/gen/go/rvbox/v1;rvboxv1"; + // An inclusive protocol-version range advertised during registration. message ProtocolRange { uint32 major = 1; @@ -52,6 +52,8 @@ enum CommandLifecycle { COMMAND_TERMINATED = 7; COMMAND_CANCELLED = 8; COMMAND_INTERRUPTED = 9; + COMMAND_EXPIRED = 10; + COMMAND_REJECTED = 11; } enum StreamKind { @@ -109,16 +111,23 @@ message CommandRecord { ExecutionSpec spec = 5; CommandLifecycle lifecycle = 6; uint64 last_event_seq = 7; - int32 exit_code = 8; + optional int32 exit_code = 8; google.protobuf.Timestamp terminal_time = 9; bool output_truncated = 10; uint64 retained_compressed_bytes = 11; + google.protobuf.Timestamp queue_expiry_time = 12; + bool output_incomplete = 13; + // Present when lifecycle is COMMAND_REJECTED. + ControlError rejection = 14; + uint64 command_revision = 15; + bool late_after_expiry = 16; } message LifecycleChange { CommandLifecycle lifecycle = 1; - int32 exit_code = 2; + optional int32 exit_code = 2; string detail = 3; + uint64 command_revision = 4; } // data is compressed according to compression. uncompressed_size is mandatory @@ -132,15 +141,16 @@ message OutputChunk { } message ResourceSnapshot { - uint64 resident_memory_bytes = 1; - uint64 virtual_memory_bytes = 2; + optional uint64 resident_memory_bytes = 1; + optional uint64 virtual_memory_bytes = 2; google.protobuf.Duration cpu_time = 3; - uint64 read_bytes = 4; - uint64 write_bytes = 5; - string process_state = 6; - string wait_reason = 7; + optional uint64 read_bytes = 4; + optional uint64 write_bytes = 5; + optional string process_state = 6; + optional string wait_reason = 7; bool suspected_hung = 8; string diagnostic_detail = 9; + optional string current_cwd = 10; } enum OutputTruncationSource { @@ -148,19 +158,31 @@ enum OutputTruncationSource { OUTPUT_TRUNCATION_SOURCE_CLIENT_SPOOL = 1; OUTPUT_TRUNCATION_SOURCE_SERVER_COMMAND_WINDOW = 2; OUTPUT_TRUNCATION_SOURCE_SERVER_CLIENT_CAP = 3; + OUTPUT_TRUNCATION_SOURCE_CLIENT_OVERLOAD = 4; + OUTPUT_TRUNCATION_SOURCE_SERVER_GLOBAL_CAP = 5; } -// Persistent query metadata for a missing contiguous event range. It is not a -// CommandEvent and therefore does not consume an event_seq. +// Metadata for missing output. Server-side retention supplies the optional +// event range. Client output discarded before wire sequence assignment omits +// the range and carries this as a normally sequenced CommandEvent. message OutputTruncation { - uint64 first_removed_event_seq = 1; - uint64 last_removed_event_seq = 2; - uint64 removed_compressed_bytes = 3; + optional uint64 first_removed_event_seq = 1; + optional uint64 last_removed_event_seq = 2; + // Absent when bytes were discarded before compression. + optional uint64 removed_compressed_bytes = 3; uint64 removed_uncompressed_bytes = 4; string reason = 5; OutputTruncationSource source = 6; } +// Indicates that capture ended without a provably complete byte stream. It is +// sequenced before the terminal lifecycle event but does not claim an invented +// event or byte range. +message OutputIncomplete { + repeated StreamKind streams = 1; + string reason = 2; +} + message StdinAcknowledgement { uint64 write_seq = 1; bool stdin_closed = 2; @@ -173,6 +195,7 @@ message SignalResult { bool graceful_delivery_attempted = 3; bool forced_termination_used = 4; string detail = 5; + uint64 command_revision = 6; } message ScriptUploadStatus { @@ -188,12 +211,14 @@ message CommandEvent { uint64 event_seq = 2; google.protobuf.Timestamp observed_at = 3; oneof payload { - LifecycleChange lifecycle = 10; - OutputChunk output = 11; - ResourceSnapshot resource = 12; - StdinAcknowledgement stdin_ack = 13; - SignalResult signal_result = 14; - ScriptUploadStatus script_status = 15; + LifecycleChange lifecycle = 4; + OutputChunk output = 5; + ResourceSnapshot resource = 6; + StdinAcknowledgement stdin_ack = 7; + SignalResult signal_result = 8; + ScriptUploadStatus script_status = 9; + OutputTruncation output_truncation = 10; + OutputIncomplete output_incomplete = 11; } } @@ -209,6 +234,7 @@ message ControlError { PROTOCOL_ERROR = 7; TRANSIENT = 8; INTERNAL = 9; + CODE_ALREADY_EXECUTED = 10; } Code code = 1; string message = 2; diff --git a/protos/rvbox/v1/control.proto b/protos/rvbox/v1/control.proto index b09a118..225cfb7 100644 --- a/protos/rvbox/v1/control.proto +++ b/protos/rvbox/v1/control.proto @@ -2,22 +2,27 @@ syntax = "proto3"; package rvbox.v1; -option go_package = "github.com/rvbox/rvbox/gen/go/rvbox/v1;rvboxv1"; - +import "google/protobuf/duration.proto"; +import "google/protobuf/timestamp.proto"; import "rvbox/v1/common.proto"; +option go_package = "github.com/rvbox/rvbox/gen/go/rvbox/v1;rvboxv1"; + service Control { rpc ListClients(ListClientsRequest) returns (ListClientsResponse); rpc GetClient(GetClientRequest) returns (GetClientResponse); rpc ListCommands(ListCommandsRequest) returns (ListCommandsResponse); rpc GetCommand(GetCommandRequest) returns (GetCommandResponse); rpc RunCommand(RunCommandRequest) returns (RunCommandResponse); - rpc RunCommandAndFollow(RunCommandRequest) returns (stream CommandEvent); - rpc FollowCommand(FollowCommandRequest) returns (stream CommandEvent); + rpc FollowCommand(FollowCommandRequest) returns (stream FollowCommandResponse); rpc AppendStdin(AppendStdinRequest) returns (AppendStdinResponse); rpc CloseStdin(CloseStdinRequest) returns (CloseStdinResponse); rpc SignalCommand(ControlSignalCommandRequest) returns (ControlSignalCommandResponse); rpc GetOutput(GetOutputRequest) returns (GetOutputResponse); + rpc ListStorageIncidents(ListStorageIncidentsRequest) returns (ListStorageIncidentsResponse); + rpc RepairStorageIncident(RepairStorageIncidentRequest) returns (RepairStorageIncidentResponse); + rpc AcknowledgeStorageIncident(AcknowledgeStorageIncidentRequest) returns (AcknowledgeStorageIncidentResponse); + rpc AuthorizeClientTakeover(AuthorizeClientTakeoverRequest) returns (AuthorizeClientTakeoverResponse); } message ClientSummary { @@ -32,6 +37,10 @@ message ClientSummary { string daemon_version = 9; string daemon_cwd = 10; repeated ShellType supported_shells = 11; + string client_instance_id = 12; + // Most recently rejected different instance while this client is live. + string pending_instance_id = 13; + google.protobuf.Timestamp pending_instance_seen_at = 14; } message ListClientsRequest { @@ -50,7 +59,6 @@ message GetClientRequest { message GetClientResponse { ClientSummary client = 1; - ControlError error = 2; } message ListCommandsRequest { @@ -63,7 +71,6 @@ message ListCommandsRequest { message ListCommandsResponse { repeated CommandRecord commands = 1; string next_page_token = 2; - ControlError error = 3; } message GetCommandRequest { @@ -74,7 +81,6 @@ message GetCommandRequest { message GetCommandResponse { CommandRecord command = 1; ResourceSnapshot latest_resource = 2; - ControlError error = 3; } // For script execution, script_content contains the bytes whose descriptor is @@ -83,12 +89,15 @@ message RunCommandRequest { string target_client_id = 1; ExecutionSpec spec = 2; bytes script_content = 3; + // Becomes issue_uuid when supplied; generated by the server when omitted. + string request_id = 4; + // Omitted uses the server default; zero explicitly requests no expiry. + google.protobuf.Duration queue_ttl = 5; } message RunCommandResponse { string issue_uuid = 1; CommandLifecycle lifecycle = 2; - ControlError error = 3; } message FollowCommandRequest { @@ -98,37 +107,47 @@ message FollowCommandRequest { bool include_existing = 4; } +message FollowCommandResponse { + oneof item { + CommandEvent event = 1; + // Server retention metadata; does not advance the client event cursor. + RetentionTruncation retention_truncation = 2; + } + // Set for a client event; server-created retention uses its own recorded_at. + google.protobuf.Timestamp server_receipt_time = 3; +} + message AppendStdinRequest { string client_id = 1; string issue_uuid = 2; bytes data = 3; bool append_newline = 4; + string request_id = 5; } message AppendStdinResponse { uint64 write_seq = 1; - ControlError error = 2; } message CloseStdinRequest { string client_id = 1; string issue_uuid = 2; + string request_id = 3; } message CloseStdinResponse { uint64 write_seq = 1; - ControlError error = 2; } message ControlSignalCommandRequest { string client_id = 1; string issue_uuid = 2; SignalKind signal = 3; + string request_id = 4; } message ControlSignalCommandResponse { uint64 command_revision = 1; - ControlError error = 2; } message GetOutputRequest { @@ -136,13 +155,91 @@ message GetOutputRequest { string issue_uuid = 2; repeated StreamKind streams = 3; uint64 after_event_seq = 4; - uint32 page_size = 5; + uint64 max_bytes = 5; + string page_token = 6; +} + +// An uncompressed slice of one persisted output event. Opaque pagination may +// resume within an event; event_byte_offset identifies that position. +message OutputSlice { + uint64 event_seq = 1; + google.protobuf.Timestamp observed_at = 2; + StreamKind stream = 3; + bytes data = 4; + uint64 event_byte_offset = 5; + bool end_of_event = 6; + google.protobuf.Timestamp server_receipt_time = 7; +} + +message RetentionTruncation { + OutputTruncation truncation = 1; + google.protobuf.Timestamp server_recorded_at = 2; } message GetOutputResponse { - repeated CommandEvent events = 1; + repeated OutputSlice output = 1; string next_page_token = 2; bool output_truncated = 3; - ControlError error = 4; - repeated OutputTruncation truncations = 5; + repeated RetentionTruncation truncations = 4; + OutputIncomplete incomplete = 5; +} + +enum StorageIncidentState { + STORAGE_INCIDENT_STATE_UNSPECIFIED = 0; + STORAGE_INCIDENT_STATE_OPEN = 1; + STORAGE_INCIDENT_STATE_REPAIRED = 2; + STORAGE_INCIDENT_STATE_ACKNOWLEDGED = 3; +} + +message StorageIncident { + string incident_id = 1; + google.protobuf.Timestamp detected_at = 2; + google.protobuf.Timestamp resolved_at = 3; + StorageIncidentState state = 4; + string scope = 5; + string client_id = 6; + string issue_uuid = 7; + string summary = 8; + bool data_loss = 9; + bool automatically_repairable = 10; +} + +message ListStorageIncidentsRequest { + bool include_resolved = 1; + uint32 page_size = 2; + string page_token = 3; +} + +message ListStorageIncidentsResponse { + repeated StorageIncident incidents = 1; + string next_page_token = 2; +} + +message RepairStorageIncidentRequest { + string incident_id = 1; + string request_id = 2; +} + +message RepairStorageIncidentResponse { + StorageIncident incident = 1; +} + +message AcknowledgeStorageIncidentRequest { + string incident_id = 1; + string note = 2; + string request_id = 3; +} + +message AcknowledgeStorageIncidentResponse { + StorageIncident incident = 1; +} + +message AuthorizeClientTakeoverRequest { + string client_id = 1; + string client_instance_id = 2; + string request_id = 3; +} + +message AuthorizeClientTakeoverResponse { + google.protobuf.Timestamp expires_at = 1; }