test: harden harness reproducibility and cleanup

This commit is contained in:
2026-09-11 06:05:52 +00:00
parent 51fa88d05b
commit e7b9625210
5 changed files with 523 additions and 77 deletions
+13 -10
View File
@@ -34,8 +34,8 @@ scripts/build recover
The second command removes only `bin/`, `rvbox-dev-toolchain:latest`, and the
two named RVBox Go-cache volumes. `recover` rebuilds the pinned toolchain and
all binaries from source. It intentionally does not remove `.test-runs/`, which
may contain resumable environments; use the exact-run `scripts/test-env purge`
workflow for those.
may contain resumable environments; use the exact-run
`scripts/test-env purge --run-id ID --execute --yes` workflow for those.
For a native Windows run, create the immutable per-run test bundle with the
pinned toolchain. Supply a test-specific config whose endpoint and CA path are
@@ -104,6 +104,7 @@ respective durability or wire boundary:
scripts/test-env doctor
scripts/test-integration --suite sample --run-id my-sample
scripts/test-integration --list
scripts/test-integration --suite all --run-id integration-all
scripts/test-integration --suite store --case '^TestStore' --run-id store-one-case
scripts/test-env status --run-id my-sample
scripts/test-env collect --run-id my-sample
@@ -111,19 +112,19 @@ scripts/test-env reset --run-id my-sample
scripts/test-env reuse --run-id my-sample
scripts/test-integration --suite sample --run-id my-sample --resume
scripts/test-env reset --run-id my-sample
scripts/test-env purge --run-id my-sample
scripts/test-env purge --run-id my-sample --execute --yes
scripts/test-integration --suite store --run-id store-smoke
scripts/test-env logs --run-id store-smoke
scripts/test-env collect --run-id store-smoke
scripts/test-env reset --run-id store-smoke
scripts/test-env purge --run-id store-smoke
scripts/test-env purge --run-id store-smoke --execute --yes
scripts/test-integration --suite server-session --run-id session-smoke
scripts/test-env logs --run-id session-smoke
scripts/test-env collect --run-id session-smoke
scripts/test-env reset --run-id session-smoke
scripts/test-env purge --run-id session-smoke
scripts/test-env purge --run-id session-smoke --execute --yes
scripts/test-e2e --scenario smoke --run-id e2e-smoke
scripts/test-e2e --list
@@ -132,7 +133,7 @@ scripts/test-env status --run-id e2e-smoke
scripts/test-env recover --run-id e2e-smoke
scripts/test-e2e --scenario smoke --run-id e2e-smoke --resume
scripts/test-env reset --run-id e2e-smoke
scripts/test-env purge --run-id e2e-smoke
scripts/test-env purge --run-id e2e-smoke --execute --yes
```
The client runtime unit lane also exercises a real child process through the
@@ -159,10 +160,12 @@ never removes the shared Go module or build-cache volumes.
Each run owns only `.test-runs/<run-id>` and resources explicitly recorded in
that run's versioned manifest. The journal is append-only and fsynced. `purge`
validates the run ID and manifest identity, refuses symlink targets or manifests
that still list runtime resources, and then removes only that exact run. Purged
artifacts are not recoverable. Dependency cache volumes are never part of run
cleanup.
first prints the exact validated target and is dry-run by default. It requires
`--execute --yes`, refuses symlink targets or manifests that still list runtime
resources, and then removes only one completed/reset run (or eligible runs
selected by `--all`). `gc --older-than DURATION` uses the same explicit
execution confirmation. Purged artifacts are not recoverable. Dependency cache
volumes are never part of run cleanup.
`test/coverage.toml` is the incremental requirement-to-test inventory. The
`make verify` lint stage checks unique stable IDs and verifies every implemented
+41 -1
View File
@@ -4,11 +4,51 @@ set -eu
repo_root=$(CDPATH= cd -- "$(dirname -- "$0")/.." && pwd)
command=${1:-}
if [ -z "$command" ]; then
echo "usage: scripts/test-env doctor|coverage|status|logs|collect|recover|reuse|stop|reset|purge [--run-id ID]" >&2
echo "usage: scripts/test-env doctor [--e2e] | coverage|status|logs|collect|recover|reuse|stop|reset|purge|gc [options]" >&2
exit 2
fi
shift
if [ "$command" = --help ] || [ "$command" = -h ]; then
echo "usage: scripts/test-env doctor [--e2e] | coverage|status|logs|collect|recover|reuse|stop|reset|purge|gc [options]"
exit 0
fi
fail() { printf '%s\n' "test-env: $*" >&2; exit 2; }
doctor() {
e2e=no
while [ "$#" -gt 0 ]; do
case $1 in
--e2e) e2e=yes ;;
--help|-h) echo "usage: scripts/test-env doctor [--e2e]"; return 0 ;;
*) fail "unknown doctor option $1" ;;
esac
shift
done
docker version >/dev/null || fail "Docker is unavailable"
docker compose version >/dev/null || fail "Docker Compose is unavailable"
defaults=$repo_root/test/harness/defaults.toml
[ -r "$defaults" ] || fail "harness defaults are unreadable"
floor=$(awk -F '=' '/^free_space_floor_bytes[[:space:]]*=/{gsub(/[[:space:]]/, "", $2); print $2; exit}' "$defaults")
case $floor in ''|*[!0-9]*) fail "harness free-space floor is invalid" ;; esac
available_kib=$(df -Pk "$repo_root" | awk 'NR == 2 { print $4 }')
case $available_kib in ''|*[!0-9]*) fail "could not determine available disk space" ;; esac
available_bytes=$((available_kib * 1024))
[ "$available_bytes" -ge "$floor" ] || fail "available disk space $available_bytes is below harness floor $floor"
printf 'docker=ready compose=ready disk_available_bytes=%s disk_floor_bytes=%s cpus=%s\n' "$available_bytes" "$floor" "$(getconf _NPROCESSORS_ONLN 2>/dev/null || printf '?')"
if [ "$e2e" = yes ]; then
"$repo_root/scripts/windows/test-host" status
fi
cd "$repo_root"
exec docker compose -f deploy/compose.yaml run --rm toolchain go run ./test/harness doctor
}
if [ "$command" = doctor ]; then
doctor "$@"
exit $?
fi
docker version >/dev/null
docker compose version >/dev/null
+23
View File
@@ -0,0 +1,23 @@
# Local default budgets for the RVBox test harness. CI may override these only
# through an explicitly recorded policy file in a later dedicated CI adapter.
version = 1
[limits]
# One mutable integration/E2E environment prevents accidental resource overlap.
max_concurrent_runs = 1
# Go package parallelism inside the pinned toolchain container.
go_parallelism = 2
# Aggregate limits reserved for an integration/E2E Compose environment.
runtime_cpus = 2
runtime_memory_bytes = 2147483648
runtime_pids = 512
# Bounded diagnostics and disk admission.
component_log_tail_bytes = 10485760
failure_bundle_bytes = 104857600
success_report_bytes = 20971520
free_space_floor_bytes = 2147483648
# Human-readable, recorded timeout budgets.
unit_timeout = "10m"
integration_timeout = "30m"
e2e_timeout = "45m"
soak_timeout = "2h"
+360 -65
View File
@@ -7,6 +7,7 @@ import (
"crypto/sha256"
"encoding/hex"
"encoding/json"
"encoding/xml"
"errors"
"flag"
"fmt"
@@ -25,7 +26,7 @@ import (
)
const (
manifestVersion = 1
manifestVersion = 2
repositoryID = "rvbox"
maxSuiteLogSize = 1 << 20
)
@@ -33,21 +34,45 @@ const (
var runIDPattern = regexp.MustCompile(`^[a-z0-9][a-z0-9-]{0,63}$`)
type manifest struct {
Version uint32 `toml:"version" json:"version"`
RepositoryID string `toml:"repository_id" json:"repository_id"`
RunID string `toml:"run_id" json:"run_id"`
Layer string `toml:"layer" json:"layer"`
Suite string `toml:"suite" json:"suite"`
TestCase string `toml:"test_case,omitempty" json:"test_case,omitempty"`
Seed int64 `toml:"seed" json:"seed"`
GitCommit string `toml:"git_commit" json:"git_commit"`
DirtyDiffSHA256 string `toml:"dirty_diff_sha256" json:"dirty_diff_sha256"`
ToolchainImage string `toml:"toolchain_image" json:"toolchain_image"`
ComposeProject string `toml:"compose_project" json:"compose_project"`
Phase string `toml:"phase" json:"phase"`
CreatedAt time.Time `toml:"created_at" json:"created_at"`
UpdatedAt time.Time `toml:"updated_at" json:"updated_at"`
OwnedResources []string `toml:"owned_resources" json:"owned_resources"`
Version uint32 `toml:"version" json:"version"`
RepositoryID string `toml:"repository_id" json:"repository_id"`
RunID string `toml:"run_id" json:"run_id"`
Layer string `toml:"layer" json:"layer"`
Suite string `toml:"suite" json:"suite"`
TestCase string `toml:"test_case,omitempty" json:"test_case,omitempty"`
Seed int64 `toml:"seed" json:"seed"`
GitCommit string `toml:"git_commit" json:"git_commit"`
DirtyDiffSHA256 string `toml:"dirty_diff_sha256" json:"dirty_diff_sha256"`
ToolchainImage string `toml:"toolchain_image" json:"toolchain_image"`
HarnessDefaultsSHA256 string `toml:"harness_defaults_sha256" json:"harness_defaults_sha256"`
ComposeDefinitionSHA256 string `toml:"compose_definition_sha256" json:"compose_definition_sha256"`
EffectiveLimits harnessLimits `toml:"effective_limits" json:"effective_limits"`
ComposeProject string `toml:"compose_project" json:"compose_project"`
Phase string `toml:"phase" json:"phase"`
CreatedAt time.Time `toml:"created_at" json:"created_at"`
UpdatedAt time.Time `toml:"updated_at" json:"updated_at"`
OwnedResources []string `toml:"owned_resources" json:"owned_resources"`
}
type harnessDefaults struct {
Version uint32 `toml:"version"`
Limits harnessLimits `toml:"limits"`
}
type harnessLimits struct {
MaxConcurrentRuns uint32 `toml:"max_concurrent_runs" json:"max_concurrent_runs"`
GoParallelism uint32 `toml:"go_parallelism" json:"go_parallelism"`
RuntimeCPUs uint32 `toml:"runtime_cpus" json:"runtime_cpus"`
RuntimeMemoryBytes uint64 `toml:"runtime_memory_bytes" json:"runtime_memory_bytes"`
RuntimePIDs uint32 `toml:"runtime_pids" json:"runtime_pids"`
ComponentLogTailBytes uint64 `toml:"component_log_tail_bytes" json:"component_log_tail_bytes"`
FailureBundleBytes uint64 `toml:"failure_bundle_bytes" json:"failure_bundle_bytes"`
SuccessReportBytes uint64 `toml:"success_report_bytes" json:"success_report_bytes"`
FreeSpaceFloorBytes uint64 `toml:"free_space_floor_bytes" json:"free_space_floor_bytes"`
UnitTimeout string `toml:"unit_timeout" json:"unit_timeout"`
IntegrationTimeout string `toml:"integration_timeout" json:"integration_timeout"`
E2ETimeout string `toml:"e2e_timeout" json:"e2e_timeout"`
SoakTimeout string `toml:"soak_timeout" json:"soak_timeout"`
}
type journalEntry struct {
@@ -58,13 +83,33 @@ type journalEntry struct {
Data map[string]any `json:"data,omitempty"`
}
type harness struct {
root string
now func() time.Time
out io.Writer
type junitSuite struct {
XMLName xml.Name `xml:"testsuite"`
Name string `xml:"name,attr"`
Tests int `xml:"tests,attr"`
Failures int `xml:"failures,attr"`
TestCase junitCase `xml:"testcase"`
}
var integrationSuites = []string{"sample", "store", "server-session", "control", "client-agent"}
type junitCase struct {
Name string `xml:"name,attr"`
Class string `xml:"classname,attr"`
Failure *junitFailure `xml:"failure,omitempty"`
}
type junitFailure struct {
Message string `xml:"message,attr"`
}
type harness struct {
root string
repoRoot string
defaults harnessDefaults
now func() time.Time
out io.Writer
}
var integrationSuites = []string{"sample", "store", "server-session", "control", "client-agent", "all"}
var e2eScenarios = []string{"smoke", "interactive", "idempotency", "reconnect", "retention", "expiry-and-incidents", "script", "recovery", "all"}
func containsChoice(choices []string, value string) bool {
@@ -98,12 +143,83 @@ func validateTestCase(value string) error {
return nil
}
func loadHarnessDefaults(path string) (harnessDefaults, error) {
data, err := os.ReadFile(path)
if err != nil {
return harnessDefaults{}, fmt.Errorf("read harness defaults: %w", err)
}
var defaults harnessDefaults
decoder := toml.NewDecoder(bytes.NewReader(data))
decoder.DisallowUnknownFields()
if err := decoder.Decode(&defaults); err != nil {
return harnessDefaults{}, fmt.Errorf("decode harness defaults: %w", err)
}
if defaults.Version != 1 {
return harnessDefaults{}, fmt.Errorf("unsupported harness defaults version %d", defaults.Version)
}
limits := defaults.Limits
if limits.MaxConcurrentRuns != 1 || limits.GoParallelism == 0 || limits.RuntimeCPUs == 0 || limits.RuntimeMemoryBytes == 0 || limits.RuntimePIDs == 0 || limits.ComponentLogTailBytes == 0 || limits.FailureBundleBytes == 0 || limits.SuccessReportBytes == 0 || limits.FreeSpaceFloorBytes == 0 {
return harnessDefaults{}, errors.New("harness defaults contain a zero or unsupported resource limit")
}
if limits.SuccessReportBytes > limits.FailureBundleBytes {
return harnessDefaults{}, errors.New("success report budget exceeds failure bundle budget")
}
for field, value := range map[string]string{"unit_timeout": limits.UnitTimeout, "integration_timeout": limits.IntegrationTimeout, "e2e_timeout": limits.E2ETimeout, "soak_timeout": limits.SoakTimeout} {
duration, durationErr := time.ParseDuration(value)
if durationErr != nil || duration <= 0 {
return harnessDefaults{}, fmt.Errorf("harness default %s must be a positive duration", field)
}
}
return defaults, nil
}
func (h *harness) definitionHash(relative string) string {
if h.repoRoot == "" {
return "unavailable"
}
data, err := os.ReadFile(filepath.Join(h.repoRoot, relative))
if err != nil {
return "unavailable"
}
sum := sha256.Sum256(data)
return hex.EncodeToString(sum[:])
}
func (h *harness) verifyResume(current *manifest) error {
commit := strings.TrimSpace(commandOutput("git", "rev-parse", "HEAD"))
if current.GitCommit != "unavailable" && commit != current.GitCommit {
return fmt.Errorf("run commit %q does not match current commit %q; start a new run", current.GitCommit, commit)
}
dirty := repositoryDirtyHash()
if current.DirtyDiffSHA256 != hex.EncodeToString(dirty[:]) {
return errors.New("run dirty-diff hash does not match the working tree; start a new run")
}
if want := h.definitionHash("test/harness/defaults.toml"); current.HarnessDefaultsSHA256 != "" && current.HarnessDefaultsSHA256 != "unavailable" && want != current.HarnessDefaultsSHA256 {
return errors.New("harness defaults changed since the run began; start a new run")
}
if want := h.definitionHash("deploy/compose.test.yaml"); current.ComposeDefinitionSHA256 != "" && current.ComposeDefinitionSHA256 != "unavailable" && want != current.ComposeDefinitionSHA256 {
return errors.New("test Compose definition changed since the run began; start a new run")
}
return nil
}
func (h *harness) componentLogLimit() int {
if h.defaults.Limits.ComponentLogTailBytes == 0 || h.defaults.Limits.ComponentLogTailBytes > uint64(^uint(0)>>1) {
return maxSuiteLogSize
}
return int(h.defaults.Limits.ComponentLogTailBytes)
}
func runCLI(ctx context.Context, args []string) error {
repoRoot, err := os.Getwd()
if err != nil {
return err
}
h := &harness{root: filepath.Join(repoRoot, ".test-runs"), now: func() time.Time { return time.Now().UTC() }, out: os.Stdout}
defaults, err := loadHarnessDefaults(filepath.Join(repoRoot, "test", "harness", "defaults.toml"))
if err != nil {
return err
}
h := &harness{root: filepath.Join(repoRoot, ".test-runs"), repoRoot: repoRoot, defaults: defaults, now: func() time.Time { return time.Now().UTC() }, out: os.Stdout}
if len(args) == 0 {
return usageError()
}
@@ -123,7 +239,7 @@ func runCLI(ctx context.Context, args []string) error {
return h.integration(ctx, args[1:])
case "e2e":
return h.e2e(ctx, args[1:])
case "status", "logs", "collect", "recover", "reuse", "stop", "reset", "purge":
case "status", "logs", "collect", "recover", "reuse", "stop", "reset", "purge", "gc":
return h.environmentCommand(args[0], args[1:])
default:
return usageError()
@@ -131,7 +247,7 @@ func runCLI(ctx context.Context, args []string) error {
}
func usageError() error {
return errors.New("usage: harness doctor|coverage|integration|e2e|status|logs|collect|recover|reuse|stop|reset|purge")
return errors.New("usage: harness doctor|coverage|integration|e2e|status|logs|collect|recover|reuse|stop|reset|purge|gc")
}
func (h *harness) doctor(repoRoot string) error {
@@ -140,17 +256,11 @@ func (h *harness) doctor(repoRoot string) error {
return fmt.Errorf("repository check %s: %w", file, err)
}
}
if err := os.MkdirAll(h.root, 0o700); err != nil {
return err
if h.defaults.Version != 1 {
return errors.New("harness defaults were not loaded")
}
probe, err := os.CreateTemp(h.root, ".doctor-")
if err != nil {
return fmt.Errorf("test run root is not writable: %w", err)
}
name := probe.Name()
_ = probe.Close()
_ = os.Remove(name)
fmt.Fprintln(h.out, "RVBox test harness is ready; native Windows scenarios use the explicit scripts/windows/test-host.ps1 host lane.")
fmt.Fprintf(h.out, "RVBox test harness policy: max_runs=%d go_parallelism=%d free_space_floor_bytes=%d\n", h.defaults.Limits.MaxConcurrentRuns, h.defaults.Limits.GoParallelism, h.defaults.Limits.FreeSpaceFloorBytes)
fmt.Fprintln(h.out, "RVBox test harness is ready; native Windows scenarios use the explicit scripts/windows/test-host host lane.")
return nil
}
@@ -177,8 +287,8 @@ func (h *harness) integration(ctx context.Context, args []string) error {
if err := validateTestCase(*testCase); err != nil {
return err
}
if *suite == "sample" && *testCase != "" {
return errors.New("--case is only supported by executable integration suites")
if (*suite == "sample" || *suite == "all") && *testCase != "" {
return errors.New("--case is only supported by one executable integration suite")
}
var current *manifest
@@ -191,6 +301,9 @@ func (h *harness) integration(ctx context.Context, args []string) error {
if err != nil {
return err
}
if err := h.verifyResume(current); err != nil {
return err
}
if current.Layer != "integration" || current.Suite != *suite {
return errors.New("run layer/suite does not match resume request")
}
@@ -220,30 +333,36 @@ func (h *harness) integration(ctx context.Context, args []string) error {
return ctx.Err()
default:
}
if *suite == "sample" {
if err := h.appendJournal(current.RunID, journalEntry{At: h.now(), Step: "invariant", Status: "passed", Detail: "manifest ownership and journal durability verified"}); err != nil {
return err
}
} else {
var suiteErr error
switch *suite {
case "store":
suiteErr = h.runStoreSuite(ctx, current, *testCase)
case "server-session":
suiteErr = h.runServerSessionSuite(ctx, current, *testCase)
case "control":
suiteErr = h.runGoSuite(ctx, current, "control-real-grpc", "running gRPC control and JSON-RPC compatibility cases", "./internal/server/control", *testCase)
case "client-agent":
suiteErr = h.runGoSuite(ctx, current, "client-agent-real-websocket", "running real client/server WebSocket control-flow cases", "./test/integration/clientagent", *testCase)
}
if suiteErr != nil {
_ = h.transition(current, "failed", *suite+"-failed", suiteErr.Error())
suites := []string{*suite}
if *suite == "all" {
suites = []string{"sample", "store", "server-session", "control", "client-agent"}
}
for _, item := range suites {
if suiteErr := h.runIntegrationSuite(ctx, current, item, *testCase); suiteErr != nil {
_ = h.transition(current, "failed", item+"-failed", suiteErr.Error())
return suiteErr
}
}
return h.transition(current, "completed", *suite+"-complete", *suite+" integration run completed")
}
func (h *harness) runIntegrationSuite(ctx context.Context, current *manifest, suite, testCase string) error {
switch suite {
case "sample":
return h.appendJournal(current.RunID, journalEntry{At: h.now(), Step: "invariant", Status: "passed", Detail: "manifest ownership and journal durability verified"})
case "store":
return h.runStoreSuite(ctx, current, testCase)
case "server-session":
return h.runServerSessionSuite(ctx, current, testCase)
case "control":
return h.runGoSuite(ctx, current, "control-real-grpc", "running gRPC control and JSON-RPC compatibility cases", "./internal/server/control", testCase)
case "client-agent":
return h.runGoSuite(ctx, current, "client-agent-real-websocket", "running real client/server WebSocket control-flow cases", "./test/integration/clientagent", testCase)
default:
return fmt.Errorf("unknown integration suite %q", suite)
}
}
// e2e runs production-shaped Go scenarios against real local listeners and
// durable stores. Native Windows work is an additional host lane invoked by
// scripts/windows/test-host.ps1; it is never silently replaced by Wine or a
@@ -285,6 +404,9 @@ func (h *harness) e2e(ctx context.Context, args []string) error {
if err != nil {
return err
}
if err := h.verifyResume(current); err != nil {
return err
}
if current.Layer != "e2e" || current.Suite != *scenario {
return errors.New("run layer/scenario does not match resume request")
}
@@ -367,7 +489,7 @@ func (h *harness) runStoreSuite(ctx context.Context, current *manifest, testCase
if err := os.MkdirAll(artifactDir, 0o700); err != nil {
return err
}
capture := &limitedCapture{limit: maxSuiteLogSize}
capture := &limitedCapture{limit: h.componentLogLimit()}
arguments := []string{"test", "-count=1", "-tags=integration", "-shuffle=" + strconv.FormatInt(current.Seed, 10), "-timeout=2m"}
if testCase != "" {
arguments = append(arguments, "-run", testCase)
@@ -399,7 +521,7 @@ func (h *harness) runGoSuite(ctx context.Context, current *manifest, step, detai
if err := os.MkdirAll(artifactDir, 0o700); err != nil {
return err
}
capture := &limitedCapture{limit: maxSuiteLogSize}
capture := &limitedCapture{limit: h.componentLogLimit()}
arguments := []string{"test", "-race", "-count=1", "-shuffle=" + strconv.FormatInt(current.Seed, 10), "-timeout=2m"}
if testCase != "" {
arguments = append(arguments, "-run", testCase)
@@ -420,12 +542,27 @@ func (h *harness) runGoSuite(ctx context.Context, current *manifest, step, detai
}
func (h *harness) environmentCommand(command string, args []string) error {
if command == "gc" {
return h.gc(args)
}
flags := flag.NewFlagSet(command, flag.ContinueOnError)
flags.SetOutput(io.Discard)
runID := flags.String("run-id", "", "run ID")
all := flags.Bool("all", false, "select every eligible run")
execute := flags.Bool("execute", false, "perform a destructive cleanup")
yes := flags.Bool("yes", false, "confirm destructive cleanup in noninteractive use")
if err := flags.Parse(args); err != nil {
return err
}
if command != "purge" && (*all || *execute || *yes) {
return fmt.Errorf("%s does not accept cleanup options", command)
}
if command == "purge" {
if (*runID == "") == (!*all) {
return errors.New("purge requires exactly one of --run-id or --all")
}
return h.purgeCommand(*runID, *all, *execute, *yes)
}
if *runID == "" {
return fmt.Errorf("%s requires --run-id", command)
}
@@ -435,7 +572,12 @@ func (h *harness) environmentCommand(command string, args []string) error {
}
switch command {
case "status":
encoded, _ := json.MarshalIndent(current, "", " ")
resumeErr := h.verifyResume(current)
status := map[string]any{"manifest": current, "run_bytes": directorySize(h.runDir(current.RunID)), "resume_valid": resumeErr == nil}
if resumeErr != nil {
status["resume_error"] = resumeErr.Error()
}
encoded, _ := json.MarshalIndent(status, "", " ")
fmt.Fprintln(h.out, string(encoded))
return nil
case "logs":
@@ -461,13 +603,118 @@ func (h *harness) environmentCommand(command string, args []string) error {
return h.transition(current, "stopped", "stop", "owned runtime resources stopped")
case "reset":
return h.transition(current, "reset", "reset", "owned runtime state reset; manifest and journal retained")
case "purge":
return h.purge(current)
default:
return usageError()
}
}
func (h *harness) purgeCommand(runID string, all, execute, yes bool) error {
var candidates []*manifest
if all {
loaded, err := h.eligiblePurgeManifests(time.Time{})
if err != nil {
return err
}
candidates = loaded
} else {
current, err := h.load(runID)
if err != nil {
return err
}
if err := h.validatePurge(current); err != nil {
return err
}
candidates = []*manifest{current}
}
if len(candidates) == 0 {
fmt.Fprintln(h.out, "no eligible RVBox test runs")
return nil
}
for _, current := range candidates {
fmt.Fprintf(h.out, "purge_target=%s phase=%s bytes=%d\n", h.runDir(current.RunID), current.Phase, directorySize(h.runDir(current.RunID)))
}
if !execute {
fmt.Fprintln(h.out, "dry run only; rerun with --execute --yes to remove exactly these validated targets")
return nil
}
if !yes {
return errors.New("purge --execute requires --yes in the noninteractive harness")
}
for _, current := range candidates {
if err := h.purge(current); err != nil {
return err
}
}
return nil
}
func (h *harness) gc(args []string) error {
flags := flag.NewFlagSet("gc", flag.ContinueOnError)
flags.SetOutput(io.Discard)
olderThan := flags.String("older-than", "", "minimum age of completed/reset runs")
execute := flags.Bool("execute", false, "perform a destructive cleanup")
yes := flags.Bool("yes", false, "confirm destructive cleanup in noninteractive use")
if err := flags.Parse(args); err != nil {
return err
}
if *olderThan == "" {
return errors.New("gc requires --older-than")
}
age, err := time.ParseDuration(*olderThan)
if err != nil || age <= 0 {
return errors.New("gc --older-than must be a positive duration")
}
candidates, err := h.eligiblePurgeManifests(h.now().Add(-age))
if err != nil {
return err
}
if len(candidates) == 0 {
fmt.Fprintln(h.out, "no eligible RVBox test runs")
return nil
}
for _, current := range candidates {
fmt.Fprintf(h.out, "gc_target=%s phase=%s bytes=%d\n", h.runDir(current.RunID), current.Phase, directorySize(h.runDir(current.RunID)))
}
if !*execute {
fmt.Fprintln(h.out, "dry run only; rerun with --execute --yes to remove exactly these validated targets")
return nil
}
if !*yes {
return errors.New("gc --execute requires --yes in the noninteractive harness")
}
for _, current := range candidates {
if err := h.purge(current); err != nil {
return err
}
}
return nil
}
func (h *harness) eligiblePurgeManifests(before time.Time) ([]*manifest, error) {
entries, err := os.ReadDir(h.root)
if os.IsNotExist(err) {
return nil, nil
}
if err != nil {
return nil, err
}
result := make([]*manifest, 0, len(entries))
for _, entry := range entries {
if !entry.IsDir() || entry.Type()&os.ModeSymlink != 0 || validateRunID(entry.Name()) != nil {
continue
}
current, loadErr := h.load(entry.Name())
if loadErr != nil || h.validatePurge(current) != nil {
continue
}
if !before.IsZero() && !current.UpdatedAt.Before(before) {
continue
}
result = append(result, current)
}
return result, nil
}
type limitedCapture struct {
data []byte
limit int
@@ -525,7 +772,10 @@ func (h *harness) create(requestedID, layer, suite string) (*manifest, error) {
Version: manifestVersion, RepositoryID: repositoryID, RunID: requestedID,
Layer: layer, Suite: suite, Seed: seedValue.Int64(), GitCommit: strings.TrimSpace(commit),
DirtyDiffSHA256: hex.EncodeToString(diffHash[:]), ToolchainImage: os.Getenv("RVBOX_TOOLCHAIN_IMAGE"),
ComposeProject: "rvbox-test-" + requestedID, Phase: "created", CreatedAt: now, UpdatedAt: now,
HarnessDefaultsSHA256: h.definitionHash("test/harness/defaults.toml"),
ComposeDefinitionSHA256: h.definitionHash("deploy/compose.test.yaml"),
EffectiveLimits: h.defaults.Limits,
ComposeProject: "rvbox-test-" + requestedID, Phase: "created", CreatedAt: now, UpdatedAt: now,
OwnedResources: []string{},
}
if err := h.writeManifest(current); err != nil {
@@ -596,12 +846,32 @@ func (h *harness) collect(current *manifest) error {
if err := os.MkdirAll(reportDir, 0o700); err != nil {
return err
}
report := map[string]any{"run_id": current.RunID, "layer": current.Layer, "suite": current.Suite, "phase": current.Phase, "collected_at": h.now(), "payloads_included": false}
resumeErr := h.verifyResume(current)
report := map[string]any{"run_id": current.RunID, "layer": current.Layer, "suite": current.Suite, "phase": current.Phase, "collected_at": h.now(), "payloads_included": false, "run_bytes": directorySize(h.runDir(current.RunID)), "resume_valid": resumeErr == nil}
if resumeErr != nil {
report["resume_error"] = resumeErr.Error()
}
data, _ := json.MarshalIndent(report, "", " ")
path := filepath.Join(reportDir, "report.json")
if err := atomicWrite(path, append(data, '\n'), 0o600); err != nil {
return err
}
failed := current.Phase == "failed"
suite := junitSuite{Name: current.Layer + "/" + current.Suite, Tests: 1, TestCase: junitCase{Name: current.TestCase, Class: current.Layer + "/" + current.Suite}}
if suite.TestCase.Name == "" {
suite.TestCase.Name = current.Suite
}
if failed {
suite.Failures = 1
suite.TestCase.Failure = &junitFailure{Message: "harness run failed; inspect bounded suite.log and journal.jsonl"}
}
encodedJUnit, err := xml.MarshalIndent(suite, "", " ")
if err != nil {
return err
}
if err := atomicWrite(filepath.Join(reportDir, "junit.xml"), append(append([]byte(xml.Header), encodedJUnit...), '\n'), 0o600); err != nil {
return err
}
if err := h.appendJournal(current.RunID, journalEntry{At: h.now(), Step: "collect", Status: "passed", Detail: "bounded redacted report written"}); err != nil {
return err
}
@@ -610,6 +880,21 @@ func (h *harness) collect(current *manifest) error {
}
func (h *harness) purge(current *manifest) error {
if err := h.validatePurge(current); err != nil {
return err
}
runDir := h.runDir(current.RunID)
if err := os.RemoveAll(runDir); err != nil {
return err
}
fmt.Fprintf(h.out, "purged %s (not recoverable)\n", runDir)
return nil
}
func (h *harness) validatePurge(current *manifest) error {
if current.Phase != "completed" && current.Phase != "reset" {
return fmt.Errorf("refusing to purge run in phase %q; reset it first", current.Phase)
}
runDir := h.runDir(current.RunID)
info, err := os.Lstat(runDir)
if err != nil {
@@ -621,13 +906,23 @@ func (h *harness) purge(current *manifest) error {
if len(current.OwnedResources) != 0 {
return errors.New("refusing to purge while manifest still lists owned runtime resources; reset first")
}
if err := os.RemoveAll(runDir); err != nil {
return err
}
fmt.Fprintf(h.out, "purged %s (not recoverable)\n", runDir)
return nil
}
func directorySize(path string) uint64 {
var total uint64
_ = filepath.WalkDir(path, func(_ string, entry os.DirEntry, err error) error {
if err != nil || entry.Type().IsDir() || entry.Type()&os.ModeSymlink != 0 {
return nil
}
if info, infoErr := entry.Info(); infoErr == nil && info.Size() > 0 {
total += uint64(info.Size())
}
return nil
})
return total
}
func (h *harness) runDir(runID string) string { return filepath.Join(h.root, runID) }
func (h *harness) manifestPath(runID string) string {
return filepath.Join(h.runDir(runID), "run.toml")
+86 -1
View File
@@ -36,6 +36,9 @@ func TestSampleRunLifecycle_HP_CFG_01(t *testing.T) {
if _, err := os.Stat(filepath.Join(root, "sample-run", "artifacts", "report.json")); err != nil {
t.Fatal(err)
}
if _, err := os.Stat(filepath.Join(root, "sample-run", "artifacts", "junit.xml")); err != nil {
t.Fatal(err)
}
if err := h.transition(loaded, "reset", "reset", "test"); err != nil {
t.Fatal(err)
}
@@ -59,6 +62,9 @@ func TestRunIsolationAndManifestIdentity_HP_CFG_01(t *testing.T) {
if err != nil {
t.Fatal(err)
}
if err := h.transition(first, "reset", "reset", "test"); err != nil {
t.Fatal(err)
}
if err := h.purge(first); err != nil {
t.Fatal(err)
}
@@ -81,6 +87,63 @@ func TestRunIsolationAndManifestIdentity_HP_CFG_01(t *testing.T) {
}
}
func TestPurgeRequiresExplicitExecution_HP_CFG_01(t *testing.T) {
t.Parallel()
var output bytes.Buffer
h := &harness{root: t.TempDir(), now: time.Now, out: &output}
current, err := h.create("safe-purge", "integration", "sample")
if err != nil {
t.Fatal(err)
}
if err := h.transition(current, "reset", "reset", "test"); err != nil {
t.Fatal(err)
}
if err := h.environmentCommand("purge", []string{"--run-id", current.RunID}); err != nil {
t.Fatalf("purge dry run: %v", err)
}
if _, err := h.load(current.RunID); err != nil {
t.Fatalf("dry-run purge removed run: %v", err)
}
if err := h.environmentCommand("purge", []string{"--run-id", current.RunID, "--execute"}); err == nil {
t.Fatal("unconfirmed purge execution succeeded")
}
if err := h.environmentCommand("purge", []string{"--run-id", current.RunID, "--execute", "--yes"}); err != nil {
t.Fatalf("confirmed purge: %v", err)
}
if _, err := os.Stat(h.runDir(current.RunID)); !os.IsNotExist(err) {
t.Fatalf("confirmed purge retained run: %v", err)
}
}
func TestGCOnlyRemovesEligibleOldRuns_HP_CFG_01(t *testing.T) {
t.Parallel()
now := time.Date(2026, time.September, 11, 6, 0, 0, 0, time.UTC)
h := &harness{root: t.TempDir(), now: func() time.Time { return now }, out: &bytes.Buffer{}}
old, err := h.create("old-reset", "integration", "sample")
if err != nil {
t.Fatal(err)
}
if err := h.transition(old, "reset", "reset", "test"); err != nil {
t.Fatal(err)
}
now = now.Add(2 * time.Hour)
active, err := h.create("active-run", "integration", "sample")
if err != nil {
t.Fatal(err)
}
if err := h.gc([]string{"--older-than", "1h", "--execute", "--yes"}); err != nil {
t.Fatalf("gc: %v", err)
}
if _, err := os.Stat(h.runDir(old.RunID)); !os.IsNotExist(err) {
t.Fatalf("gc retained old reset run: %v", err)
}
if _, err := h.load(active.RunID); err != nil {
t.Fatalf("gc affected active run: %v", err)
}
}
func TestRunIDTraversalAndSymlinkPurgeRejected_HP_CFG_01(t *testing.T) {
t.Parallel()
@@ -131,7 +194,7 @@ func TestStableSuiteAndScenarioListing_HP_CFG_01(t *testing.T) {
if err := h.integration(context.Background(), []string{"--list"}); err != nil {
t.Fatalf("list integration suites: %v", err)
}
if got, want := output.String(), "integration suites:\nsample\nstore\nserver-session\ncontrol\nclient-agent\n"; got != want {
if got, want := output.String(), "integration suites:\nsample\nstore\nserver-session\ncontrol\nclient-agent\nall\n"; got != want {
t.Fatalf("integration list = %q, want %q", got, want)
}
output.Reset()
@@ -149,6 +212,28 @@ func TestStableSuiteAndScenarioListing_HP_CFG_01(t *testing.T) {
}
}
func TestHarnessDefaultsAndResumeIdentity_HP_CFG_01(t *testing.T) {
t.Parallel()
defaults, err := loadHarnessDefaults("defaults.toml")
if err != nil {
t.Fatalf("load defaults: %v", err)
}
if defaults.Limits.MaxConcurrentRuns != 1 || defaults.Limits.FreeSpaceFloorBytes == 0 {
t.Fatalf("defaults = %#v", defaults)
}
h := &harness{root: t.TempDir(), now: time.Now, out: &bytes.Buffer{}}
current, err := h.create("resume-identity", "integration", "sample")
if err != nil {
t.Fatal(err)
}
current.GitCommit = "different-commit"
if err := h.verifyResume(current); err == nil || !strings.Contains(err.Error(), "commit") {
t.Fatalf("verifyResume mismatched commit = %v", err)
}
}
func TestBoundedSuiteLogAndFailedRunRecovery_BH_STORE_03(t *testing.T) {
t.Parallel()