test: harden harness lifecycle recovery

This commit is contained in:
2026-09-11 06:11:17 +00:00
parent e7b9625210
commit 110a329652
3 changed files with 152 additions and 12 deletions
+15 -7
View File
@@ -109,10 +109,11 @@ scripts/test-integration --suite store --case '^TestStore' --run-id store-one-ca
scripts/test-env status --run-id my-sample scripts/test-env status --run-id my-sample
scripts/test-env collect --run-id my-sample scripts/test-env collect --run-id my-sample
scripts/test-env reset --run-id my-sample scripts/test-env reset --run-id my-sample
scripts/test-env reuse --run-id my-sample scripts/test-env reuse --run-id my-sample --new-run-id my-sample-retry
scripts/test-integration --suite sample --run-id my-sample --resume scripts/test-integration --suite sample --run-id my-sample-retry --resume
scripts/test-env reset --run-id my-sample scripts/test-env reset --run-id my-sample-retry
scripts/test-env purge --run-id my-sample --execute --yes scripts/test-env purge --run-id my-sample --execute --yes
scripts/test-env purge --run-id my-sample-retry --execute --yes
scripts/test-integration --suite store --run-id store-smoke scripts/test-integration --suite store --run-id store-smoke
scripts/test-env logs --run-id store-smoke scripts/test-env logs --run-id store-smoke
@@ -153,10 +154,17 @@ The Windows artifact is linked with the GUI subsystem (`-H=windowsgui`) so
service, launcher, and tray startup do not flash a console. Human-facing modes service, launcher, and tray startup do not flash a console. Human-facing modes
still attach to a parent console explicitly when one exists. still attach to a parent console explicitly when one exists.
Suite output is capped at 1 MiB and stored as `artifacts/suite.log`. A failed Suite output is capped by the checked-in harness policy (10 MiB by default) and
run remains inspectable and can be moved back to `ready` with `recover`, then stored as `artifacts/suite.log`. A failed run remains inspectable and can be
resumed with the same run ID and deterministic shuffle seed. Test-run cleanup moved back to `ready` with `recover`, then resumed with the same run ID and
never removes the shared Go module or build-cache volumes. deterministic shuffle seed only while its source definitions and working tree
still match its manifest. `reuse` is intentionally different: after a reset or
completed run, it creates a fresh manifest/run ID with the same layer, suite,
and case selection, leaving the original evidence unchanged. `reset` refuses a
running run or one with recorded owned runtime resources, removes only its
ephemeral `runtime`, `pki`, and `scratch` directories, and retains the
manifest, journal, reports, and artifacts. Test-run cleanup never removes the
shared Go module or build-cache volumes.
Each run owns only `.test-runs/<run-id>` and resources explicitly recorded in Each run owns only `.test-runs/<run-id>` and resources explicitly recorded in
that run's versioned manifest. The journal is append-only and fsynced. `purge` that run's versioned manifest. The journal is append-only and fsynced. `purge`
+57 -5
View File
@@ -548,6 +548,7 @@ func (h *harness) environmentCommand(command string, args []string) error {
flags := flag.NewFlagSet(command, flag.ContinueOnError) flags := flag.NewFlagSet(command, flag.ContinueOnError)
flags.SetOutput(io.Discard) flags.SetOutput(io.Discard)
runID := flags.String("run-id", "", "run ID") runID := flags.String("run-id", "", "run ID")
newRunID := flags.String("new-run-id", "", "fresh run ID for reuse")
all := flags.Bool("all", false, "select every eligible run") all := flags.Bool("all", false, "select every eligible run")
execute := flags.Bool("execute", false, "perform a destructive cleanup") execute := flags.Bool("execute", false, "perform a destructive cleanup")
yes := flags.Bool("yes", false, "confirm destructive cleanup in noninteractive use") yes := flags.Bool("yes", false, "confirm destructive cleanup in noninteractive use")
@@ -557,6 +558,9 @@ func (h *harness) environmentCommand(command string, args []string) error {
if command != "purge" && (*all || *execute || *yes) { if command != "purge" && (*all || *execute || *yes) {
return fmt.Errorf("%s does not accept cleanup options", command) return fmt.Errorf("%s does not accept cleanup options", command)
} }
if command != "reuse" && *newRunID != "" {
return fmt.Errorf("%s does not accept --new-run-id", command)
}
if command == "purge" { if command == "purge" {
if (*runID == "") == (!*all) { if (*runID == "") == (!*all) {
return errors.New("purge requires exactly one of --run-id or --all") return errors.New("purge requires exactly one of --run-id or --all")
@@ -590,24 +594,72 @@ func (h *harness) environmentCommand(command string, args []string) error {
case "collect": case "collect":
return h.collect(current) return h.collect(current)
case "recover": case "recover":
if err := h.verifyResume(current); err != nil {
return err
}
if current.Phase != "interrupted" && current.Phase != "stopped" && current.Phase != "running" && current.Phase != "failed" { if current.Phase != "interrupted" && current.Phase != "stopped" && current.Phase != "running" && current.Phase != "failed" {
return fmt.Errorf("run in phase %q does not need recovery", current.Phase) return fmt.Errorf("run in phase %q does not need recovery", current.Phase)
} }
return h.transition(current, "ready", "recover", "run recovered and ready to resume") return h.transition(current, "ready", "recover", "run recovered and ready to resume")
case "reuse": case "reuse":
if current.Phase != "reset" && current.Phase != "completed" && current.Phase != "ready" { return h.reuse(current, *newRunID)
return fmt.Errorf("run in phase %q cannot be reused", current.Phase)
}
return h.transition(current, "ready", "reuse", "run retained for reuse")
case "stop": case "stop":
return h.transition(current, "stopped", "stop", "owned runtime resources stopped") return h.transition(current, "stopped", "stop", "owned runtime resources stopped")
case "reset": case "reset":
return h.transition(current, "reset", "reset", "owned runtime state reset; manifest and journal retained") return h.reset(current)
default: default:
return usageError() return usageError()
} }
} }
func (h *harness) reuse(previous *manifest, newRunID string) error {
if previous.Phase != "reset" && previous.Phase != "completed" {
return fmt.Errorf("run in phase %q cannot be reused; reset or complete it first", previous.Phase)
}
if err := h.verifyResume(previous); err != nil {
return err
}
fresh, err := h.create(newRunID, previous.Layer, previous.Suite)
if err != nil {
return err
}
fresh.TestCase = previous.TestCase
if err := h.writeManifest(fresh); err != nil {
return err
}
if err := h.appendJournal(fresh.RunID, journalEntry{At: h.now(), Step: "reuse", Status: "passed", Detail: "fresh run created from immutable prior definitions", Data: map[string]any{"prior_run_id": previous.RunID}}); err != nil {
return err
}
fmt.Fprintf(h.out, "reused_from=%s new_run_id=%s\n", previous.RunID, fresh.RunID)
return nil
}
func (h *harness) reset(current *manifest) error {
if current.Phase == "running" {
return errors.New("refusing to reset a running run; stop it first")
}
if len(current.OwnedResources) != 0 {
return errors.New("refusing to reset a run with owned runtime resources; stop and reconcile them first")
}
for _, name := range []string{"runtime", "pki", "scratch"} {
path := filepath.Join(h.runDir(current.RunID), name)
info, err := os.Lstat(path)
if os.IsNotExist(err) {
continue
}
if err != nil {
return err
}
if info.Mode()&os.ModeSymlink != 0 || !info.IsDir() {
return fmt.Errorf("refusing unsafe reset path %s", path)
}
if err := os.RemoveAll(path); err != nil {
return err
}
}
return h.transition(current, "reset", "reset", "owned runtime scratch reset; manifest, journal, and reports retained")
}
func (h *harness) purgeCommand(runID string, all, execute, yes bool) error { func (h *harness) purgeCommand(runID string, all, execute, yes bool) error {
var candidates []*manifest var candidates []*manifest
if all { if all {
+80
View File
@@ -262,6 +262,86 @@ func TestBoundedSuiteLogAndFailedRunRecovery_BH_STORE_03(t *testing.T) {
} }
} }
func TestResetOnlyRemovesEphemeralState_HP_CFG_02(t *testing.T) {
t.Parallel()
h := &harness{root: t.TempDir(), now: time.Now, out: &bytes.Buffer{}}
current, err := h.create("reset-scratch", "integration", "sample")
if err != nil {
t.Fatal(err)
}
for _, name := range []string{"runtime", "pki", "scratch", "artifacts"} {
path := filepath.Join(h.runDir(current.RunID), name)
if err := os.Mkdir(path, 0o700); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(path, "marker"), []byte(name), 0o600); err != nil {
t.Fatal(err)
}
}
if err := h.reset(current); err != nil {
t.Fatalf("reset: %v", err)
}
for _, name := range []string{"runtime", "pki", "scratch"} {
if _, err := os.Stat(filepath.Join(h.runDir(current.RunID), name)); !os.IsNotExist(err) {
t.Fatalf("reset retained %s: %v", name, err)
}
}
if _, err := os.Stat(filepath.Join(h.runDir(current.RunID), "artifacts", "marker")); err != nil {
t.Fatalf("reset removed preserved artifacts: %v", err)
}
loaded, err := h.load(current.RunID)
if err != nil || loaded.Phase != "reset" {
t.Fatalf("reset manifest = (%+v, %v)", loaded, err)
}
active, err := h.create("running-scratch", "integration", "sample")
if err != nil {
t.Fatal(err)
}
if err := h.transition(active, "running", "start", "test"); err != nil {
t.Fatal(err)
}
if err := h.reset(active); err == nil {
t.Fatal("reset allowed an active run")
}
}
func TestReuseCreatesFreshRun_HP_CFG_03(t *testing.T) {
t.Parallel()
var output bytes.Buffer
h := &harness{root: t.TempDir(), now: time.Now, out: &output}
previous, err := h.create("prior-run", "e2e", "reconnect")
if err != nil {
t.Fatal(err)
}
previous.TestCase = "^TestReconnect$"
if err := h.writeManifest(previous); err != nil {
t.Fatal(err)
}
if err := h.transition(previous, "reset", "reset", "test"); err != nil {
t.Fatal(err)
}
if err := h.environmentCommand("reuse", []string{"--run-id", previous.RunID, "--new-run-id", "fresh-run"}); err != nil {
t.Fatalf("reuse: %v", err)
}
stillPrevious, err := h.load(previous.RunID)
if err != nil || stillPrevious.Phase != "reset" {
t.Fatalf("reuse mutated source run = (%+v, %v)", stillPrevious, err)
}
fresh, err := h.load("fresh-run")
if err != nil {
t.Fatal(err)
}
if fresh.Layer != previous.Layer || fresh.Suite != previous.Suite || fresh.TestCase != previous.TestCase || fresh.Phase != "created" {
t.Fatalf("fresh reused manifest = %+v", fresh)
}
if !strings.Contains(output.String(), "reused_from=prior-run new_run_id=fresh-run") {
t.Fatalf("reuse output = %q", output.String())
}
}
func TestCoverageInventoryReferencesExistingTests_HP_CFG_01(t *testing.T) { func TestCoverageInventoryReferencesExistingTests_HP_CFG_01(t *testing.T) {
t.Parallel() t.Parallel()