Files
boshu2__agentops/cli/internal/goals/measure_test.go
T
Bo 62cc3b6ee0 Clear the operations-layer alignment residuals (#1054)
Closes out the six residual items #1051 disclosed: terminology residue
on non-authority surfaces, the eval command-surface fixture that failed
when executed (#{3,4} -> #{3,5}), the vacuous retrieval-quality canary
and the nightly job that ran it, the consumer-free dream config block
and its exclusive helpers, the remaining knowledge-shaped writers moved
to the scratch tier, and the MEMORY.md consumer audit.

bin/ralph still resumes legacy .agents/ralph/ checkpoints so the
documented backwards-compat contract holds without a migration; both
paths and the outside-both refusal are now tested.

Fresh author-distinct validation returned PASS with empty not_checked,
after an earlier revision failed on a dangling nightly invoker and a
back-compat test regression that were fixed and independently
re-verified.

Test-Removal-Reason: the dream config subsystem was deleted with its tests (operations-layer residuals)
2026-08-08 14:34:14 -04:00

637 lines
19 KiB
Go

package goals
import (
"context"
"errors"
"os"
"strconv"
"strings"
"sync"
"syscall"
"testing"
"time"
)
func TestMeasureOne_Pass(t *testing.T) {
goal := Goal{
ID: "test-pass",
Check: "exit 0",
Weight: 5,
}
m := MeasureOne(goal, 5*time.Second)
if m.Result != "pass" {
t.Errorf("expected pass, got %q", m.Result)
}
if m.GoalID != "test-pass" {
t.Errorf("GoalID = %q, want %q", m.GoalID, "test-pass")
}
if m.Weight != 5 {
t.Errorf("Weight = %d, want 5", m.Weight)
}
if m.Duration < 0 {
t.Errorf("Duration should be >= 0, got %f", m.Duration)
}
}
func TestMeasureOne_Fail(t *testing.T) {
goal := Goal{
ID: "test-fail",
Check: "exit 1",
Weight: 3,
}
m := MeasureOne(goal, 5*time.Second)
if m.Result != "fail" {
t.Errorf("expected fail, got %q", m.Result)
}
}
func TestMeasureOne_Timeout(t *testing.T) {
goal := Goal{
ID: "test-timeout",
Check: "sleep 10",
Weight: 1,
}
m := MeasureOne(goal, 100*time.Millisecond)
if m.Result != "skip" {
t.Errorf("expected skip on timeout, got %q", m.Result)
}
}
func TestMeasureOne_OutputTruncated(t *testing.T) {
goal := Goal{
ID: "test-truncate",
Check: "printf '%600s' | tr ' ' 'A'",
Weight: 1,
}
m := MeasureOne(goal, 5*time.Second)
if len([]rune(m.Output)) > 500 {
t.Errorf("output should be truncated to <= 500 runes, got %d", len([]rune(m.Output)))
}
}
// TestTruncateOutput_PreservesTailHint guards against the regression seen in
// the 2026-04-26 nightly retro: head-only truncation cut diagnostic hints
// that operators rely on (e.g. "sessions must use 'ao lookup --cite ...'").
// A 1000-rune input must keep both the leading FAIL label and the trailing
// HINT_AT_END marker.
func TestTruncateOutput_PreservesTailHint(t *testing.T) {
prefix := "FAIL: "
suffix := " — HINT_AT_END"
fillerRunes := 1000 - len([]rune(prefix)) - len([]rune(suffix))
input := prefix + strings.Repeat("x", fillerRunes) + suffix
if len([]rune(input)) != 1000 {
t.Fatalf("test fixture wrong size: got %d runes, want 1000", len([]rune(input)))
}
got := truncateOutput([]byte(input))
if !strings.Contains(got, "HINT_AT_END") {
t.Errorf("trailing hint dropped; output = %q", got)
}
if !strings.Contains(got, "FAIL:") {
t.Errorf("leading FAIL label dropped; output = %q", got)
}
if !strings.Contains(got, "[truncated]") {
t.Errorf("truncation marker missing; output = %q", got)
}
if got == input {
t.Errorf("output unchanged from input; truncation did not run")
}
}
// TestTruncateOutput_ShortInputUntouched verifies the fast path is unchanged
// for inputs at or below the cap.
func TestTruncateOutput_ShortInputUntouched(t *testing.T) {
short := "FAIL: small output — operator hint here"
got := truncateOutput([]byte(short))
if got != short {
t.Errorf("short input mutated: got %q, want %q", got, short)
}
}
func TestMeasureOne_ContinuousMetric_ParsesValue(t *testing.T) {
threshold := 0.5
goal := Goal{
ID: "test-continuous",
Check: "echo 0.75",
Weight: 2,
Continuous: &ContinuousMetric{
Metric: "my_metric",
Threshold: threshold,
},
}
m := MeasureOne(goal, 5*time.Second)
if m.Value == nil {
t.Fatal("expected Value to be set for continuous metric")
}
if *m.Value != 0.75 {
t.Errorf("Value = %f, want 0.75", *m.Value)
}
if m.Threshold == nil {
t.Fatal("expected Threshold to be set for continuous metric")
}
if *m.Threshold != threshold {
t.Errorf("Threshold = %f, want %f", *m.Threshold, threshold)
}
}
func TestMeasureOne_ContinuousMetric_NonNumericOutput(t *testing.T) {
goal := Goal{
ID: "test-nonnumeric",
Check: "echo hello",
Weight: 1,
Continuous: &ContinuousMetric{
Metric: "my_metric",
Threshold: 0.5,
},
}
m := MeasureOne(goal, 5*time.Second)
if m.Value != nil {
t.Errorf("expected Value to be nil for non-numeric output, got %f", *m.Value)
}
}
func TestClassifyResult(t *testing.T) {
tests := []struct {
name string
ctxErr error
cmdErr error
want string
}{
{name: "pass", want: resultPass},
{name: "fail", cmdErr: errors.New("boom"), want: resultFail},
{name: "timeout", ctxErr: context.DeadlineExceeded, want: resultSkip},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
if got := classifyResult(tt.ctxErr, tt.cmdErr); got != tt.want {
t.Fatalf("classifyResult(%v, %v) = %q, want %q", tt.ctxErr, tt.cmdErr, got, tt.want)
}
})
}
}
func TestTruncateOutput_TrimWithoutTruncation(t *testing.T) {
got := truncateOutput([]byte(" ok \n"))
if got != "ok" {
t.Fatalf("truncateOutput returned %q, want %q", got, "ok")
}
}
func TestApplyContinuousMetric(t *testing.T) {
goal := Goal{
Continuous: &ContinuousMetric{
Metric: "coverage",
Threshold: 90.0,
},
}
m := &Measurement{Output: "91.5"}
applyContinuousMetric(m, goal)
if m.Value == nil || *m.Value != 91.5 {
t.Fatalf("Value = %v, want 91.5", m.Value)
}
if m.Threshold == nil || *m.Threshold != 90.0 {
t.Fatalf("Threshold = %v, want 90.0", m.Threshold)
}
blank := &Measurement{Output: ""}
applyContinuousMetric(blank, goal)
if blank.Value != nil || blank.Threshold != nil {
t.Fatal("blank output should not set continuous metric values")
}
}
func TestMeasure_MetaGoalsRunFirst(t *testing.T) {
gf := &GoalFile{
Version: 2,
Goals: []Goal{
{ID: "non-meta-1", Check: "exit 0", Weight: 1, Type: GoalTypeHealth},
{ID: "meta-1", Check: "exit 0", Weight: 1, Type: GoalTypeMeta},
{ID: "non-meta-2", Check: "exit 0", Weight: 1, Type: GoalTypeQuality},
},
}
snap := Measure(gf, 5*time.Second)
if len(snap.Goals) != 3 {
t.Fatalf("expected 3 measurements, got %d", len(snap.Goals))
}
if snap.Goals[0].GoalID != "meta-1" {
t.Errorf("expected meta-1 first, got %q", snap.Goals[0].GoalID)
}
}
func TestMeasure_SummaryCorrect(t *testing.T) {
gf := &GoalFile{
Version: 2,
Goals: []Goal{
{ID: "pass-1", Check: "exit 0", Weight: 5, Type: GoalTypeHealth},
{ID: "pass-2", Check: "exit 0", Weight: 3, Type: GoalTypeHealth},
{ID: "fail-1", Check: "exit 1", Weight: 2, Type: GoalTypeHealth},
},
}
snap := Measure(gf, 5*time.Second)
if snap.Summary.Total != 3 {
t.Errorf("Total = %d, want 3", snap.Summary.Total)
}
if snap.Summary.Passing != 2 {
t.Errorf("Passing = %d, want 2", snap.Summary.Passing)
}
if snap.Summary.Failing != 1 {
t.Errorf("Failing = %d, want 1", snap.Summary.Failing)
}
if snap.Summary.Score != 80.0 {
t.Errorf("Score = %f, want 80.0", snap.Summary.Score)
}
}
func TestMeasure_SkippedGoalsExcludedFromScore(t *testing.T) {
gf := &GoalFile{
Version: 2,
Goals: []Goal{
{ID: "pass-1", Check: "exit 0", Weight: 5, Type: GoalTypeHealth},
{ID: "skip-1", Check: "sleep 5", Weight: 10, Type: GoalTypeHealth},
},
}
snap := Measure(gf, 2*time.Second)
if snap.Summary.Skipped != 1 {
t.Errorf("Skipped = %d, want 1", snap.Summary.Skipped)
}
if snap.Summary.Score != 100.0 {
t.Errorf("Score = %f, want 100.0 (skipped excluded)", snap.Summary.Score)
}
}
func TestMeasureWithTotalTimeoutSkipsQueuedGoals(t *testing.T) {
gf := &GoalFile{
Version: 2,
Goals: []Goal{
{ID: "slow-1", Check: "sleep 5", Weight: 1, Type: GoalTypeHealth},
{ID: "slow-2", Check: "sleep 5", Weight: 1, Type: GoalTypeHealth},
{ID: "slow-3", Check: "sleep 5", Weight: 1, Type: GoalTypeHealth},
},
}
start := time.Now()
snap := MeasureWithTotalTimeout(gf, 5*time.Second, 150*time.Millisecond)
elapsed := time.Since(start)
if elapsed > 2*time.Second {
t.Fatalf("MeasureWithTotalTimeout took %s, want under 2s", elapsed)
}
if snap.Summary.Total != 3 {
t.Fatalf("Total = %d, want 3", snap.Summary.Total)
}
if snap.Summary.Skipped != 3 {
t.Fatalf("Skipped = %d, want 3", snap.Summary.Skipped)
}
for _, m := range snap.Goals {
if m.Result != resultSkip {
t.Fatalf("%s result = %q, want skip", m.GoalID, m.Result)
}
}
}
func TestMeasure_EmptyGoals(t *testing.T) {
gf := &GoalFile{Version: 2, Goals: []Goal{}}
snap := Measure(gf, 5*time.Second)
if snap.Summary.Total != 0 {
t.Errorf("Total = %d, want 0", snap.Summary.Total)
}
if snap.Summary.Score != 0 {
t.Errorf("Score = %f, want 0 for empty goals", snap.Summary.Score)
}
if snap.Timestamp == "" {
t.Error("Timestamp should not be empty")
}
}
func TestRequiresExclusiveExecution(t *testing.T) {
if !requiresExclusiveExecution(Goal{Check: "go test ./..."}) {
t.Fatal("go test checks should require exclusive execution")
}
if !requiresExclusiveExecution(Goal{Check: "./scripts/check-cmdao-coverage-floor.sh"}) {
t.Fatal("coverage floor check should require exclusive execution")
}
if requiresExclusiveExecution(Goal{Check: "echo ok"}) {
t.Fatal("simple commands should not require exclusive execution")
}
}
func TestComputeSummary_AllOutcomes(t *testing.T) {
summary := computeSummary([]Measurement{
{Result: resultPass, Weight: 3},
{Result: resultFail, Weight: 1},
{Result: resultSkip, Weight: 9},
})
if summary.Total != 3 || summary.Passing != 1 || summary.Failing != 1 || summary.Skipped != 1 {
t.Fatalf("unexpected summary counts: %+v", summary)
}
if summary.Score != 75.0 {
t.Fatalf("Score = %f, want 75.0", summary.Score)
}
}
func TestTruncateOutput_MultiByteRunes(t *testing.T) {
runes := make([]rune, 501)
for i := range runes {
runes[i] = '世'
}
input := []byte(string(runes))
result := truncateOutput(input)
runeCount := len([]rune(result))
// New shape: head (200 CJK) + marker (ASCII + ellipses) + tail (200 CJK).
expected := truncateHead + len([]rune(truncateMarker)) + truncateTail
if runeCount != expected {
t.Errorf("rune count = %d, want %d (head+marker+tail shape)", runeCount, expected)
}
for i, r := range result {
if r == '\uFFFD' {
t.Errorf("invalid UTF-8 at byte %d (replacement character found)", i)
break
}
}
}
// TestMeasureOne_StartError exercises the cmd.Start() error path in MeasureOne
// (lines 109-113). When bash itself cannot start, the function should return
// a fail result with the error message in Output.
func TestMeasureOne_StartError(t *testing.T) {
// Set PATH empty to make "bash" unresolvable; t.Setenv auto-restores.
t.Setenv("PATH", "")
goal := Goal{
ID: "start-error",
Check: "echo hello",
Weight: 4,
}
m := MeasureOne(goal, 5*time.Second)
if m.Result != "fail" {
t.Errorf("Result = %q, want %q for start error", m.Result, "fail")
}
if m.GoalID != "start-error" {
t.Errorf("GoalID = %q, want %q", m.GoalID, "start-error")
}
if m.Weight != 4 {
t.Errorf("Weight = %d, want 4", m.Weight)
}
if m.Output == "" {
t.Error("Output should contain the start error message")
}
if m.Duration < 0 {
t.Errorf("Duration should be >= 0, got %f", m.Duration)
}
}
// TestRunGoals_OnlyMetaGoals_EarlyReturn exercises the early return at line 186
// when all goals are meta-type and the nonMeta slice is empty.
func TestRunGoals_OnlyMetaGoals_EarlyReturn(t *testing.T) {
goals := []Goal{
{ID: "meta-a", Check: "echo a", Weight: 2, Type: GoalTypeMeta},
{ID: "meta-b", Check: "echo b", Weight: 3, Type: GoalTypeMeta},
}
measurements := runGoals(goals, 5*time.Second)
if len(measurements) != 2 {
t.Fatalf("got %d measurements, want 2", len(measurements))
}
if measurements[0].GoalID != "meta-a" {
t.Errorf("measurements[0].GoalID = %q, want %q", measurements[0].GoalID, "meta-a")
}
if measurements[1].GoalID != "meta-b" {
t.Errorf("measurements[1].GoalID = %q, want %q", measurements[1].GoalID, "meta-b")
}
for i, m := range measurements {
if m.Result != "pass" {
t.Errorf("measurements[%d].Result = %q, want %q", i, m.Result, "pass")
}
}
}
// TestRunGoals_EmptyGoals exercises runGoals with no goals at all.
func TestRunGoals_EmptyGoals(t *testing.T) {
measurements := runGoals([]Goal{}, 5*time.Second)
if len(measurements) != 0 {
t.Errorf("got %d measurements, want 0 for empty goals", len(measurements))
}
}
func TestGitSHA_OutsideGitRepo(t *testing.T) {
tmpDir := t.TempDir()
t.Chdir(tmpDir)
sha := gitSHA()
if sha != "" {
t.Errorf("expected empty SHA outside git repo, got %q", sha)
}
}
func TestChildGroupsInitialized(t *testing.T) {
// Bug #7: childGroups.pids should be non-nil at package init time.
// Before the fix, it starts nil and relies on lazy init in trackChild.
if childGroups.pids == nil {
t.Fatal("childGroups.pids is nil at package init; expected eager initialization")
}
}
func TestRunGoals_SignalHandlerCallsExit(t *testing.T) {
// Exercise the signal handler branch in runGoals: when a SIGINT arrives
// during goal execution, the handler calls killAllChildren() and osExitFn(130).
// We override osExitFn to capture the exit code instead of terminating.
var exitCode int
exitCalled := make(chan struct{})
origExit := osExitFn
osExitFn = func(code int) {
exitCode = code
close(exitCalled)
// Block forever so the goroutine doesn't return and cause races.
select {}
}
defer func() { osExitFn = origExit }()
// Use a goal that sleeps long enough for us to send a signal.
goals := []Goal{
{ID: "slow", Check: "sleep 30", Weight: 1, Type: GoalTypeHealth},
}
done := make(chan struct{})
go func() {
runGoals(goals, 30*time.Second)
close(done)
}()
// Give the goroutine time to start and install the signal handler,
// then send SIGINT to ourselves.
time.Sleep(100 * time.Millisecond)
proc, err := os.FindProcess(os.Getpid())
if err != nil {
t.Fatalf("FindProcess: %v", err)
}
if err := proc.Signal(syscall.SIGINT); err != nil {
t.Fatalf("sending SIGINT: %v", err)
}
select {
case <-exitCalled:
if exitCode != 130 {
t.Errorf("exit code = %d, want 130", exitCode)
}
case <-time.After(5 * time.Second):
t.Fatal("timed out waiting for signal handler to call osExitFn")
}
}
func TestTrackChild_ConcurrentAccess(t *testing.T) {
// Bug #7: Verify trackChild/untrackChild are safe under concurrent access.
// Must pass with -race flag.
const goroutines = 10
var wg sync.WaitGroup
wg.Add(goroutines * 2)
for i := 0; i < goroutines; i++ {
pid := 10000 + i
go func(p int) {
defer wg.Done()
trackChild(p)
}(pid)
go func(p int) {
defer wg.Done()
untrackChild(p)
}(pid)
}
wg.Wait()
// Clean up: remove any leftover tracked pids from this test.
childGroups.mu.Lock()
for pid := 10000; pid < 10000+goroutines; pid++ {
delete(childGroups.pids, pid)
}
childGroups.mu.Unlock()
}
func TestMeasureOne_Skip_ExitCode77(t *testing.T) {
// When a gate exits 77 (autotools skip convention), the goals runner
// must classify the measurement as `skip`, not `fail`. This is the
// quarantine-by-precondition path skip-aware gate scripts use when a
// precondition is absent — failing under "no signal" is a
// misclassification that artificially drags fitness scores.
g := Goal{ID: "skip77", Check: "exit 77", Weight: 3}
m := MeasureOne(g, time.Second)
if m.Result != resultSkip {
t.Fatalf("Result = %q, want %q for exit 77", m.Result, resultSkip)
}
if m.GoalID != "skip77" {
t.Errorf("GoalID = %q, want skip77", m.GoalID)
}
if m.Weight != 3 {
t.Errorf("Weight = %d, want 3", m.Weight)
}
}
func TestMeasureOne_FailOnOtherNonZeroExitCodes(t *testing.T) {
// Non-77 non-zero exits must still classify as fail. SKIP must be
// opt-in via the explicit autotools convention, not a default for
// every gate that returns >0.
for _, code := range []int{1, 2, 7, 76, 78, 99} {
t.Run("exit_"+strconv.Itoa(code), func(t *testing.T) {
g := Goal{ID: "g", Check: "exit " + strconv.Itoa(code), Weight: 1}
m := MeasureOne(g, time.Second)
if m.Result != resultFail {
t.Errorf("exit %d: Result = %q, want %q", code, m.Result, resultFail)
}
})
}
}
func TestIsSkipExit(t *testing.T) {
tests := []struct {
name string
err error
want bool
}{
{name: "nil", err: nil, want: false},
{name: "non-exit-error", err: errors.New("boom"), want: false},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
if got := isSkipExit(tt.err); got != tt.want {
t.Errorf("isSkipExit(%v) = %v, want %v", tt.err, got, tt.want)
}
})
}
}
func TestComputeSummary_CodeDrivenAndRuntimeArtifactSplit(t *testing.T) {
// Mixed corpus: two code-driven (one pass, one fail), two
// runtime-artifact (one pass, one fail), one skip in each lane.
summary := computeSummary([]Measurement{
{Result: resultPass, Weight: 6}, // code-driven pass
{Result: resultFail, Weight: 8}, // code-driven fail
{Result: resultSkip, Weight: 3}, // code-driven skip
{Result: resultPass, Weight: 4, Tags: []string{"runtime-artifact"}}, // RA pass
{Result: resultFail, Weight: 4, Tags: []string{"runtime-artifact"}}, // RA fail
})
// Headline (raw): 2 pass (6+4) / 2 fail (8+4) / 1 skip (3, excluded);
// weighted 10 / 22 = 45.45%
if summary.Total != 5 || summary.Passing != 2 || summary.Failing != 2 || summary.Skipped != 1 {
t.Fatalf("headline counts: %+v", summary)
}
wantScore := 100.0 * 10 / 22
if got := summary.Score; got < wantScore-0.01 || got > wantScore+0.01 {
t.Errorf("Score = %f, want ~%f", got, wantScore)
}
// Code-driven: 1 pass / 1 fail / 1 skip; weighted 6 / 14 = 42.86%
if summary.CodeDrivenTotal != 3 || summary.CodeDrivenPassing != 1 || summary.CodeDrivenFailing != 1 || summary.CodeDrivenSkipped != 1 {
t.Fatalf("code-driven counts: %+v", summary)
}
wantCode := 100.0 * 6 / 14
if got := summary.CodeDrivenScore; got < wantCode-0.01 || got > wantCode+0.01 {
t.Errorf("CodeDrivenScore = %f, want ~%f", got, wantCode)
}
// Runtime-artifact: 1 pass / 1 fail / 0 skip
if summary.RuntimeArtifactTotal != 2 || summary.RuntimeArtifactPassing != 1 || summary.RuntimeArtifactFailing != 1 {
t.Errorf("runtime-artifact counts: %+v", summary)
}
}
func TestComputeSummary_AllRuntimeArtifact_CodeDrivenScoreZeroDenominator(t *testing.T) {
// When every goal is runtime-artifact, the code-driven slice is empty.
// CodeDrivenTotal=0 must keep CodeDrivenScore at 0 (no NaN, no panic).
summary := computeSummary([]Measurement{
{Result: resultPass, Weight: 4, Tags: []string{"runtime-artifact"}},
{Result: resultFail, Weight: 4, Tags: []string{"runtime-artifact"}},
})
if summary.CodeDrivenTotal != 0 {
t.Errorf("CodeDrivenTotal = %d, want 0", summary.CodeDrivenTotal)
}
if summary.CodeDrivenScore != 0 {
t.Errorf("CodeDrivenScore = %f, want 0 (empty denominator)", summary.CodeDrivenScore)
}
}
func TestIsRuntimeArtifact(t *testing.T) {
tests := []struct {
name string
tags []string
want bool
}{
{name: "nil_tags", tags: nil, want: false},
{name: "empty_tags", tags: []string{}, want: false},
{name: "exact_match", tags: []string{"runtime-artifact"}, want: true},
{name: "case_insensitive", tags: []string{"Runtime-Artifact"}, want: true},
{name: "with_whitespace", tags: []string{" runtime-artifact "}, want: true},
{name: "alongside_other_tags", tags: []string{"long-cycle", "runtime-artifact"}, want: true},
{name: "no_match", tags: []string{"long-cycle", "corpus-state"}, want: false},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
if got := isRuntimeArtifact(tt.tags); got != tt.want {
t.Errorf("isRuntimeArtifact(%v) = %v, want %v", tt.tags, got, tt.want)
}
})
}
}