Some checks failed
ci / build (push) Has been cancelled
Completes FLUID-WP-0007. The seven minimal-conformance requirements and the mechanically checkable architectural invariants are asserted as tests rather than claimed in a README, because a conformance claim nobody re-checks is one that quietly stops being true. Only the checkable subset of the invariants is asserted; pretending a test can settle the rest would be worse than leaving them to review. TestFirstVerticalSlice runs all eleven steps of Blueprint 50 with no human steps: two revisions, explicit routing, telemetry, a cohort dimension, detected pressure, a hypothesis, a candidate, a 90/10 experiment, fitness comparison, promotion, and a complete audit trail. Requests per completed task fall from 5.65 to 1.00 against a 1.20 target. A companion test runs the loop twice and requires the same verdict, since a loop whose conclusion depended on run order would be measuring the harness rather than the interface. The failure-containment matrix covers Blueprint 34 directly: the data plane keeps serving with the evidence store closed, with telemetry wedged against a sink that never returns, after a failed build, after an experiment rollback, and with the adaptive concurrency limit saturated. Fixes a real bug the suite exposed. Drain closed the emitter outright, so every request after the first flush emitted into a dead emitter and was silently lost -- the kind of fault that makes a later measurement quietly wrong rather than loudly broken. Emitter.Flush now waits for delivery without stopping it. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_014KmVxhJ35tCo7rE7UnLwWu Assistant: claude-code Assistant-Model: opus Assistant-Process: 1116572@bnt-lap001 Assistant-Session: 8ba9bb93-a72a-4883-b189-2499cce5c400
222 lines
6.5 KiB
Go
222 lines
6.5 KiB
Go
package suite
|
|
|
|
import (
|
|
"context"
|
|
"strings"
|
|
"testing"
|
|
|
|
"github.com/tegwick/fluid-core/internal/contract"
|
|
"github.com/tegwick/fluid-core/internal/fitness"
|
|
"github.com/tegwick/fluid-core/internal/intent"
|
|
"github.com/tegwick/fluid-core/internal/policy"
|
|
"github.com/tegwick/fluid-core/internal/promotion"
|
|
)
|
|
|
|
// TestFirstVerticalSlice is the whole point of the conformance suite.
|
|
//
|
|
// ArchitectureBlueprint.md section 50 lists eleven things the first FLUID
|
|
// deployment must demonstrate, and says plainly what success means:
|
|
//
|
|
// The first success criterion is not autonomous coding.
|
|
// It is proving that the revision-experiment-fitness loop works cleanly
|
|
// and safely.
|
|
//
|
|
// This test runs all eleven with no human steps.
|
|
func TestFirstVerticalSlice(t *testing.T) {
|
|
h := New(t)
|
|
loop := RunFullLoop(t, h)
|
|
|
|
// 1. Two deterministic revisions.
|
|
for _, id := range []contract.RevisionID{"R-1", "R-2"} {
|
|
if _, err := h.Registry.Revision(id); err != nil {
|
|
t.Fatalf("revision %s is not published: %v", id, err)
|
|
}
|
|
}
|
|
|
|
// 2. Explicit revision routing.
|
|
policy, err := h.Registry.Policy()
|
|
if err != nil {
|
|
t.Fatalf("no routing policy: %v", err)
|
|
}
|
|
if policy.DefaultRevision == "" {
|
|
t.Error("the routing policy names no default revision")
|
|
}
|
|
|
|
// 3. Telemetry.
|
|
telemetry := h.Telemetry()
|
|
if len(telemetry) == 0 {
|
|
t.Fatal("no telemetry recorded")
|
|
}
|
|
|
|
// 4. One consumer cohort dimension.
|
|
var sawCohort bool
|
|
for _, ev := range telemetry {
|
|
if ev.Cohort != nil && *ev.Cohort != "" {
|
|
sawCohort = true
|
|
break
|
|
}
|
|
}
|
|
if !sawCohort {
|
|
t.Error("no telemetry carries a cohort assignment")
|
|
}
|
|
|
|
// 5. Pressure detection.
|
|
if loop.Pressure == "" {
|
|
t.Fatal("no pressure was detected")
|
|
}
|
|
|
|
// 6. A hypothesis, linked to that pressure.
|
|
hyp, err := h.Hypotheses.Get(context.Background(), loop.Hypothesis)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if hyp.Explanation.Claim == "" {
|
|
t.Error("the hypothesis offers no explanation")
|
|
}
|
|
|
|
// 7. A candidate revision, claimed by that hypothesis.
|
|
var claimsCandidate bool
|
|
for _, r := range hyp.CandidateRevisionRefs {
|
|
if r == loop.Candidate {
|
|
claimsCandidate = true
|
|
}
|
|
}
|
|
if !claimsCandidate {
|
|
t.Errorf("the hypothesis does not claim %s", loop.Candidate)
|
|
}
|
|
|
|
// 8. A controlled 90/10 experiment.
|
|
exp, err := h.Experiments.Get(context.Background(), loop.Experiment)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if exp.Allocation["control"] != 0.9 || exp.Allocation["candidate"] != 0.1 {
|
|
t.Errorf("allocation was %v, want a 90/10 split", exp.Allocation)
|
|
}
|
|
|
|
// 9. Fitness comparison.
|
|
if loop.Evaluation.Verdict != fitness.VerdictSucceeded {
|
|
t.Fatalf("verdict %s: %v", loop.Evaluation.Verdict, loop.Evaluation.Reasons)
|
|
}
|
|
primary := loop.Evaluation.PrimaryResults()
|
|
if len(primary) != 1 {
|
|
t.Fatalf("expected one primary metric, got %d", len(primary))
|
|
}
|
|
// The candidate should show the reduction the hypothesis predicted.
|
|
if primary[0].Current >= primary[0].Baseline {
|
|
t.Errorf("requests per task did not fall: %v -> %v",
|
|
primary[0].Baseline, primary[0].Current)
|
|
}
|
|
t.Logf("requests per completed task: %.2f -> %.2f (target %.2f)",
|
|
primary[0].Baseline, primary[0].Current, *primary[0].Target)
|
|
|
|
// 10. Promotion.
|
|
if loop.Decision.Outcome != promotion.Promote {
|
|
t.Errorf("decision was %s, want PROMOTE", loop.Decision.Outcome)
|
|
}
|
|
if loop.Decision.Override {
|
|
t.Error("the promotion was an override; the evidence should have carried it")
|
|
}
|
|
|
|
// 11. A complete audit trail.
|
|
//
|
|
// Every stage of the loop must have left a reconstructable record. This is
|
|
// the assertion that would fail first if any part of the framework started
|
|
// deciding things without saying so.
|
|
events := h.Events(string(loop.Candidate))
|
|
required := map[string]bool{
|
|
"SIGN_PASSED": false,
|
|
"POLICY_CHECK_PASSED": false,
|
|
"INTENT_BOUND": false,
|
|
"REVISION_PUBLISHED": false,
|
|
"PROMOTION_DECIDED_PROMOTE": false,
|
|
}
|
|
for _, ev := range events {
|
|
if _, tracked := required[ev.EventType]; tracked {
|
|
required[ev.EventType] = true
|
|
}
|
|
}
|
|
for name, seen := range required {
|
|
if !seen {
|
|
t.Errorf("the audit trail is missing %s", name)
|
|
}
|
|
}
|
|
|
|
// The pressure, hypothesis and experiment each left their own trail.
|
|
for _, entity := range []string{
|
|
string(loop.Pressure), string(loop.Hypothesis), string(loop.Experiment),
|
|
} {
|
|
if len(h.Events(entity)) == 0 {
|
|
t.Errorf("%s left no audit events", entity)
|
|
}
|
|
}
|
|
|
|
t.Logf("slice complete: %s -> %s -> %s -> %s (%s)",
|
|
loop.Pressure, loop.Hypothesis, loop.Experiment, loop.Candidate, loop.Decision.Outcome)
|
|
}
|
|
|
|
// TestSliceIsReproducible runs the loop twice over fresh state and requires the
|
|
// same conclusion both times.
|
|
//
|
|
// A loop whose verdict depended on run order would be measuring the harness
|
|
// rather than the interface.
|
|
func TestSliceIsReproducible(t *testing.T) {
|
|
var verdicts []fitness.Verdict
|
|
var ratios []float64
|
|
|
|
for i := 0; i < 2; i++ {
|
|
h := New(t)
|
|
loop := RunFullLoop(t, h)
|
|
verdicts = append(verdicts, loop.Evaluation.Verdict)
|
|
ratios = append(ratios, loop.Evaluation.PrimaryResults()[0].Current)
|
|
}
|
|
|
|
if verdicts[0] != verdicts[1] {
|
|
t.Errorf("verdict varied between runs: %s then %s", verdicts[0], verdicts[1])
|
|
}
|
|
if ratios[0] != ratios[1] {
|
|
t.Errorf("measured ratio varied between runs: %v then %v", ratios[0], ratios[1])
|
|
}
|
|
}
|
|
|
|
// TestSliceRefusesPromotionWithoutEvidence confirms the loop cannot be
|
|
// short-circuited: the same candidate, promoted without its experiment, is
|
|
// refused.
|
|
func TestSliceRefusesPromotionWithoutEvidence(t *testing.T) {
|
|
h := New(t)
|
|
|
|
descriptor, err := h.Registry.Revision("R-2")
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
governing, err := h.Intents.GoverningIntent(context.Background(), "R-2")
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
limits := policy.DefaultLimits()
|
|
limits.AllowedAdaptationClasses = append(limits.AllowedAdaptationClasses,
|
|
contract.AdaptationClassContract)
|
|
limits.RequiredMode = intent.ModeExperimental
|
|
|
|
_, err = promotion.NewController(h.Store, policy.NewGate(limits)).Decide(context.Background(), promotion.Request{
|
|
Revision: "R-2",
|
|
Outcome: promotion.Promote,
|
|
Reason: "it looks better",
|
|
Actor: Operator,
|
|
GateInput: &policy.Input{
|
|
Descriptor: descriptor,
|
|
GoverningMode: governing.Mode,
|
|
AdaptationClasses: []contract.AdaptationClass{contract.AdaptationClassContract},
|
|
RequestedTrafficShare: 0.2,
|
|
Approved: true,
|
|
ApprovedBy: &Operator,
|
|
},
|
|
})
|
|
if err == nil {
|
|
t.Fatal("a candidate was promoted with no fitness evidence")
|
|
}
|
|
if !strings.Contains(err.Error(), "fitness") {
|
|
t.Errorf("refusal does not cite missing evidence: %v", err)
|
|
}
|
|
}
|