fluid-core/conformance/suite/slice_test.go
tegwick 55363905bc
Some checks failed
ci / build (push) Has been cancelled
Add the conformance suite, echo fixture and integration guide
Completes FLUID-WP-0007. The seven minimal-conformance requirements and
the mechanically checkable architectural invariants are asserted as
tests rather than claimed in a README, because a conformance claim
nobody re-checks is one that quietly stops being true. Only the
checkable subset of the invariants is asserted; pretending a test can
settle the rest would be worse than leaving them to review.

TestFirstVerticalSlice runs all eleven steps of Blueprint 50 with no
human steps: two revisions, explicit routing, telemetry, a cohort
dimension, detected pressure, a hypothesis, a candidate, a 90/10
experiment, fitness comparison, promotion, and a complete audit trail.
Requests per completed task fall from 5.65 to 1.00 against a 1.20
target. A companion test runs the loop twice and requires the same
verdict, since a loop whose conclusion depended on run order would be
measuring the harness rather than the interface.

The failure-containment matrix covers Blueprint 34 directly: the data
plane keeps serving with the evidence store closed, with telemetry
wedged against a sink that never returns, after a failed build, after an
experiment rollback, and with the adaptive concurrency limit saturated.

Fixes a real bug the suite exposed. Drain closed the emitter outright,
so every request after the first flush emitted into a dead emitter and
was silently lost -- the kind of fault that makes a later measurement
quietly wrong rather than loudly broken. Emitter.Flush now waits for
delivery without stopping it.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_014KmVxhJ35tCo7rE7UnLwWu

Assistant: claude-code
Assistant-Model: opus
Assistant-Process: 1116572@bnt-lap001
Assistant-Session: 8ba9bb93-a72a-4883-b189-2499cce5c400
2026-09-04 08:21:49 +02:00

222 lines
6.5 KiB
Go

package suite
import (
"context"
"strings"
"testing"
"github.com/tegwick/fluid-core/internal/contract"
"github.com/tegwick/fluid-core/internal/fitness"
"github.com/tegwick/fluid-core/internal/intent"
"github.com/tegwick/fluid-core/internal/policy"
"github.com/tegwick/fluid-core/internal/promotion"
)
// TestFirstVerticalSlice is the whole point of the conformance suite.
//
// ArchitectureBlueprint.md section 50 lists eleven things the first FLUID
// deployment must demonstrate, and says plainly what success means:
//
// The first success criterion is not autonomous coding.
// It is proving that the revision-experiment-fitness loop works cleanly
// and safely.
//
// This test runs all eleven with no human steps.
func TestFirstVerticalSlice(t *testing.T) {
h := New(t)
loop := RunFullLoop(t, h)
// 1. Two deterministic revisions.
for _, id := range []contract.RevisionID{"R-1", "R-2"} {
if _, err := h.Registry.Revision(id); err != nil {
t.Fatalf("revision %s is not published: %v", id, err)
}
}
// 2. Explicit revision routing.
policy, err := h.Registry.Policy()
if err != nil {
t.Fatalf("no routing policy: %v", err)
}
if policy.DefaultRevision == "" {
t.Error("the routing policy names no default revision")
}
// 3. Telemetry.
telemetry := h.Telemetry()
if len(telemetry) == 0 {
t.Fatal("no telemetry recorded")
}
// 4. One consumer cohort dimension.
var sawCohort bool
for _, ev := range telemetry {
if ev.Cohort != nil && *ev.Cohort != "" {
sawCohort = true
break
}
}
if !sawCohort {
t.Error("no telemetry carries a cohort assignment")
}
// 5. Pressure detection.
if loop.Pressure == "" {
t.Fatal("no pressure was detected")
}
// 6. A hypothesis, linked to that pressure.
hyp, err := h.Hypotheses.Get(context.Background(), loop.Hypothesis)
if err != nil {
t.Fatal(err)
}
if hyp.Explanation.Claim == "" {
t.Error("the hypothesis offers no explanation")
}
// 7. A candidate revision, claimed by that hypothesis.
var claimsCandidate bool
for _, r := range hyp.CandidateRevisionRefs {
if r == loop.Candidate {
claimsCandidate = true
}
}
if !claimsCandidate {
t.Errorf("the hypothesis does not claim %s", loop.Candidate)
}
// 8. A controlled 90/10 experiment.
exp, err := h.Experiments.Get(context.Background(), loop.Experiment)
if err != nil {
t.Fatal(err)
}
if exp.Allocation["control"] != 0.9 || exp.Allocation["candidate"] != 0.1 {
t.Errorf("allocation was %v, want a 90/10 split", exp.Allocation)
}
// 9. Fitness comparison.
if loop.Evaluation.Verdict != fitness.VerdictSucceeded {
t.Fatalf("verdict %s: %v", loop.Evaluation.Verdict, loop.Evaluation.Reasons)
}
primary := loop.Evaluation.PrimaryResults()
if len(primary) != 1 {
t.Fatalf("expected one primary metric, got %d", len(primary))
}
// The candidate should show the reduction the hypothesis predicted.
if primary[0].Current >= primary[0].Baseline {
t.Errorf("requests per task did not fall: %v -> %v",
primary[0].Baseline, primary[0].Current)
}
t.Logf("requests per completed task: %.2f -> %.2f (target %.2f)",
primary[0].Baseline, primary[0].Current, *primary[0].Target)
// 10. Promotion.
if loop.Decision.Outcome != promotion.Promote {
t.Errorf("decision was %s, want PROMOTE", loop.Decision.Outcome)
}
if loop.Decision.Override {
t.Error("the promotion was an override; the evidence should have carried it")
}
// 11. A complete audit trail.
//
// Every stage of the loop must have left a reconstructable record. This is
// the assertion that would fail first if any part of the framework started
// deciding things without saying so.
events := h.Events(string(loop.Candidate))
required := map[string]bool{
"SIGN_PASSED": false,
"POLICY_CHECK_PASSED": false,
"INTENT_BOUND": false,
"REVISION_PUBLISHED": false,
"PROMOTION_DECIDED_PROMOTE": false,
}
for _, ev := range events {
if _, tracked := required[ev.EventType]; tracked {
required[ev.EventType] = true
}
}
for name, seen := range required {
if !seen {
t.Errorf("the audit trail is missing %s", name)
}
}
// The pressure, hypothesis and experiment each left their own trail.
for _, entity := range []string{
string(loop.Pressure), string(loop.Hypothesis), string(loop.Experiment),
} {
if len(h.Events(entity)) == 0 {
t.Errorf("%s left no audit events", entity)
}
}
t.Logf("slice complete: %s -> %s -> %s -> %s (%s)",
loop.Pressure, loop.Hypothesis, loop.Experiment, loop.Candidate, loop.Decision.Outcome)
}
// TestSliceIsReproducible runs the loop twice over fresh state and requires the
// same conclusion both times.
//
// A loop whose verdict depended on run order would be measuring the harness
// rather than the interface.
func TestSliceIsReproducible(t *testing.T) {
var verdicts []fitness.Verdict
var ratios []float64
for i := 0; i < 2; i++ {
h := New(t)
loop := RunFullLoop(t, h)
verdicts = append(verdicts, loop.Evaluation.Verdict)
ratios = append(ratios, loop.Evaluation.PrimaryResults()[0].Current)
}
if verdicts[0] != verdicts[1] {
t.Errorf("verdict varied between runs: %s then %s", verdicts[0], verdicts[1])
}
if ratios[0] != ratios[1] {
t.Errorf("measured ratio varied between runs: %v then %v", ratios[0], ratios[1])
}
}
// TestSliceRefusesPromotionWithoutEvidence confirms the loop cannot be
// short-circuited: the same candidate, promoted without its experiment, is
// refused.
func TestSliceRefusesPromotionWithoutEvidence(t *testing.T) {
h := New(t)
descriptor, err := h.Registry.Revision("R-2")
if err != nil {
t.Fatal(err)
}
governing, err := h.Intents.GoverningIntent(context.Background(), "R-2")
if err != nil {
t.Fatal(err)
}
limits := policy.DefaultLimits()
limits.AllowedAdaptationClasses = append(limits.AllowedAdaptationClasses,
contract.AdaptationClassContract)
limits.RequiredMode = intent.ModeExperimental
_, err = promotion.NewController(h.Store, policy.NewGate(limits)).Decide(context.Background(), promotion.Request{
Revision: "R-2",
Outcome: promotion.Promote,
Reason: "it looks better",
Actor: Operator,
GateInput: &policy.Input{
Descriptor: descriptor,
GoverningMode: governing.Mode,
AdaptationClasses: []contract.AdaptationClass{contract.AdaptationClassContract},
RequestedTrafficShare: 0.2,
Approved: true,
ApprovedBy: &Operator,
},
})
if err == nil {
t.Fatal("a candidate was promoted with no fitness evidence")
}
if !strings.Contains(err.Error(), "fitness") {
t.Errorf("refusal does not cite missing evidence: %v", err)
}
}