Add the conformance suite, echo fixture and integration guide
Some checks failed
ci / build (push) Has been cancelled
Some checks failed
ci / build (push) Has been cancelled
Completes FLUID-WP-0007. The seven minimal-conformance requirements and the mechanically checkable architectural invariants are asserted as tests rather than claimed in a README, because a conformance claim nobody re-checks is one that quietly stops being true. Only the checkable subset of the invariants is asserted; pretending a test can settle the rest would be worse than leaving them to review. TestFirstVerticalSlice runs all eleven steps of Blueprint 50 with no human steps: two revisions, explicit routing, telemetry, a cohort dimension, detected pressure, a hypothesis, a candidate, a 90/10 experiment, fitness comparison, promotion, and a complete audit trail. Requests per completed task fall from 5.65 to 1.00 against a 1.20 target. A companion test runs the loop twice and requires the same verdict, since a loop whose conclusion depended on run order would be measuring the harness rather than the interface. The failure-containment matrix covers Blueprint 34 directly: the data plane keeps serving with the evidence store closed, with telemetry wedged against a sink that never returns, after a failed build, after an experiment rollback, and with the adaptive concurrency limit saturated. Fixes a real bug the suite exposed. Drain closed the emitter outright, so every request after the first flush emitted into a dead emitter and was silently lost -- the kind of fault that makes a later measurement quietly wrong rather than loudly broken. Emitter.Flush now waits for delivery without stopping it. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_014KmVxhJ35tCo7rE7UnLwWu Assistant: claude-code Assistant-Model: opus Assistant-Process: 1116572@bnt-lap001 Assistant-Session: 8ba9bb93-a72a-4883-b189-2499cce5c400
This commit is contained in:
parent
61d8d8cabe
commit
55363905bc
16 changed files with 1885 additions and 23 deletions
222
conformance/suite/slice_test.go
Normal file
222
conformance/suite/slice_test.go
Normal file
|
|
@ -0,0 +1,222 @@
|
|||
package suite
|
||||
|
||||
import (
|
||||
"context"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/tegwick/fluid-core/internal/contract"
|
||||
"github.com/tegwick/fluid-core/internal/fitness"
|
||||
"github.com/tegwick/fluid-core/internal/intent"
|
||||
"github.com/tegwick/fluid-core/internal/policy"
|
||||
"github.com/tegwick/fluid-core/internal/promotion"
|
||||
)
|
||||
|
||||
// TestFirstVerticalSlice is the whole point of the conformance suite.
|
||||
//
|
||||
// ArchitectureBlueprint.md section 50 lists eleven things the first FLUID
|
||||
// deployment must demonstrate, and says plainly what success means:
|
||||
//
|
||||
// The first success criterion is not autonomous coding.
|
||||
// It is proving that the revision-experiment-fitness loop works cleanly
|
||||
// and safely.
|
||||
//
|
||||
// This test runs all eleven with no human steps.
|
||||
func TestFirstVerticalSlice(t *testing.T) {
|
||||
h := New(t)
|
||||
loop := RunFullLoop(t, h)
|
||||
|
||||
// 1. Two deterministic revisions.
|
||||
for _, id := range []contract.RevisionID{"R-1", "R-2"} {
|
||||
if _, err := h.Registry.Revision(id); err != nil {
|
||||
t.Fatalf("revision %s is not published: %v", id, err)
|
||||
}
|
||||
}
|
||||
|
||||
// 2. Explicit revision routing.
|
||||
policy, err := h.Registry.Policy()
|
||||
if err != nil {
|
||||
t.Fatalf("no routing policy: %v", err)
|
||||
}
|
||||
if policy.DefaultRevision == "" {
|
||||
t.Error("the routing policy names no default revision")
|
||||
}
|
||||
|
||||
// 3. Telemetry.
|
||||
telemetry := h.Telemetry()
|
||||
if len(telemetry) == 0 {
|
||||
t.Fatal("no telemetry recorded")
|
||||
}
|
||||
|
||||
// 4. One consumer cohort dimension.
|
||||
var sawCohort bool
|
||||
for _, ev := range telemetry {
|
||||
if ev.Cohort != nil && *ev.Cohort != "" {
|
||||
sawCohort = true
|
||||
break
|
||||
}
|
||||
}
|
||||
if !sawCohort {
|
||||
t.Error("no telemetry carries a cohort assignment")
|
||||
}
|
||||
|
||||
// 5. Pressure detection.
|
||||
if loop.Pressure == "" {
|
||||
t.Fatal("no pressure was detected")
|
||||
}
|
||||
|
||||
// 6. A hypothesis, linked to that pressure.
|
||||
hyp, err := h.Hypotheses.Get(context.Background(), loop.Hypothesis)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if hyp.Explanation.Claim == "" {
|
||||
t.Error("the hypothesis offers no explanation")
|
||||
}
|
||||
|
||||
// 7. A candidate revision, claimed by that hypothesis.
|
||||
var claimsCandidate bool
|
||||
for _, r := range hyp.CandidateRevisionRefs {
|
||||
if r == loop.Candidate {
|
||||
claimsCandidate = true
|
||||
}
|
||||
}
|
||||
if !claimsCandidate {
|
||||
t.Errorf("the hypothesis does not claim %s", loop.Candidate)
|
||||
}
|
||||
|
||||
// 8. A controlled 90/10 experiment.
|
||||
exp, err := h.Experiments.Get(context.Background(), loop.Experiment)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if exp.Allocation["control"] != 0.9 || exp.Allocation["candidate"] != 0.1 {
|
||||
t.Errorf("allocation was %v, want a 90/10 split", exp.Allocation)
|
||||
}
|
||||
|
||||
// 9. Fitness comparison.
|
||||
if loop.Evaluation.Verdict != fitness.VerdictSucceeded {
|
||||
t.Fatalf("verdict %s: %v", loop.Evaluation.Verdict, loop.Evaluation.Reasons)
|
||||
}
|
||||
primary := loop.Evaluation.PrimaryResults()
|
||||
if len(primary) != 1 {
|
||||
t.Fatalf("expected one primary metric, got %d", len(primary))
|
||||
}
|
||||
// The candidate should show the reduction the hypothesis predicted.
|
||||
if primary[0].Current >= primary[0].Baseline {
|
||||
t.Errorf("requests per task did not fall: %v -> %v",
|
||||
primary[0].Baseline, primary[0].Current)
|
||||
}
|
||||
t.Logf("requests per completed task: %.2f -> %.2f (target %.2f)",
|
||||
primary[0].Baseline, primary[0].Current, *primary[0].Target)
|
||||
|
||||
// 10. Promotion.
|
||||
if loop.Decision.Outcome != promotion.Promote {
|
||||
t.Errorf("decision was %s, want PROMOTE", loop.Decision.Outcome)
|
||||
}
|
||||
if loop.Decision.Override {
|
||||
t.Error("the promotion was an override; the evidence should have carried it")
|
||||
}
|
||||
|
||||
// 11. A complete audit trail.
|
||||
//
|
||||
// Every stage of the loop must have left a reconstructable record. This is
|
||||
// the assertion that would fail first if any part of the framework started
|
||||
// deciding things without saying so.
|
||||
events := h.Events(string(loop.Candidate))
|
||||
required := map[string]bool{
|
||||
"SIGN_PASSED": false,
|
||||
"POLICY_CHECK_PASSED": false,
|
||||
"INTENT_BOUND": false,
|
||||
"REVISION_PUBLISHED": false,
|
||||
"PROMOTION_DECIDED_PROMOTE": false,
|
||||
}
|
||||
for _, ev := range events {
|
||||
if _, tracked := required[ev.EventType]; tracked {
|
||||
required[ev.EventType] = true
|
||||
}
|
||||
}
|
||||
for name, seen := range required {
|
||||
if !seen {
|
||||
t.Errorf("the audit trail is missing %s", name)
|
||||
}
|
||||
}
|
||||
|
||||
// The pressure, hypothesis and experiment each left their own trail.
|
||||
for _, entity := range []string{
|
||||
string(loop.Pressure), string(loop.Hypothesis), string(loop.Experiment),
|
||||
} {
|
||||
if len(h.Events(entity)) == 0 {
|
||||
t.Errorf("%s left no audit events", entity)
|
||||
}
|
||||
}
|
||||
|
||||
t.Logf("slice complete: %s -> %s -> %s -> %s (%s)",
|
||||
loop.Pressure, loop.Hypothesis, loop.Experiment, loop.Candidate, loop.Decision.Outcome)
|
||||
}
|
||||
|
||||
// TestSliceIsReproducible runs the loop twice over fresh state and requires the
|
||||
// same conclusion both times.
|
||||
//
|
||||
// A loop whose verdict depended on run order would be measuring the harness
|
||||
// rather than the interface.
|
||||
func TestSliceIsReproducible(t *testing.T) {
|
||||
var verdicts []fitness.Verdict
|
||||
var ratios []float64
|
||||
|
||||
for i := 0; i < 2; i++ {
|
||||
h := New(t)
|
||||
loop := RunFullLoop(t, h)
|
||||
verdicts = append(verdicts, loop.Evaluation.Verdict)
|
||||
ratios = append(ratios, loop.Evaluation.PrimaryResults()[0].Current)
|
||||
}
|
||||
|
||||
if verdicts[0] != verdicts[1] {
|
||||
t.Errorf("verdict varied between runs: %s then %s", verdicts[0], verdicts[1])
|
||||
}
|
||||
if ratios[0] != ratios[1] {
|
||||
t.Errorf("measured ratio varied between runs: %v then %v", ratios[0], ratios[1])
|
||||
}
|
||||
}
|
||||
|
||||
// TestSliceRefusesPromotionWithoutEvidence confirms the loop cannot be
|
||||
// short-circuited: the same candidate, promoted without its experiment, is
|
||||
// refused.
|
||||
func TestSliceRefusesPromotionWithoutEvidence(t *testing.T) {
|
||||
h := New(t)
|
||||
|
||||
descriptor, err := h.Registry.Revision("R-2")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
governing, err := h.Intents.GoverningIntent(context.Background(), "R-2")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
limits := policy.DefaultLimits()
|
||||
limits.AllowedAdaptationClasses = append(limits.AllowedAdaptationClasses,
|
||||
contract.AdaptationClassContract)
|
||||
limits.RequiredMode = intent.ModeExperimental
|
||||
|
||||
_, err = promotion.NewController(h.Store, policy.NewGate(limits)).Decide(context.Background(), promotion.Request{
|
||||
Revision: "R-2",
|
||||
Outcome: promotion.Promote,
|
||||
Reason: "it looks better",
|
||||
Actor: Operator,
|
||||
GateInput: &policy.Input{
|
||||
Descriptor: descriptor,
|
||||
GoverningMode: governing.Mode,
|
||||
AdaptationClasses: []contract.AdaptationClass{contract.AdaptationClassContract},
|
||||
RequestedTrafficShare: 0.2,
|
||||
Approved: true,
|
||||
ApprovedBy: &Operator,
|
||||
},
|
||||
})
|
||||
if err == nil {
|
||||
t.Fatal("a candidate was promoted with no fitness evidence")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "fitness") {
|
||||
t.Errorf("refusal does not cite missing evidence: %v", err)
|
||||
}
|
||||
}
|
||||
Loading…
Add table
Add a link
Reference in a new issue