package observation import ( "fmt" "sort" "time" "github.com/tegwick/fluid-core/internal/contract" ) // Finding is a classified observation, before it becomes a pressure record. // // A finding carries its evidence and its reasoning. FluidAPIStandards.md // section 13 is explicit that pressure is evidence and not truth, so a finding // that cannot show its working is not usable: the whole point of separating // observation from explanation (schema doc section 18) is that a later reader // can disagree with the interpretation while still trusting the observation. type Finding struct { Class contract.PressureClass `json:"class"` Summary string `json:"summary"` Cohorts []contract.CohortID `json:"cohorts"` Occurrences int `json:"occurrences"` Consumers int `json:"independent_consumers"` Severity float64 `json:"severity"` Confidence float64 `json:"confidence"` Evidence []contract.EvidenceRef `json:"evidence_refs"` FirstSeen time.Time `json:"first_seen"` LastSeen time.Time `json:"last_seen"` // Fingerprint identifies the same finding across runs, so the registry can // deduplicate rather than accumulating one record per analysis pass. Fingerprint string `json:"fingerprint"` } // ClassifierOptions tunes the deterministic thresholds. // // They are explicit configuration because Blueprint section 54 lists pressure // detection as an open question: which signals produce the best improvement // candidates is meant to be learned by operating the system, not fixed now. type ClassifierOptions struct { // RepeatThreshold is how many identical calls in one chain count as // compensating for a missing capability rather than ordinary paging. RepeatThreshold int // MinConsumers is the independent-consumer floor for any finding. MinConsumers int // ErrorRateThreshold is the share of a route's calls that must fail before // implementation failure is claimed. ErrorRateThreshold float64 } // DefaultClassifierOptions returns workable starting thresholds. func DefaultClassifierOptions() ClassifierOptions { return ClassifierOptions{ RepeatThreshold: 3, MinConsumers: 2, ErrorRateThreshold: 0.20, } } // Classifier maps evidence onto the ten standard pressure classes. // // It is deterministic on purpose. FluidAPIStandards.md section 13 allows // heuristics, statistics or agentic reasoning here, but a first implementation // that reaches for a model cannot be audited, and the classifier's output is // what a hypothesis will later cite as its observation. type Classifier struct { opts ClassifierOptions analyzer *TopologyAnalyzer } // NewClassifier returns a classifier. func NewClassifier(opts ClassifierOptions, analyzer *TopologyAnalyzer) *Classifier { if analyzer == nil { analyzer = NewTopologyAnalyzer() } return &Classifier{opts: opts, analyzer: analyzer} } // Classify examines telemetry and returns findings. func (c *Classifier) Classify(events []contract.FluidTelemetry) []Finding { var findings []Finding findings = append(findings, c.fromPatterns(events)...) findings = append(findings, c.fromErrors(events)...) // Highest severity first, with a stable tiebreak so output is diffable. sort.Slice(findings, func(i, j int) bool { if findings[i].Severity != findings[j].Severity { return findings[i].Severity > findings[j].Severity } return findings[i].Fingerprint < findings[j].Fingerprint }) return findings } // fromPatterns derives findings from interaction topology. func (c *Classifier) fromPatterns(events []contract.FluidTelemetry) []Finding { var out []Finding for _, p := range c.analyzer.Patterns(events) { if p.Consumers < c.opts.MinConsumers { continue } switch { case p.MaxRepeats >= c.opts.RepeatThreshold: // Independent consumers calling one route repeatedly inside a // single task is the classic shape of an interface that makes them // assemble something it could have handed them. out = append(out, Finding{ Class: contract.PressureClassSuccessfulButInefficientUsage, Summary: fmt.Sprintf( "%d independent consumers call %s up to %d times within one interaction", p.Consumers, p.RepeatedStep, p.MaxRepeats), Cohorts: p.Cohorts, Occurrences: p.Count, Consumers: p.Consumers, Severity: severityFrom(p.Consumers, p.Count, float64(p.MaxRepeats)/10), Confidence: confidenceFrom(p.Consumers, p.Count), Evidence: []contract.EvidenceRef{contract.EvidenceRef("topology:" + p.Signature)}, FirstSeen: p.FirstSeen, LastSeen: p.LastSeen, Fingerprint: fingerprint("inefficient", p.RepeatedStep), }) case p.RecoveredError == contract.FluidTelemetryErrorClassValidation, p.RecoveredError == contract.FluidTelemetryErrorClassUnknownField, p.RecoveredError == contract.FluidTelemetryErrorClassUnsupportedParameter: // Consumers who fail, correct themselves, and succeed have // understood the interface eventually. That is a documentation and // discoverability problem, not a capability gap. out = append(out, Finding{ Class: contract.PressureClassRecoverableMisunderstanding, Summary: fmt.Sprintf( "%d independent consumers recover from a %s error within the same interaction", p.Consumers, p.RecoveredError), Cohorts: p.Cohorts, Occurrences: p.Count, Consumers: p.Consumers, Severity: severityFrom(p.Consumers, p.Count, 0.1), Confidence: confidenceFrom(p.Consumers, p.Count), Evidence: []contract.EvidenceRef{contract.EvidenceRef("topology:" + p.Signature)}, FirstSeen: p.FirstSeen, LastSeen: p.LastSeen, Fingerprint: fingerprint("recoverable", string(p.RecoveredError)), }) } } return out } // routeStats accumulates per-route outcomes. type routeStats struct { total int errors map[contract.FluidTelemetryErrorClass]int consumers map[string]struct{} cohorts map[contract.CohortID]struct{} first time.Time last time.Time } // fromErrors derives findings from error classes on individual routes. func (c *Classifier) fromErrors(events []contract.FluidTelemetry) []Finding { byRoute := map[string]*routeStats{} for _, ev := range events { route := "(unknown)" if ev.Request != nil && ev.Request.Route != "" { route = ev.Request.Method + " " + ev.Request.Route } s, ok := byRoute[route] if !ok { s = &routeStats{ errors: map[contract.FluidTelemetryErrorClass]int{}, consumers: map[string]struct{}{}, cohorts: map[contract.CohortID]struct{}{}, first: ev.OccurredAt, last: ev.OccurredAt, } byRoute[route] = s } s.total++ if ev.ConsumerRef != "" { s.consumers[ev.ConsumerRef] = struct{}{} } if ev.Cohort != nil { s.cohorts[*ev.Cohort] = struct{}{} } if ev.OccurredAt.Before(s.first) { s.first = ev.OccurredAt } if ev.OccurredAt.After(s.last) { s.last = ev.OccurredAt } if ev.Error != nil { s.errors[ev.Error.Class]++ } } routes := make([]string, 0, len(byRoute)) for r := range byRoute { routes = append(routes, r) } sort.Strings(routes) var out []Finding for _, route := range routes { s := byRoute[route] consumers := len(s.consumers) if consumers < c.opts.MinConsumers { continue } classes := make([]contract.FluidTelemetryErrorClass, 0, len(s.errors)) for cl := range s.errors { classes = append(classes, cl) } sort.Slice(classes, func(i, j int) bool { return classes[i] < classes[j] }) for _, class := range classes { count := s.errors[class] rate := float64(count) / float64(s.total) pressure, ok := pressureForError(class) if !ok { continue } // Backend failures and timeouts are reported on any recurrence; // consumer-side errors need to be a meaningful share of traffic // before they say something about the interface rather than about // one confused client. if !alwaysReport(class) && rate < c.opts.ErrorRateThreshold { continue } cohorts := make([]contract.CohortID, 0, len(s.cohorts)) for co := range s.cohorts { cohorts = append(cohorts, co) } sort.Slice(cohorts, func(i, j int) bool { return cohorts[i] < cohorts[j] }) out = append(out, Finding{ Class: pressure, Summary: fmt.Sprintf("%s returns %s for %.0f%% of calls across %d consumers", route, class, rate*100, consumers), Cohorts: cohorts, Occurrences: count, Consumers: consumers, Severity: severityFrom(consumers, count, rate), Confidence: confidenceFrom(consumers, count), Evidence: []contract.EvidenceRef{ contract.EvidenceRef("route:" + route), contract.EvidenceRef("error:" + string(class)), }, FirstSeen: s.first, LastSeen: s.last, Fingerprint: fingerprint(string(pressure), route+"/"+string(class)), }) } } return out } // pressureForError maps a telemetry error class onto a pressure class. // // Not every error is pressure. An authorization failure usually means the // interface worked exactly as designed, so it is left unmapped rather than // dressed up as a defect. func pressureForError(class contract.FluidTelemetryErrorClass) (contract.PressureClass, bool) { switch class { case contract.FluidTelemetryErrorClassUnknownPath: return contract.PressureClassMissingInterfaceCapability, true case contract.FluidTelemetryErrorClassUnknownField, contract.FluidTelemetryErrorClassUnsupportedParameter: return contract.PressureClassRepeatedExpectationMismatch, true case contract.FluidTelemetryErrorClassValidation: return contract.PressureClassPoorDiscoverability, true case contract.FluidTelemetryErrorClassBackendFailure, contract.FluidTelemetryErrorClassTimeout: return contract.PressureClassImplementationFailure, true case contract.FluidTelemetryErrorClassPolicyRejection: return contract.PressureClassProhibitedDemand, true } return "", false } // alwaysReport marks classes worth surfacing at any rate. func alwaysReport(class contract.FluidTelemetryErrorClass) bool { switch class { case contract.FluidTelemetryErrorClassBackendFailure, contract.FluidTelemetryErrorClassTimeout: return true } return false } // severityFrom combines breadth, volume and intensity into a bounded score. // // The weighting favours breadth: a problem five consumers hit occasionally // says more about the interface than one consumer hitting it constantly. func severityFrom(consumers, occurrences int, intensity float64) float64 { breadth := saturate(float64(consumers) / 10) volume := saturate(float64(occurrences) / 100) return round2(saturate(0.5*breadth + 0.2*volume + 0.3*saturate(intensity))) } // confidenceFrom expresses how much the evidence supports any conclusion. func confidenceFrom(consumers, occurrences int) float64 { return round2(saturate(0.6*saturate(float64(consumers)/5) + 0.4*saturate(float64(occurrences)/50))) } func saturate(v float64) float64 { switch { case v < 0: return 0 case v > 1: return 1 } return v } func round2(v float64) float64 { return float64(int(v*100+0.5)) / 100 } // fingerprint identifies a finding stably across analysis runs. func fingerprint(kind, subject string) string { return kind + ":" + subject }