Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
120 changes: 120 additions & 0 deletions internal/investigate/decision_path_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,120 @@
// SPDX-License-Identifier: Apache-2.0

package investigate

import (
"context"
"math"
"testing"
"time"

"go.opentelemetry.io/otel/attribute"
"go.opentelemetry.io/otel/sdk/metric/metricdata"

"github.com/Smana/runlore/internal/catalog"
"github.com/Smana/runlore/internal/providers"
)

// gatingModel signals entered when Complete is called and answers only once the
// decider has been asked. With rendezvousDecider this pins that the two shadow arms
// overlap: under either serial order one side waits for a signal the other has not
// yet been reached to give, the context deadline expires, rank falls through, and
// the assertion catches it.
type gatingModel struct {
entered chan struct{}
asked <-chan struct{}
resp providers.CompletionResponse
}

func (m *gatingModel) Complete(ctx context.Context, _ providers.CompletionRequest) (providers.CompletionResponse, error) {
close(m.entered)
select {
case <-m.asked:
return m.resp, nil
case <-ctx.Done():
return providers.CompletionResponse{}, ctx.Err()
}
}

// rendezvousDecider is a fakeDecider that waits until the model has been entered
// before answering, and signals asked once it has.
type rendezvousDecider struct {
fakeDecider
entered <-chan struct{}
asked chan struct{}
}

func (d *rendezvousDecider) Decide(ctx context.Context, state string, qs []providers.Question) (providers.Answers, error) {
select {
case <-d.entered:
case <-ctx.Done():
return nil, ctx.Err()
}
close(d.asked)
return d.fakeDecider.Decide(ctx, state, qs)
}

// TestRerankShadowAsksTheDeciderWhileTheLLMRuns: shadow mode costs the recall path
// max(LLM, decider), not their sum (see rankShadow).
func TestRerankShadowAsksTheDeciderWhileTheLLMRuns(t *testing.T) {
entered, asked := make(chan struct{}), make(chan struct{})
dec := &rendezvousDecider{
fakeDecider: fakeDecider{answers: providers.Answers{rerankQuestionID: {Choice: "a.md", Confidence: 0.9}}},
entered: entered,
asked: asked,
}
rr := &Reranker{
Model: &gatingModel{entered: entered, asked: asked, resp: rerankResp(`{"match":true,"entry_id":"a.md","confidence":0.95}`)},
Decider: dec,
Backend: "shadow",
Threshold: 0.7, ThresholdJev: 0.7,
}
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Second)
defer cancel()
got, conf, ok := rr.rank(ctx, Request{Title: "t"}, []catalog.ScoredEntry{webHit("a.md", 1)}, &recallSpend{})
if !ok || got.Path != "a.md" || conf != 0.95 {
t.Fatalf("the arms did not overlap (one waited on the other): ok=%v path=%s conf=%v", ok, got.Path, conf)
}
if dec.calls != 1 {
t.Fatalf("the shadow arm must be asked exactly once, got %d", dec.calls)
}
}

// TestDecisionConfidenceSeparatesNoneFromACandidate: the rerank consumer labels each
// confidence sample by choice (see decideRerank), so a bar is read off the candidate
// answers alone.
func TestDecisionConfidenceSeparatesNoneFromACandidate(t *testing.T) {
m, reader := installMeterReader(t)
cands := []catalog.ScoredEntry{webHit("a.md", 1)}
for _, a := range []providers.Answer{
{Choice: rerankNoneOption, Confidence: 0.9},
{Choice: "a.md", Confidence: 0.6},
} {
rr := &Reranker{Decider: &fakeDecider{answers: providers.Answers{rerankQuestionID: a}}, Backend: "jev", Metrics: m}
if _, err := rr.decideRerank(context.Background(), Request{Title: "t"}, cands); err != nil {
t.Fatalf("decideRerank: %v", err)
}
}
md, ok := findSeries(t, reader, "runlore_decision_model_confidence")
if !ok {
t.Fatal("no confidence sample was recorded")
}
h, ok := md.Data.(metricdata.Histogram[float64])
if !ok {
t.Fatalf("runlore_decision_model_confidence is not a float64 histogram (%T)", md.Data)
}
got := map[string]metricdata.HistogramDataPoint[float64]{}
for _, dp := range h.DataPoints {
choice, ok := dp.Attributes.Value(attribute.Key("choice"))
if !ok {
t.Fatalf("a rerank sample carries no `choice` label: %v", dp.Attributes.ToSlice())
}
got[choice.AsString()] = dp
}
for choice, want := range map[string]float64{"none": 0.9, "candidate": 0.6} {
dp, ok := got[choice]
if !ok || dp.Count != 1 || math.Abs(dp.Sum-want) > 1e-9 {
t.Errorf("want one %s sample at %v, got count=%d sum=%v", choice, want, dp.Count, dp.Sum)
}
}
}
4 changes: 4 additions & 0 deletions internal/investigate/recall.go
Original file line number Diff line number Diff line change
Expand Up @@ -564,6 +564,10 @@ func nearMissAgrees(reqW providers.Workload, entryResource string, requireWorklo
// a recall REJECTION instead puts it beside rerank_no_signal and rerank_low_confidence,
// where an operator already looks to learn why a rerank did not happen.
func (r *Recall) affordRerank(ctx context.Context, req Request, cands []catalog.ScoredEntry, spend *recallSpend) bool {
// Gated whatever the backend, including jev, whose rank call itself spends no LLM
// tokens: the recall it produces still runs confirmRecall and verifyFindings, which
// are paid completions the ceiling never sees, so letting a crossed ceiling fire a
// free rank call would have it buy a verify it was about to refuse.
reason := spend.refuses(r.Rerank.requestEstimate(req, cands))
if reason == "" {
return true
Expand Down
69 changes: 43 additions & 26 deletions internal/investigate/recall_tokens_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -18,38 +18,55 @@ import (
"github.com/Smana/runlore/internal/telemetry"
)

// meteredInstruments installs a REAL SDK meter provider backed by a manual reader and
// returns the instrument set bound to it, plus a reader that sums an int64 counter by
// its exported series name (0, false when the series was never recorded). The provider
// is global, so the returned instruments must be used by exactly one test at a time
// (no t.Parallel here) and the cleanup restores the no-op provider.
func meteredInstruments(t *testing.T) (*telemetry.Metrics, func(series string) (int64, bool)) {
// installMeterReader installs a REAL SDK meter provider backed by a manual reader and
// returns the instrument set bound to it, plus the reader. The provider is global, so
// the returned instruments must be used by exactly one test at a time (no t.Parallel
// here) and the cleanup restores the no-op provider.
func installMeterReader(t *testing.T) (*telemetry.Metrics, *sdkmetric.ManualReader) {
t.Helper()
reader := sdkmetric.NewManualReader()
otel.SetMeterProvider(sdkmetric.NewMeterProvider(sdkmetric.WithReader(reader)))
t.Cleanup(func() { otel.SetMeterProvider(noop.NewMeterProvider()) })
return telemetry.NewMetrics(), func(series string) (int64, bool) {
var rm metricdata.ResourceMetrics
if err := reader.Collect(context.Background(), &rm); err != nil {
t.Fatalf("collect metrics: %v", err)
}
for _, sm := range rm.ScopeMetrics {
for _, md := range sm.Metrics {
if md.Name != series {
continue
}
sum, ok := md.Data.(metricdata.Sum[int64])
if !ok {
t.Fatalf("series %q is not an int64 sum (%T)", series, md.Data)
}
var total int64
for _, dp := range sum.DataPoints {
total += dp.Value
}
return total, true
return telemetry.NewMetrics(), reader
}

// findSeries collects what the reader has and returns the series by its exported
// name (false when it was never recorded).
func findSeries(t *testing.T, reader *sdkmetric.ManualReader, series string) (metricdata.Metrics, bool) {
t.Helper()
var rm metricdata.ResourceMetrics
if err := reader.Collect(context.Background(), &rm); err != nil {
t.Fatalf("collect metrics: %v", err)
}
for _, sm := range rm.ScopeMetrics {
for _, md := range sm.Metrics {
if md.Name == series {
return md, true
}
}
return 0, false
}
return metricdata.Metrics{}, false
}

// meteredInstruments is installMeterReader plus a reader that sums an int64 counter
// by its exported series name (0, false when the series was never recorded).
func meteredInstruments(t *testing.T) (*telemetry.Metrics, func(series string) (int64, bool)) {
t.Helper()
m, reader := installMeterReader(t)
return m, func(series string) (int64, bool) {
md, ok := findSeries(t, reader, series)
if !ok {
return 0, false
}
sum, ok := md.Data.(metricdata.Sum[int64])
if !ok {
t.Fatalf("series %q is not an int64 sum (%T)", series, md.Data)
}
var total int64
for _, dp := range sum.DataPoints {
total += dp.Value
}
return total, true
}
}

Expand Down
122 changes: 80 additions & 42 deletions internal/investigate/rerank.go
Original file line number Diff line number Diff line change
Expand Up @@ -83,16 +83,10 @@ type Reranker struct {
// confidences are not comparable — the LLM asserts its number, the decider derives one
// from probability spread — so the caller must gate on the bar belonging to the backend
// that produced the number, never on one shared field. In shadow mode the LLM decides,
// so the LLM's bar applies.
//
// The Decider != nil check mirrors rank()'s own dispatch: rank falls back to rankLLM
// whenever Decider is nil, REGARDLESS of Backend, so a verdict can carry Backend ==
// "jev" while the LLM actually produced it. ThresholdJev must apply only when the
// decider could actually have decided — the same condition rank dispatches on. Two
// functions answering "which backend decided" with different predicates is how they
// drift; checking Decider here too keeps them in sync.
// so the LLM's bar applies. Resolved through effectiveBackend — the same rule rank
// dispatches on — so the bar can never belong to a backend that did not decide.
func (rr *Reranker) fireThreshold() float64 {
if rr.Decider != nil && rr.Backend == "jev" {
if rr.effectiveBackend() == "jev" {
return rr.ThresholdJev
}
return rr.Threshold
Expand Down Expand Up @@ -241,6 +235,11 @@ const rerankQuestionID = "runbook"
// option — without it the decider would be forced to name a runbook it does not believe.
const rerankNoneOption = "none"

// named reports whether the decider named a candidate rather than the none option —
// the one predicate the fire decision, the shadow comparison and the confidence
// label all turn on.
func named(a providers.Answer) bool { return a.Choice != rerankNoneOption }

// decideRerank makes the raw call and returns the decider's answer. Split out from
// rankJev so the shadow arm can tell an ERROR from a no-match: rankJev collapses both
// into ok=false, which is right for a fire decision and useless for measuring agreement.
Expand Down Expand Up @@ -280,8 +279,17 @@ func (rr *Reranker) decideRerank(ctx context.Context, req Request, cands []catal
return providers.Answer{}, fmt.Errorf("no answer for question %q", rerankQuestionID)
}
if rr.Metrics != nil {
rr.Metrics.DecisionConfidence.Record(ctx, a.Confidence,
metric.WithAttributes(attribute.String("consumer", "rerank")))
// Which kind of answer this sample is. The histogram is what the promotion
// procedure reads a fire bar off, and "confident there is nothing to recall" and
// "confident candidate X matches" are different distributions: on a corpus where
// most incidents have no runbook the none answers carry the mass, and a bar set
// against it lands above every candidate answer, so in jev mode nothing fires.
choice := "none"
if named(a) {
choice = "candidate"
}
rr.Metrics.DecisionConfidence.Record(ctx, a.Confidence, metric.WithAttributes(
attribute.String("consumer", "rerank"), attribute.String("choice", choice)))
}
return a, nil
}
Expand All @@ -308,7 +316,7 @@ func (rr *Reranker) rankJev(ctx context.Context, req Request, cands []catalog.Sc
rr.Log.Info("decision-model rerank decision",
"title", req.Title, "choice", a.Choice, "confidence", a.Confidence)
}
if a.Choice == rerankNoneOption || a.Confidence < rr.ThresholdJev {
if !named(a) || a.Confidence < rr.ThresholdJev {
return catalog.Entry{}, 0, false
}
// Only ever an id it was offered: a backend that names something else must never
Expand Down Expand Up @@ -353,6 +361,17 @@ func shadowAgreement(llmFired bool, llmPath string, shadowFired bool, shadowPath
}
}

// effectiveBackend is the one place the Backend string and the Decider's presence
// resolve to the path taken: "llm" (the default, and the kill-switch fallback when a
// non-llm Backend has no Decider), "jev" or "shadow". rank dispatches on it and
// fireThreshold reads it, so the bar always belongs to the backend that decided.
func (rr *Reranker) effectiveBackend() string {
if rr.Decider != nil && (rr.Backend == "jev" || rr.Backend == "shadow") {
return rr.Backend
}
return "llm"
}

// rank routes the decision to the configured backend. The LLM path is the default and
// is unchanged; a decider that is configured but absent (the kill switch) falls back to
// it rather than failing, so an investigation's outcome never depends on the decider
Expand All @@ -361,44 +380,63 @@ func shadowAgreement(llmFired bool, llmPath string, shadowFired bool, shadowPath
// is unreachable in production, but the safe default is the proven path, and a typo
// must never silently double the decider's call volume.
func (rr *Reranker) rank(ctx context.Context, req Request, cands []catalog.ScoredEntry, spend *recallSpend) (catalog.Entry, float64, bool) {
if rr.Decider == nil {
return rr.rankLLM(ctx, req, cands, spend)
}
switch rr.Backend {
switch rr.effectiveBackend() {
case "jev":
return rr.rankJev(ctx, req, cands, spend)
case "shadow":
// the LLM decides, the decider is measured against it. The shadow arm must
// never change the outcome, including when it errors, so its answer is recorded
// and discarded. decideRerank rather than rankJev, because agreement needs to
// distinguish a genuine disagreement from an outage.
entry, conf, ok := rr.rankLLM(ctx, req, cands, spend)
a, err := rr.decideRerank(ctx, req, cands)
shadowFired := err == nil && a.Choice != rerankNoneOption && a.Confidence >= rr.ThresholdJev
shadowPath := ""
if shadowFired {
shadowPath = a.Choice
}
agreement := shadowAgreement(ok, entry.Path, shadowFired, shadowPath, err)
if spend != nil {
spend.shadow = shadowOutcome{Ran: true, Agreed: agreement == "agree"}
}
if rr.Metrics != nil {
rr.Metrics.DecisionShadow.Add(ctx, 1, metric.WithAttributes(
attribute.String("consumer", "rerank"),
attribute.String("agreement", agreement)))
}
if rr.Log != nil {
rr.Log.Info("rerank shadow comparison",
"title", req.Title, "llm_fired", ok, "llm_entry", entry.Path,
"shadow_fired", shadowFired, "shadow_entry", shadowPath, "shadow_err", err)
}
return entry, conf, ok
return rr.rankShadow(ctx, req, cands, spend)
default:
return rr.rankLLM(ctx, req, cands, spend)
}
}

// rankShadow: the LLM decides, the decider is measured against it. The shadow arm must
// never change the outcome, including when it errors, so its answer is recorded and
// discarded. decideRerank rather than rankJev, because agreement needs to distinguish
// a genuine disagreement from an outage.
//
// The decider is asked WHILE the LLM runs, so shadow costs the recall path
// max(LLM, decider) rather than their sum: run serially, a slow or black-holed
// endpoint added up to its client timeout to every instant recall — twice per
// investigation on the outcomeFallback path — for a verdict the LLM had already
// given. Joined rather than abandoned: the comparison is what the metric and the log
// line below exist to record, and a goroutine outliving rank would attribute it to
// nothing. The close/receive pair is the happens-before that makes a and err safe to
// read after the join.
func (rr *Reranker) rankShadow(ctx context.Context, req Request, cands []catalog.ScoredEntry, spend *recallSpend) (catalog.Entry, float64, bool) {
var (
a providers.Answer
err error
)
done := make(chan struct{})
go func() {
defer close(done)
a, err = rr.decideRerank(ctx, req, cands)
}()
entry, conf, ok := rr.rankLLM(ctx, req, cands, spend)
<-done
shadowFired := err == nil && named(a) && a.Confidence >= rr.ThresholdJev
shadowPath := ""
if shadowFired {
shadowPath = a.Choice
}
agreement := shadowAgreement(ok, entry.Path, shadowFired, shadowPath, err)
if spend != nil {
spend.shadow = shadowOutcome{Ran: true, Agreed: agreement == "agree"}
}
if rr.Metrics != nil {
rr.Metrics.DecisionShadow.Add(ctx, 1, metric.WithAttributes(
attribute.String("consumer", "rerank"),
attribute.String("agreement", agreement)))
}
if rr.Log != nil {
rr.Log.Info("rerank shadow comparison",
"title", req.Title, "llm_fired", ok, "llm_entry", entry.Path,
"shadow_fired", shadowFired, "shadow_entry", shadowPath, "shadow_err", err)
}
return entry, conf, ok
}

// rankLLM asks the reranker which of the (already structurally-agreeing) candidates is the
// correct runbook for this incident, returning the matched entry and a calibrated
// confidence. ok is false — meaning DO NOT short-circuit, fall through to a full
Expand Down
Loading
Loading