From 5ff6902bbe3987cfb26e9c0aeb6697ed92f0c8dc Mon Sep 17 00:00:00 2001 From: Ashwin Gopalsamy <47941624+ashwingopalsamy@users.noreply.github.com> Date: Tue, 22 Sep 2026 09:45:15 +0530 Subject: [PATCH 01/20] fix(intelligence): ignore git directory markers in snapshots --- internal/intelligence/snapshot.go | 13 +++++++++++-- internal/intelligence/snapshot_test.go | 11 +++++++++++ 2 files changed, 22 insertions(+), 2 deletions(-) diff --git a/internal/intelligence/snapshot.go b/internal/intelligence/snapshot.go index 1fef1f7..a475466 100644 --- a/internal/intelligence/snapshot.go +++ b/internal/intelligence/snapshot.go @@ -508,13 +508,13 @@ func (s *Snapshotter) snapshotPaths(ctx context.Context, index []byte, scope str if err != nil { return nil, fmt.Errorf("reading untracked content: %w", err) } - addNULPaths(paths, untracked) + addNULFilePaths(paths, untracked) ignoredInputs, err := s.gitBytes(ctx, "ls-files", "--others", "--ignored", "--exclude-standard", "-z", "--", "go.mod", "go.sum", "go.work", "go.work.sum", ":(glob)**/*.go", ":(glob)**/go.mod", ":(glob)**/go.sum", ":(glob)**/go.work", ":(glob)**/go.work.sum") if err != nil { return nil, fmt.Errorf("reading ignored Go inputs: %w", err) } - addNULPaths(paths, ignoredInputs) + addNULFilePaths(paths, ignoredInputs) for _, entry := range bytes.Split(index, []byte{0}) { if len(entry) > 2 && entry[0] >= 'a' && entry[0] <= 'z' && entry[1] == ' ' { paths[string(entry[2:])] = struct{}{} @@ -776,6 +776,15 @@ func addNULPaths(destination map[string]struct{}, data []byte) { } } +func addNULFilePaths(destination map[string]struct{}, data []byte) { + for _, value := range bytes.Split(data, []byte{0}) { + if len(value) == 0 || bytes.HasSuffix(value, []byte{'/'}) { + continue + } + destination[filepath.ToSlash(string(value))] = struct{}{} + } +} + func cleanSnapshotPath(path string) (string, error) { path = filepath.ToSlash(path) if path == "" || strings.ContainsRune(path, 0) || filepath.IsAbs(filepath.FromSlash(path)) { diff --git a/internal/intelligence/snapshot_test.go b/internal/intelligence/snapshot_test.go index 64ca368..54f8a64 100644 --- a/internal/intelligence/snapshot_test.go +++ b/internal/intelligence/snapshot_test.go @@ -349,6 +349,17 @@ func TestSnapshotIncludesIgnoredActiveInputsButNotInactiveFiles(t *testing.T) { } } +func TestSnapshotIgnoresGitDirectoryMarkers(t *testing.T) { + root := snapshotRepository(t) + writeSnapshotFile(t, root, ".gitignore", "ignored/\n") + writeSnapshotFile(t, root, "ignored/checkout/go.mod", "module example.test/ignored\n\ngo 1.25.0\n") + + snapshotter := newTestSnapshotter(t, root) + if _, err := snapshotter.Capture(context.Background(), SnapshotRequest{Semantic: SemanticIdentity{Version: "test"}}); err != nil { + t.Fatalf("Capture() with ignored nested checkout = %v", err) + } +} + func newTestSnapshotter(t *testing.T, root string) *Snapshotter { t.Helper() goPath, err := exec.LookPath("go") From 43814562b442d86097974f365a256e36f25b03ec Mon Sep 17 00:00:00 2001 From: Ashwin Gopalsamy <47941624+ashwingopalsamy@users.noreply.github.com> Date: Tue, 22 Sep 2026 09:52:57 +0530 Subject: [PATCH 02/20] fix(intelligence): gate focus facets by declaration kind --- docs/continuation/go-intelligence.md | 29 +++-- internal/intelligence/core_test.go | 32 +++-- internal/intelligence/focus.go | 40 +++++- internal/intelligence/focus_test.go | 175 +++++++++++++++++++++++++++ 4 files changed, 256 insertions(+), 20 deletions(-) diff --git a/docs/continuation/go-intelligence.md b/docs/continuation/go-intelligence.md index b5426d7..21bfc94 100644 --- a/docs/continuation/go-intelligence.md +++ b/docs/continuation/go-intelligence.md @@ -35,9 +35,9 @@ become stale. ## Current status -As of 2026-09-06 at reviewed HEAD `284df97`, **stage 2, coherent -observation**, and the additive focus slice are **implemented in the current -uncommitted change**. +As of 2026-09-22, this branch is based on the signed `v1.1.0` release. **Stage +2, coherent observation**, and the additive focus slice are shipped in that +release. The observation-correctness follow-up closed the source-confirmed gaps recorded by the Astra review. The change preserves the existing exact content-based Snapshot Ref and v1 @@ -48,7 +48,7 @@ post-v1 capabilities and are not frozen v1 interfaces. The broader roadmap stages remain partially implemented or pending and must not be inferred as complete from this slice. -Implemented in the current uncommitted change: +Implemented in v1.1.0: - `Snapshotter.observe` performs the existing two-pass capture, rejects drift, retains the exact manifest, and returns captured source bytes for the @@ -82,14 +82,29 @@ Focused correctness tests cover captured-source precedence, same-size and A→B→A rewrites on a source-cap miss, guidance identity and mismatch rejection, Brief observation forwarding, Symbol position-error lease release, active manifest protection, fail-closed admission, and replacement byte accounting. -The focused Slice 1 test command and `git diff --check` passed. Broader repository -verification is recorded separately at handoff; no benchmark, evaluation, -publication, commit, tag, or push was performed. +Those historical checks qualify the v1.1.0 release work; they do not qualify +later changes. The current v1.2 candidate separately passed the full +repository gates (`go test ./...`, `go test -race ./...`, `go vet ./...`, +`go build ./...`, and `git diff --check`). No benchmark, evaluation, +publication, tag, or push was performed. Stage 2 does not introduce a public interface or a general derived cache. Derived parsing/semantic caching, useful-context selection, richer relationships, refresh, and verification lineage remain later-stage work. +## Current post-v1.1.0 follow-up + +The current `codex/v1.2-reliability` follow-up narrows focus semantic expansion +by the selected declaration kind. Function and method selections do not request +type definitions, and non-callable declarations do not request call hierarchy. +Skipped facets are reported as unexamined rather than as examined-and-absent +evidence; incomplete provider evidence is not reported as absent. This is a +bounded reliability fix for observed provider failures; the public MCP +inventory and `agentic.focus/v1` schema remain unchanged. Focused package +validation and full repository gates have passed. The prerequisite snapshot +input fix is committed locally as `5ff6902`; the focus follow-up remains the +current gated slice. + ## Findings recorded from the documentation and targeted source inspection Use symbol names to relocate sections if line numbers change. These are diff --git a/internal/intelligence/core_test.go b/internal/intelligence/core_test.go index d10177c..2885537 100644 --- a/internal/intelligence/core_test.go +++ b/internal/intelligence/core_test.go @@ -44,16 +44,20 @@ func (p *fakeSemanticProvider) ReadObservation(_ context.Context, observation *s } type fakeSemanticReader struct { - hover string - diagnostics []Diagnostic - search semanticSymbols - definitions semanticLocations - typeDefinitions semanticLocations - references semanticLocations - implementations semanticSymbols - calls semanticCalls - symbol SymbolMatch - position Position + hover string + diagnostics []Diagnostic + search semanticSymbols + definitions semanticLocations + typeDefinitions semanticLocations + references semanticLocations + implementations semanticSymbols + calls semanticCalls + symbol SymbolMatch + position Position + typeDefinitionCalls int + callsCalls int + typeDefinitionErr error + callsErr error } func (r *fakeSemanticReader) Search(context.Context, string) (semanticSymbols, error) { @@ -74,6 +78,10 @@ func (r *fakeSemanticReader) Definition(context.Context, string, Position) (sema } func (r *fakeSemanticReader) TypeDefinition(context.Context, string, Position) (semanticLocations, error) { + r.typeDefinitionCalls++ + if r.typeDefinitionErr != nil { + return semanticLocations{}, r.typeDefinitionErr + } return r.typeDefinitions, nil } @@ -90,6 +98,10 @@ func (r *fakeSemanticReader) Diagnostics(context.Context, string) ([]Diagnostic, } func (r *fakeSemanticReader) Calls(context.Context, string, Position) (semanticCalls, error) { + r.callsCalls++ + if r.callsErr != nil { + return semanticCalls{}, r.callsErr + } return r.calls, nil } diff --git a/internal/intelligence/focus.go b/internal/intelligence/focus.go index daf1513..869a412 100644 --- a/internal/intelligence/focus.go +++ b/internal/intelligence/focus.go @@ -494,7 +494,28 @@ func (c *Core) focusContext(ctx context.Context, observation *snapshotObservatio result.Excerpts = excerpts result.Uncertainties = append(result.Uncertainties, uncertainties...) if len(result.CallSites) == 0 { - result.EvidenceStates = append(result.EvidenceStates, EvidenceState{Facet: "direct_callers", State: "examined_and_absent"}) + switch { + case !observation.snapshot.Capabilities.CallHierarchy: + result.EvidenceStates = append(result.EvidenceStates, EvidenceState{Facet: "direct_callers", State: "unavailable", Reason: "the active semantic provider does not support call hierarchy"}) + case !isCallableSymbol(focused.Symbol.Kind): + result.EvidenceStates = append(result.EvidenceStates, EvidenceState{Facet: "direct_callers", State: "unexamined", Reason: "call hierarchy expansion is limited to function and method declarations"}) + case focused.Calls.Truncated || focused.Calls.Total > len(focused.Calls.Items) || hasIncomingCallWithoutSites(focused.Calls.Items): + result.EvidenceStates = append(result.EvidenceStates, EvidenceState{Facet: "direct_callers", State: "gathered_but_omitted", Reason: "call hierarchy evidence was incomplete or lacked source call sites"}) + default: + result.EvidenceStates = append(result.EvidenceStates, EvidenceState{Facet: "direct_callers", State: "examined_and_absent"}) + } + } + if isCallableSymbol(focused.Symbol.Kind) { + result.EvidenceStates = append(result.EvidenceStates, EvidenceState{Facet: "type_definition", State: "unexamined", Reason: "type definitions are not applicable to function and method declarations"}) + } else { + switch { + case !observation.snapshot.Capabilities.TypeDefinition: + result.EvidenceStates = append(result.EvidenceStates, EvidenceState{Facet: "type_definition", State: "unavailable", Reason: "the active semantic provider does not support type definitions"}) + case focused.TypeDefinitions.Truncated || focused.TypeDefinitions.Total > len(focused.TypeDefinitions.Items): + result.EvidenceStates = append(result.EvidenceStates, EvidenceState{Facet: "type_definition", State: "gathered_but_omitted", Reason: "type-definition evidence was bounded or omitted"}) + case focused.TypeDefinitions.Total == 0: + result.EvidenceStates = append(result.EvidenceStates, EvidenceState{Facet: "type_definition", State: "examined_and_absent"}) + } } if len(tests) == 0 { result.EvidenceStates = append(result.EvidenceStates, EvidenceState{Facet: "related_tests", State: "examined_and_absent"}) @@ -547,7 +568,7 @@ func (c *Core) focusSymbol(ctx context.Context, reader semanticReader, observati result.Definitions = locationSet(locations) omitted["definitions"] = locations.Omitted } - if observation.snapshot.Capabilities.TypeDefinition { + if observation.snapshot.Capabilities.TypeDefinition && !isCallableSymbol(result.Symbol.Kind) { locations, readErr := reader.TypeDefinition(ctx, file, position) if readErr != nil { return nil, nil, nil, nil, readErr @@ -588,7 +609,7 @@ func (c *Core) focusSymbol(ctx context.Context, reader semanticReader, observati } result.DiagnosticsTotal = len(result.Diagnostics) } - if observation.snapshot.Capabilities.CallHierarchy { + if observation.snapshot.Capabilities.CallHierarchy && isCallableSymbol(result.Symbol.Kind) { calls, readErr := reader.Calls(ctx, file, position) if readErr != nil { return nil, nil, nil, nil, readErr @@ -668,6 +689,19 @@ func isGoTestEntry(name string) bool { return false } +func isCallableSymbol(kind string) bool { + return kind == "go.function" || kind == "go.method" +} + +func hasIncomingCallWithoutSites(calls []CallEdge) bool { + for _, call := range calls { + if call.Direction == "incoming" && len(call.CallSites) == 0 { + return true + } + } + return false +} + func (c *Core) focusExcerpts(observation *snapshotObservation, symbol SymbolContext, tests []SymbolMatch) []SourceExcerpt { excerpts := []SourceExcerpt{} seen := make(map[string]struct{}) diff --git a/internal/intelligence/focus_test.go b/internal/intelligence/focus_test.go index 1b0cc31..d498e9d 100644 --- a/internal/intelligence/focus_test.go +++ b/internal/intelligence/focus_test.go @@ -349,3 +349,178 @@ func TestFocusUnsupportedDocumentSymbolsReturnsUnavailableEvidence(t *testing.T) t.Fatalf("unsupported focus = %#v", result) } } + +func TestFocusFacetApplicabilityFollowsDeclarationKind(t *testing.T) { + tests := []struct { + name string + kind string + typeDefinitionState string + callsState string + wantTypeCalls int + wantCalls int + }{ + {name: "function", kind: "go.function", typeDefinitionState: "unexamined", callsState: "examined_and_absent", wantTypeCalls: 0, wantCalls: 1}, + {name: "method", kind: "go.method", typeDefinitionState: "unexamined", callsState: "examined_and_absent", wantTypeCalls: 0, wantCalls: 1}, + {name: "type", kind: "go.type", typeDefinitionState: "examined_and_absent", callsState: "unexamined", wantTypeCalls: 1, wantCalls: 0}, + {name: "variable", kind: "go.variable", typeDefinitionState: "examined_and_absent", callsState: "unexamined", wantTypeCalls: 1, wantCalls: 0}, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + root := snapshotRepository(t) + writeSnapshotFile(t, root, "main.go", "package fixture\n\nfunc Value() {}\ntype Worker struct{}\nvar Count int\n") + snapshotter := newTestSnapshotter(t, root) + reader := &fakeSemanticReader{symbol: SymbolMatch{Name: test.name, Qualified: "fixture." + test.name, Kind: test.kind, Location: Location{File: "main.go", Line: 3, Column: 1}}} + core := newTestCore(t, snapshotter, reader) + core.semantic.(*fakeSemanticProvider).identity.Capabilities.CallHierarchy = true + observation, err := core.observe(context.Background(), "HEAD", "./...", "") + if err != nil { + t.Fatal(err) + } + defer observation.release() + + result, err := core.focusContext(context.Background(), &observation, FocusRequest{Position: &SourcePosition{File: "main.go", Line: 3, Column: 1}, MaxBytes: DefaultBriefBytes}) + if err != nil { + t.Fatal(err) + } + if reader.typeDefinitionCalls != test.wantTypeCalls || reader.callsCalls != test.wantCalls { + t.Fatalf("provider calls = type_definition:%d calls:%d, want type_definition:%d calls:%d", reader.typeDefinitionCalls, reader.callsCalls, test.wantTypeCalls, test.wantCalls) + } + if !hasEvidenceState(result.EvidenceStates, "type_definition", test.typeDefinitionState) || !hasEvidenceState(result.EvidenceStates, "direct_callers", test.callsState) { + t.Fatalf("evidence states = %#v, want type_definition=%s direct_callers=%s", result.EvidenceStates, test.typeDefinitionState, test.callsState) + } + }) + } +} + +func TestFocusFacetStatesDoNotClaimIncompleteEvidenceAbsent(t *testing.T) { + tests := []struct { + name string + calls semanticCalls + maxBytes int + }{ + { + name: "incoming edge without source sites", + calls: semanticCalls{Items: []CallEdge{{Direction: "incoming", Symbol: SymbolMatch{Name: "Caller", Qualified: "fixture.Caller", Kind: "go.function"}}}}, + }, + { + name: "provider omitted edges", + calls: semanticCalls{Omitted: 1}, + }, + { + name: "budget truncated edge", + calls: semanticCalls{Items: []CallEdge{{Direction: "incoming", Symbol: SymbolMatch{Name: "Caller", Qualified: "fixture." + strings.Repeat("Caller", 2000), Kind: "go.function", Package: strings.Repeat("fixture", 500)}}}}, + maxBytes: 4096, + }, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + root := snapshotRepository(t) + writeSnapshotFile(t, root, "main.go", "package fixture\n\nfunc Value() {}\n") + snapshotter := newTestSnapshotter(t, root) + reader := &fakeSemanticReader{ + symbol: SymbolMatch{Name: "Value", Qualified: "fixture.Value", Kind: "go.function", Location: Location{File: "main.go", Line: 3, Column: 1}}, + calls: test.calls, + } + core := newTestCore(t, snapshotter, reader) + core.semantic.(*fakeSemanticProvider).identity.Capabilities.CallHierarchy = true + observation, err := core.observe(context.Background(), "HEAD", "./...", "") + if err != nil { + t.Fatal(err) + } + defer observation.release() + + maxBytes := test.maxBytes + if maxBytes == 0 { + maxBytes = DefaultBriefBytes + } + result, err := core.focusContext(context.Background(), &observation, FocusRequest{Position: &SourcePosition{File: "main.go", Line: 3, Column: 1}, MaxBytes: maxBytes}) + if err != nil { + t.Fatal(err) + } + if !hasEvidenceState(result.EvidenceStates, "direct_callers", "gathered_but_omitted") || hasEvidenceState(result.EvidenceStates, "direct_callers", "examined_and_absent") { + t.Fatalf("incomplete caller evidence = %#v", result.EvidenceStates) + } + }) + } +} + +func TestFocusUnsupportedFacetStatesAreExplicit(t *testing.T) { + root := snapshotRepository(t) + writeSnapshotFile(t, root, "main.go", "package fixture\n\ntype Worker struct{}\n") + snapshotter := newTestSnapshotter(t, root) + reader := &fakeSemanticReader{symbol: SymbolMatch{Name: "Worker", Qualified: "fixture.Worker", Kind: "go.type", Location: Location{File: "main.go", Line: 3, Column: 6}}} + core := newTestCore(t, snapshotter, reader) + provider := core.semantic.(*fakeSemanticProvider) + provider.identity.Capabilities.CallHierarchy = false + provider.identity.Capabilities.TypeDefinition = false + observation, err := core.observe(context.Background(), "HEAD", "./...", "") + if err != nil { + t.Fatal(err) + } + defer observation.release() + + result, err := core.focusContext(context.Background(), &observation, FocusRequest{Position: &SourcePosition{File: "main.go", Line: 3, Column: 6}, MaxBytes: DefaultBriefBytes}) + if err != nil { + t.Fatal(err) + } + if reader.typeDefinitionCalls != 0 || reader.callsCalls != 0 || !hasEvidenceState(result.EvidenceStates, "type_definition", "unavailable") || !hasEvidenceState(result.EvidenceStates, "direct_callers", "unavailable") { + t.Fatalf("unsupported facet result = %#v, provider calls type_definition:%d calls:%d", result.EvidenceStates, reader.typeDefinitionCalls, reader.callsCalls) + } +} + +func TestFocusFacetProviderErrorsPropagate(t *testing.T) { + tests := []struct { + name string + kind string + err error + calls bool + }{ + {name: "type definition cancellation", kind: "go.type", err: context.Canceled}, + {name: "call hierarchy stale snapshot", kind: "go.function", err: ErrSnapshotChanged, calls: true}, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + root := snapshotRepository(t) + writeSnapshotFile(t, root, "main.go", "package fixture\n\ntype Worker struct{}\nfunc Value() {}\n") + snapshotter := newTestSnapshotter(t, root) + reader := &fakeSemanticReader{symbol: SymbolMatch{Name: "selected", Qualified: "fixture.selected", Kind: test.kind, Location: Location{File: "main.go", Line: 3, Column: 1}}} + if test.calls { + reader.callsErr = test.err + } else { + reader.typeDefinitionErr = test.err + } + core := newTestCore(t, snapshotter, reader) + core.semantic.(*fakeSemanticProvider).identity.Capabilities.CallHierarchy = true + observation, err := core.observe(context.Background(), "HEAD", "./...", "") + if err != nil { + t.Fatal(err) + } + defer observation.release() + + _, err = core.focusContext(context.Background(), &observation, FocusRequest{Position: &SourcePosition{File: "main.go", Line: 3, Column: 1}, MaxBytes: DefaultBriefBytes}) + if !errors.Is(err, test.err) { + t.Fatalf("focusContext() error = %v, want %v", err, test.err) + } + }) + } +} + +func TestHasIncomingCallWithoutSites(t *testing.T) { + tests := []struct { + name string + calls []CallEdge + want bool + }{ + {name: "empty", calls: nil, want: false}, + {name: "outgoing", calls: []CallEdge{{Direction: "outgoing"}}, want: false}, + {name: "incoming with site", calls: []CallEdge{{Direction: "incoming", CallSites: []Location{{File: "main.go"}}}}, want: false}, + {name: "incoming without site", calls: []CallEdge{{Direction: "incoming"}}, want: true}, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + if got := hasIncomingCallWithoutSites(test.calls); got != test.want { + t.Fatalf("hasIncomingCallWithoutSites() = %t, want %t", got, test.want) + } + }) + } +} From 7cb2117b887f24c12e32b38f0baee99fe30896cc Mon Sep 17 00:00:00 2001 From: Ashwin Gopalsamy <47941624+ashwingopalsamy@users.noreply.github.com> Date: Tue, 22 Sep 2026 10:01:35 +0530 Subject: [PATCH 03/20] feat(trace): record bounded go_context outcomes --- internal/tools/intelligence_tools.go | 53 ++++++++++++++- internal/tools/intelligence_tools_test.go | 82 +++++++++++++++++++++++ 2 files changed, 133 insertions(+), 2 deletions(-) diff --git a/internal/tools/intelligence_tools.go b/internal/tools/intelligence_tools.go index 63b4ca7..3542911 100644 --- a/internal/tools/intelligence_tools.go +++ b/internal/tools/intelligence_tools.go @@ -2,10 +2,13 @@ package tools import ( "context" + "errors" "fmt" "strings" + "time" "github.com/agentic-mcps/go/internal/intelligence" + "github.com/agentic-mcps/go/internal/trace" "github.com/agentic-mcps/go/internal/verification" "github.com/modelcontextprotocol/go-sdk/mcp" ) @@ -149,7 +152,25 @@ func (r *Runtime) symbolContext(ctx context.Context, _ *mcp.CallToolRequest, inp return &mcp.CallToolResult{Content: []mcp.Content{&mcp.TextContent{Text: fmt.Sprintf("symbol context for %s at snapshot %s; canonical context is in structuredContent", result.Symbol.Name, result.Snapshot.ID)}}}, result, nil } -func (r *Runtime) context(ctx context.Context, _ *mcp.CallToolRequest, input ContextInput) (*mcp.CallToolResult, intelligence.FocusResult, error) { +func (r *Runtime) context(ctx context.Context, _ *mcp.CallToolRequest, input ContextInput) (call *mcp.CallToolResult, result intelligence.FocusResult, returnErr error) { + started := time.Now() + var tracer *trace.Tracer + if r != nil { + tracer = r.tracer + } + defer func() { + if tracer == nil { + return + } + event := trace.Event{Tool: "go_context", Args: input, Duration: time.Since(started)} + if returnErr != nil { + event.ErrorKind = contextTraceErrorKind(returnErr) + } else { + event.ResultSummary = contextTraceSummary(result) + } + _ = tracer.Record(event) + }() + service, err := r.requireIntelligence() if err != nil { return nil, intelligence.FocusResult{}, err @@ -165,10 +186,38 @@ func (r *Runtime) context(ctx context.Context, _ *mcp.CallToolRequest, input Con if input.File != "" || input.Line != 0 || input.Column != 0 { request.Position = &intelligence.SourcePosition{File: input.File, Line: input.Line, Column: input.Column} } - result, err := service.Focus(ctx, request) + result, err = service.Focus(ctx, request) if err != nil { return nil, intelligence.FocusResult{}, fmt.Errorf("building change context: %w", err) } text := intelligence.FocusSummary(result) + "; canonical evidence is in structuredContent" return &mcp.CallToolResult{Content: []mcp.Content{&mcp.TextContent{Text: text}}}, result, nil } + +func contextTraceErrorKind(err error) trace.ErrorKind { + switch { + case errors.Is(err, context.Canceled): + return trace.ErrorCancelled + case errors.Is(err, context.DeadlineExceeded): + return trace.ErrorDeadline + default: + return trace.ErrorInternal + } +} + +func contextTraceSummary(result intelligence.FocusResult) string { + truncated := false + if result.Context != nil { + truncated = result.Context.Truncated + } + refresh := "none" + if result.Refresh != nil { + switch result.Refresh.Status { + case "replaced": + refresh = "replaced" + default: + refresh = "other" + } + } + return fmt.Sprintf("changed_files=%d; impacted_packages=%d; complete=%t; truncated=%t; verification_applicable=%t; refresh=%s", result.Change.FilesTotal, result.Impact.PackagesTotal, result.Complete, truncated, result.Verification.Applicable, refresh) +} diff --git a/internal/tools/intelligence_tools_test.go b/internal/tools/intelligence_tools_test.go index 422c11b..3216ddc 100644 --- a/internal/tools/intelligence_tools_test.go +++ b/internal/tools/intelligence_tools_test.go @@ -3,6 +3,9 @@ package tools import ( "context" "encoding/json" + "errors" + "os" + "path/filepath" "strings" "testing" @@ -21,6 +24,7 @@ type fakeIntelligence struct { //nolint:govet // Test requests are grouped by op refactor intelligence.RefactorRequest verify verification.Request focus intelligence.FocusRequest + focusErr error } func (f *fakeIntelligence) Brief(_ context.Context, request intelligence.BriefRequest) (intelligence.ContextPack, error) { @@ -40,6 +44,9 @@ func (f *fakeIntelligence) Symbol(_ context.Context, request intelligence.Symbol func (f *fakeIntelligence) Focus(_ context.Context, request intelligence.FocusRequest) (intelligence.FocusResult, error) { f.focus = request + if f.focusErr != nil { + return intelligence.FocusResult{}, f.focusErr + } return intelligence.FocusResult{SchemaVersion: intelligence.FocusSchemaVersion, Snapshot: intelligence.SnapshotRef{ID: "snap-focus"}, Change: verification.Change{Files: []verification.ChangedFile{}, Declarations: []verification.ChangedDeclaration{}, FilesTotal: 1}, Impact: verification.Impact{Packages: []verification.ImpactedPackage{}, PackagesTotal: 2}, Risks: []verification.RiskArea{}, Uncertainties: []verification.Uncertainty{}, Verification: intelligence.VerificationApplicability{Reasons: []string{}}, PackID: strings.Repeat("c", 64), Refresh: &intelligence.FocusRefresh{Status: "replaced"}}, nil } @@ -214,6 +221,81 @@ func TestIntelligenceToolsRejectInvalidInputAndMissingService(t *testing.T) { } } +func TestContextTraceRecordsBoundedSuccess(t *testing.T) { + traceRoot := t.TempDir() + tracer, err := trace.NewWithBaseDir(traceRoot) + if err != nil { + t.Fatal(err) + } + defer func() { _ = tracer.Close() }() + fake := &fakeIntelligence{} + runtime := &Runtime{intelligence: fake, tracer: tracer} + + input := ContextInput{Base: "HEAD", Query: "SensitiveSymbol"} + if _, _, err := runtime.context(context.Background(), nil, input); err != nil { + t.Fatal(err) + } + fake.focusErr = errors.New("provider failure for SensitiveSymbol in internal/private.go") + if _, _, err := runtime.context(context.Background(), nil, input); err == nil { + t.Fatal("context unexpectedly succeeded") + } + summary, err := tracer.Summary() + if err != nil { + t.Fatal(err) + } + if !summary.Enabled || summary.RecordsConsidered != 2 || len(summary.Tools) != 1 || summary.Tools[0].Tool != "go_context" || summary.Tools[0].Calls != 2 || summary.Tools[0].ErrorCount != 1 { + t.Fatalf("trace summary = %#v", summary) + } + payload := readTracePayload(t, traceRoot) + for _, forbidden := range []string{"SensitiveSymbol", "internal/private.go", "building change context"} { + if strings.Contains(payload, forbidden) { + t.Fatalf("trace payload contains %q: %s", forbidden, payload) + } + } + if !strings.Contains(payload, `"result_summary":"changed_files=1; impacted_packages=2; complete=false; truncated=false; verification_applicable=false; refresh=replaced"`) || !strings.Contains(payload, `"error_kind":"internal"`) { + t.Fatalf("trace payload = %s", payload) + } +} + +func TestContextTraceClassifiesCancellationAndPreservesToolError(t *testing.T) { + traceRoot := t.TempDir() + tracer, err := trace.NewWithBaseDir(traceRoot) + if err != nil { + t.Fatal(err) + } + defer func() { _ = tracer.Close() }() + runtime := &Runtime{intelligence: &fakeIntelligence{focusErr: context.Canceled}, tracer: tracer} + + _, _, err = runtime.context(context.Background(), nil, ContextInput{Base: "HEAD", Query: "SensitiveSymbol"}) + if !errors.Is(err, context.Canceled) { + t.Fatalf("context error = %v, want cancellation", err) + } + summary, err := tracer.Summary() + if err != nil { + t.Fatal(err) + } + if len(summary.Tools) != 1 || summary.Tools[0].ErrorCount != 1 { + t.Fatalf("trace summary = %#v", summary) + } + payload := readTracePayload(t, traceRoot) + if !strings.Contains(payload, `"error_kind":"cancelled"`) || !strings.Contains(payload, `"result_summary":""`) { + t.Fatalf("trace payload = %s", payload) + } +} + +func readTracePayload(t *testing.T, root string) string { + t.Helper() + paths, err := filepath.Glob(filepath.Join(root, "*", "trace.jsonl")) + if err != nil || len(paths) != 1 { + t.Fatalf("trace files = %v, err %v", paths, err) + } + payload, err := os.ReadFile(paths[0]) + if err != nil { + t.Fatal(err) + } + return string(payload) +} + func TestIntelligenceResourcesReturnCapabilitiesAndArtifactChunks(t *testing.T) { runtime := testIntelligenceRuntime(&fakeIntelligence{}) capabilities, err := runtime.capabilitiesResource(context.Background(), &mcp.ReadResourceRequest{Params: &mcp.ReadResourceParams{URI: capabilitiesURI}}) From 91522afb3f28be8429591c555d252b8906e86ccf Mon Sep 17 00:00:00 2001 From: Ashwin Gopalsamy <47941624+ashwingopalsamy@users.noreply.github.com> Date: Wed, 23 Sep 2026 08:34:10 +0530 Subject: [PATCH 04/20] fix(test): satisfy release lint gate --- internal/intelligence/core_test.go | 12 ++++++------ internal/tools/intelligence_tools_test.go | 4 ++-- 2 files changed, 8 insertions(+), 8 deletions(-) diff --git a/internal/intelligence/core_test.go b/internal/intelligence/core_test.go index 2885537..53ddd5e 100644 --- a/internal/intelligence/core_test.go +++ b/internal/intelligence/core_test.go @@ -44,20 +44,20 @@ func (p *fakeSemanticProvider) ReadObservation(_ context.Context, observation *s } type fakeSemanticReader struct { + callsErr error + typeDefinitionErr error hover string diagnostics []Diagnostic - search semanticSymbols - definitions semanticLocations - typeDefinitions semanticLocations - references semanticLocations implementations semanticSymbols + references semanticLocations + typeDefinitions semanticLocations calls semanticCalls + definitions semanticLocations + search semanticSymbols symbol SymbolMatch position Position typeDefinitionCalls int callsCalls int - typeDefinitionErr error - callsErr error } func (r *fakeSemanticReader) Search(context.Context, string) (semanticSymbols, error) { diff --git a/internal/tools/intelligence_tools_test.go b/internal/tools/intelligence_tools_test.go index 3216ddc..f93498e 100644 --- a/internal/tools/intelligence_tools_test.go +++ b/internal/tools/intelligence_tools_test.go @@ -232,11 +232,11 @@ func TestContextTraceRecordsBoundedSuccess(t *testing.T) { runtime := &Runtime{intelligence: fake, tracer: tracer} input := ContextInput{Base: "HEAD", Query: "SensitiveSymbol"} - if _, _, err := runtime.context(context.Background(), nil, input); err != nil { + if _, _, err = runtime.context(context.Background(), nil, input); err != nil { t.Fatal(err) } fake.focusErr = errors.New("provider failure for SensitiveSymbol in internal/private.go") - if _, _, err := runtime.context(context.Background(), nil, input); err == nil { + if _, _, err = runtime.context(context.Background(), nil, input); err == nil { t.Fatal("context unexpectedly succeeded") } summary, err := tracer.Summary() From dcee0aadd1f40bc19e201904282957367960297b Mon Sep 17 00:00:00 2001 From: Ashwin Gopalsamy <47941624+ashwingopalsamy@users.noreply.github.com> Date: Wed, 23 Sep 2026 08:34:20 +0530 Subject: [PATCH 05/20] docs(release): prepare v1.2.0 --- CHANGELOG.md | 20 +++++++++++++++++- README.md | 10 ++++----- assets/brand/pills/release.svg | 4 ++-- docs/continuation/go-intelligence.md | 31 +++++++++++++++------------- docs/module-migration.md | 6 +++--- llms.txt | 2 +- site/.well-known/mcp.json | 4 ++-- site/app.js | 16 +++++++------- site/assets/brand/pills/release.svg | 4 ++-- site/docs/contracts/index.html | 8 +++---- site/docs/index.html | 2 +- site/index.html | 2 +- site/llms-full.txt | 4 ++-- site/llms.txt | 4 ++-- 14 files changed, 69 insertions(+), 48 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 3e53adc..01a20c6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,23 @@ All notable changes to this project are documented here. The format follows ## [Unreleased] +## [1.2.0] - 2026-09-23 + +### Added + +- Bounded opt-in `go_context` trace outcomes with privacy-preserving aggregate + summaries and cancellation, deadline, and internal failure categories. +- Focus-facet evidence tests covering callable declarations, type declarations, + variables and fields, unsupported capabilities, cancellation, and stale + references. + +### Fixed + +- Snapshot observation now ignores Git directory markers emitted by status + commands while preserving strict file and content validation. +- Focus expansion now reports applicability and evidence states accurately and + preserves provider and snapshot errors. + ## [1.1.0] - 2026-09-07 ### Added @@ -75,6 +92,7 @@ All notable changes to this project are documented here. The format follows - Workspace containment, bounded execution, event-driven progress, and optional privacy-preserving local traces. -[Unreleased]: https://github.com/agentic-mcps/go/compare/v1.1.0...HEAD +[Unreleased]: https://github.com/agentic-mcps/go/compare/v1.2.0...HEAD +[1.2.0]: https://github.com/agentic-mcps/go/releases/tag/v1.2.0 [1.1.0]: https://github.com/agentic-mcps/go/releases/tag/v1.1.0 [1.0.0]: https://github.com/agentic-mcps/go/releases/tag/v1.0.0 diff --git a/README.md b/README.md index e43dfaf..9c4102a 100644 --- a/README.md +++ b/README.md @@ -10,14 +10,14 @@ Install agentic-go Connect MCP Read docs - v1.1.0 release + v1.2.0 release

Website · Docs · Install · Connect · Workflow · Capabilities · FAQ

`agentic-go` is a local Go MCP server and CLI. It gives an external coding agent semantic context, change continuity, guarded refactoring, and executed verification without embedding an LLM or becoming an agent framework. -The v1.1.0 server exposes 15 MCP tools: the frozen v1 surface of 14 tools plus the additive `go_context` tool under `agentic.focus/v1`. The seven resources, resource template, six prompts, and frozen v1 schemas remain unchanged. +The v1.2.0 server exposes 15 MCP tools: the frozen v1 surface of 14 tools plus the additive `go_context` tool under `agentic.focus/v1`. This release hardens snapshot observation, focus applicability evidence, and bounded private tracing. The seven resources, resource template, six prompts, and frozen v1 schemas remain unchanged. ## Install @@ -28,7 +28,7 @@ brew install agentic-mcps/tap/agentic-go agentic-go --version ``` -That installs `agentic-go`, the pinned `agentic-go-gopls` companion, and `agentic-go-vet`. The Homebrew tap is maintained separately; the signed v1.1.0 release archive and checksum installer below are the canonical versioned distribution path. +That installs `agentic-go`, the pinned `agentic-go-gopls` companion, and `agentic-go-vet`. The Homebrew tap is maintained separately; the signed v1.2.0 release archive and checksum installer below are the canonical versioned distribution path. For an agent workflow, install the binary first, then print the client-native MCP entry for the current workspace: @@ -44,8 +44,8 @@ edit your client configuration. Install from the release archive instead ```sh -curl -fsSL https://raw.githubusercontent.com/agentic-mcps/go/v1.1.0/scripts/install.sh \ - | bash -s -- 1.1.0 +curl -fsSL https://raw.githubusercontent.com/agentic-mcps/go/v1.2.0/scripts/install.sh \ + | bash -s -- 1.2.0 ``` The installer places the binaries in `~/.local/bin` and verifies the release checksum before replacing them. diff --git a/assets/brand/pills/release.svg b/assets/brand/pills/release.svg index 5779c3a..76e6423 100644 --- a/assets/brand/pills/release.svg +++ b/assets/brand/pills/release.svg @@ -1,6 +1,6 @@ - + - v1.1.0 + v1.2.0 diff --git a/docs/continuation/go-intelligence.md b/docs/continuation/go-intelligence.md index 21bfc94..cd73911 100644 --- a/docs/continuation/go-intelligence.md +++ b/docs/continuation/go-intelligence.md @@ -35,7 +35,8 @@ become stale. ## Current status -As of 2026-09-22, this branch is based on the signed `v1.1.0` release. **Stage +As of 2026-09-22, this branch prepares the signed `v1.2.0` release from the +`v1.1.0` baseline. **Stage 2, coherent observation**, and the additive focus slice are shipped in that release. The observation-correctness follow-up closed the source-confirmed gaps recorded @@ -83,18 +84,19 @@ A→B→A rewrites on a source-cap miss, guidance identity and mismatch rejectio Brief observation forwarding, Symbol position-error lease release, active manifest protection, fail-closed admission, and replacement byte accounting. Those historical checks qualify the v1.1.0 release work; they do not qualify -later changes. The current v1.2 candidate separately passed the full +later changes. The v1.2.0 candidate separately passed the full repository gates (`go test ./...`, `go test -race ./...`, `go vet ./...`, -`go build ./...`, and `git diff --check`). No benchmark, evaluation, -publication, tag, or push was performed. +`go build ./...`, and `git diff --check`). The v0.8 task, adoption, and pilot +definitions validate, and two available private server replays pass. No +three-tier model evaluation or productivity claim is included. Stage 2 does not introduce a public interface or a general derived cache. Derived parsing/semantic caching, useful-context selection, richer relationships, refresh, and verification lineage remain later-stage work. -## Current post-v1.1.0 follow-up +## v1.2.0 reliability release -The current `codex/v1.2-reliability` follow-up narrows focus semantic expansion +The `codex/v1.2-reliability` release narrows focus semantic expansion by the selected declaration kind. Function and method selections do not request type definitions, and non-callable declarations do not request call hierarchy. Skipped facets are reported as unexamined rather than as examined-and-absent @@ -102,8 +104,9 @@ evidence; incomplete provider evidence is not reported as absent. This is a bounded reliability fix for observed provider failures; the public MCP inventory and `agentic.focus/v1` schema remain unchanged. Focused package validation and full repository gates have passed. The prerequisite snapshot -input fix is committed locally as `5ff6902`; the focus follow-up remains the -current gated slice. +input fix is committed as `5ff6902`, the focus follow-up as `4381456`, and +bounded outcome tracing as `7cb2117`. The release metadata preserves the +existing 15-tool current surface and frozen v1 contracts. ## Findings recorded from the documentation and targeted source inspection @@ -269,9 +272,9 @@ identities, and limitations. Retain focus and full replacement; defer delta refresh. The next slice is release hardening, instruction-surface discoverability, and provider-failure investigation. Raw artifacts remain private and ignored. -This handoff predates the separately authorized publication workflow. Public -publication must preserve existing tags and history, and does not create a new -release or claim that the paid comparison ran. +The v1.2.0 publication workflow is separately authorized. Public publication +preserves existing tags and history and does not claim that the paid model +comparison ran. ## Standing continuation instruction @@ -292,7 +295,7 @@ Read docs/continuation/astra-understanding.md and this handoff first. Observatio verification applicability, declaration selection, and full-replacement refresh and focus-v1 stabilization are complete. Inspect the current diff and choose a new explicitly authorized objective. Delta refresh, general derived caches, -expanded refactoring, speculative test selection, and release creation remain -outside the completed scope. Publication is separately authorized only when a -maintainer explicitly requests it; preserve existing tags and public history. +expanded refactoring, and speculative test selection remain outside the +completed scope. Preserve existing tags and public history when continuing the +reliability work. ``` diff --git a/docs/module-migration.md b/docs/module-migration.md index 7b7f03a..edd19df 100644 --- a/docs/module-migration.md +++ b/docs/module-migration.md @@ -16,8 +16,8 @@ Install the organization release over the existing binary names: ```sh curl --fail --location --silent --show-error \ - https://raw.githubusercontent.com/agentic-mcps/go/v1.1.0/scripts/install.sh \ - | bash -s -- 1.1.0 + https://raw.githubusercontent.com/agentic-mcps/go/v1.2.0/scripts/install.sh \ + | bash -s -- 1.2.0 agentic-go --version agentic-go doctor @@ -26,7 +26,7 @@ agentic-go doctor Update GitHub Actions references to: ```yaml -- uses: agentic-mcps/go@v1.1.0 +- uses: agentic-mcps/go@v1.2.0 ``` Update source-checkout or advanced `go install` commands to use diff --git a/llms.txt b/llms.txt index ce3359f..c909613 100644 --- a/llms.txt +++ b/llms.txt @@ -7,7 +7,7 @@ This is a convenience index for humans and coding agents. It is not a ranking or ## Start - [README](https://github.com/agentic-mcps/go/blob/main/README.md): installation, MCP setup, workflow, capabilities, compatibility, and trust boundary. -- [v1.1.0 release](https://github.com/agentic-mcps/go/releases/tag/v1.1.0): supported archives and checksums. +- [v1.2.0 release](https://github.com/agentic-mcps/go/releases/tag/v1.2.0): supported archives and checksums. - [Module migration](https://github.com/agentic-mcps/go/blob/main/docs/module-migration.md): separate personal and organization module identities. ## Contracts diff --git a/site/.well-known/mcp.json b/site/.well-known/mcp.json index df3b9e3..fcaa024 100644 --- a/site/.well-known/mcp.json +++ b/site/.well-known/mcp.json @@ -1,9 +1,9 @@ { "name": "agentic-go", - "version": "1.1.0", + "version": "1.2.0", "serverInfo": { "name": "agentic-go", - "version": "1.1.0" + "version": "1.2.0" }, "description": "Source-grounded Go intelligence, change continuity, guarded refactoring, and whole-package verification for coding agents.", "homepage": "https://agentic-mcps.github.io/go/", diff --git a/site/app.js b/site/app.js index 3b478e0..b0fb4ea 100644 --- a/site/app.js +++ b/site/app.js @@ -516,7 +516,7 @@ const DOCS_SEARCH_INDEX = [ title: "Protocol Contracts & Schemas", path: "/go/docs/contracts/", relPath: "docs/contracts/", - summary: "Frozen v1 specification for 14 MCP tools, 7 resources, 6 prompts, and JSON schemas, plus the additive go_context tool in v1.1.0.", + summary: "Frozen v1 specification for 14 MCP tools, 7 resources, 6 prompts, and JSON schemas, plus the additive go_context tool in v1.2.0.", keywords: "contracts, schemas, protocol, tools, 14 tools, 15 tools, go_context, resources, prompts, agentic.context/v1, agentic.change/v1, agentic.verify/v1, agentic.focus/v1" }, { @@ -689,11 +689,11 @@ function generateInstallCommandPayload(params = {}) { command = 'brew install agentic-mcps/tap/agentic-go\nagentic-go --version'; instructions = 'Installs agentic-go, agentic-go-gopls companion, and agentic-go-vet via official Homebrew tap.'; } else if (method === 'curl') { - command = 'curl -fsSL https://raw.githubusercontent.com/agentic-mcps/go/v1.1.0/scripts/install.sh | bash -s -- 1.1.0'; - instructions = 'Downloads and installs the latest pinned v1.1.0 release binaries to /usr/local/bin or ~/.local/bin.'; + command = 'curl -fsSL https://raw.githubusercontent.com/agentic-mcps/go/v1.2.0/scripts/install.sh | bash -s -- 1.2.0'; + instructions = 'Downloads and installs the latest pinned v1.2.0 release binaries to /usr/local/bin or ~/.local/bin.'; } else { - const archiveName = `agentic-go_1.1.0_${os}_${arch}.tar.gz`; - command = `curl -LO https://github.com/agentic-mcps/go/releases/download/v1.1.0/${archiveName}\ntar -xzf ${archiveName}\nsudo mv agentic-go agentic-go-gopls agentic-go-vet /usr/local/bin/`; + const archiveName = `agentic-go_1.2.0_${os}_${arch}.tar.gz`; + command = `curl -LO https://github.com/agentic-mcps/go/releases/download/v1.2.0/${archiveName}\ntar -xzf ${archiveName}\nsudo mv agentic-go agentic-go-gopls agentic-go-vet /usr/local/bin/`; instructions = `Direct binary archive installation for ${os}/${arch}.`; } @@ -920,7 +920,7 @@ function initWebMCP() { execute: async () => { const data = { brew: "brew install agentic-mcps/tap/agentic-go", - curl: "curl -fsSL https://raw.githubusercontent.com/agentic-mcps/go/v1.1.0/scripts/install.sh | bash -s -- 1.1.0", + curl: "curl -fsSL https://raw.githubusercontent.com/agentic-mcps/go/v1.2.0/scripts/install.sh | bash -s -- 1.2.0", mcpConfig: { mcpServers: { "agentic-go": { @@ -1000,7 +1000,7 @@ function initWebMCP() { }, { name: "get_tool_catalog", - description: "Returns the current v1.1.0 surface of agentic-go: 15 tools (14 frozen v1 tools plus additive go_context), 7 resources, 1 resource template, and 6 prompts.", + description: "Returns the current v1.2.0 surface of agentic-go: 15 tools (14 frozen v1 tools plus additive go_context), 7 resources, 1 resource template, and 6 prompts.", inputSchema: { type: "object", properties: {}, @@ -1008,7 +1008,7 @@ function initWebMCP() { }, execute: async () => { const catalog = { - releaseVersion: "1.1.0", + releaseVersion: "1.2.0", toolsCount: 15, frozenToolsCount: 14, tools: [ diff --git a/site/assets/brand/pills/release.svg b/site/assets/brand/pills/release.svg index 5779c3a..76e6423 100644 --- a/site/assets/brand/pills/release.svg +++ b/site/assets/brand/pills/release.svg @@ -1,6 +1,6 @@ - + - v1.1.0 + v1.2.0 diff --git a/site/docs/contracts/index.html b/site/docs/contracts/index.html index ba2b571..5bb054d 100644 --- a/site/docs/contracts/index.html +++ b/site/docs/contracts/index.html @@ -13,12 +13,12 @@ - + - + @@ -132,7 +132,7 @@

Protocol Contracts & Schemas

- agentic-go v1.1.0 preserves a frozen protocol surface consisting of 14 MCP tools, 7 resources, 1 resource template, 6 prompts, and three durable JSON schemas. The current server also exposes additive go_context under agentic.focus/v1. + agentic-go v1.2.0 preserves a frozen protocol surface consisting of 14 MCP tools, 7 resources, 1 resource template, 6 prompts, and three durable JSON schemas. The current server also exposes additive go_context under agentic.focus/v1.

Core JSON Schemas

@@ -255,7 +255,7 @@

Frozen MCP Tools Inventory (14 Tools)

Current release addition

- v1.1.0 exposes 15 current tools: the 14-tool frozen v1 inventory above plus go_context, a snapshot-bound context and verification-applicability tool. The additive tool does not change the frozen v1 schemas or existing tool contracts. + v1.2.0 exposes 15 current tools: the 14-tool frozen v1 inventory above plus go_context, a snapshot-bound context and verification-applicability tool. The additive tool does not change the frozen v1 schemas or existing tool contracts.

diff --git a/site/docs/index.html b/site/docs/index.html index 6c1652e..ece6dc3 100644 --- a/site/docs/index.html +++ b/site/docs/index.html @@ -177,7 +177,7 @@

Documentation Roadmap

Reference Protocol Contracts - Formal specifications for 14 frozen MCP tools, 7 resources, 6 prompts, and JSON schemas, plus the additive go_context tool in v1.1.0. + Formal specifications for 14 frozen MCP tools, 7 resources, 6 prompts, and JSON schemas, plus the additive go_context tool in v1.2.0. Frequently Asked Questions diff --git a/site/index.html b/site/index.html index 09a6966..ae0903a 100644 --- a/site/index.html +++ b/site/index.html @@ -53,7 +53,7 @@ "name": "agentic-go", "applicationCategory": "DeveloperApplication", "operatingSystem": "macOS, Linux", - "softwareVersion": "1.1.0", + "softwareVersion": "1.2.0", "license": "https://www.apache.org/licenses/LICENSE-2.0", "description": "Source-grounded Go code intelligence, change continuity, guarded refactoring, and executed verification for coding agents via Model Context Protocol (MCP).", "offers": { diff --git a/site/llms-full.txt b/site/llms-full.txt index 792b745..8b1dabe 100644 --- a/site/llms-full.txt +++ b/site/llms-full.txt @@ -44,7 +44,7 @@ agentic-go --version ### Standalone Shell Installer ```sh -curl -fsSL https://raw.githubusercontent.com/agentic-mcps/go/v1.1.0/scripts/install.sh | bash -s -- 1.1.0 +curl -fsSL https://raw.githubusercontent.com/agentic-mcps/go/v1.2.0/scripts/install.sh | bash -s -- 1.2.0 ``` ### MCP Configuration (Claude Desktop, Cursor, Cline) @@ -61,7 +61,7 @@ curl -fsSL https://raw.githubusercontent.com/agentic-mcps/go/v1.1.0/scripts/inst --- -## 4. MCP Surface Inventory (v1.1.0) +## 4. MCP Surface Inventory (v1.2.0) ### 15 Current Tools 1. `go_workspace_brief` (read-only): Returns top-level module map, package list, and entry points. diff --git a/site/llms.txt b/site/llms.txt index 1eb5a56..f52d926 100644 --- a/site/llms.txt +++ b/site/llms.txt @@ -21,10 +21,10 @@ agentic-go is a local Model Context Protocol (MCP) server and CLI for Go codebas - [Verification Engine](https://agentic-mcps.github.io/go/docs/verify/): Package-level test execution, race detection, coverage gaps, and verification reports. - [Safety & Containment](https://agentic-mcps.github.io/go/docs/safety/): Workspace boundary containment, non-destructive mutation guarantees, and execution permissions. - [GitHub Action in CI](https://agentic-mcps.github.io/go/docs/action/): Running agentic-go verification in GitHub Actions pull request workflows. -- [Protocol Contracts & Schemas](https://agentic-mcps.github.io/go/docs/contracts/): Frozen specification of 14 MCP tools, 7 resources, 6 prompts, and JSON schemas, plus the additive `go_context` tool in v1.1.0. +- [Protocol Contracts & Schemas](https://agentic-mcps.github.io/go/docs/contracts/): Frozen specification of 14 MCP tools, 7 resources, 6 prompts, and JSON schemas, plus the additive `go_context` tool in v1.2.0. - [Frequently Asked Questions](https://agentic-mcps.github.io/go/docs/faq/): Answers to common questions about gopls integration, sandboxing, and licensing. -## MCP Surface (v1.1.0) +## MCP Surface (v1.2.0) - **15 current tools:** `go_workspace_brief`, `go_search`, `go_symbol_context`, `go_begin_change`, `go_checkpoint_change`, `go_refactor`, `go_verify_change`, `go_test_structured`, `go_race_report`, `go_coverage_gaps`, `go_benchmark_diff`, `go_flake_finder`, `go_audit_concurrency`, `go_audit_errors`, and additive `go_context`. - **Frozen v1 baseline:** The first 14 tools remain the frozen v1 MCP contract. `go_context` is post-v1 functionality under `agentic.focus/v1` and does not change the frozen schemas. From 283caee2cd1352cca59f07e0bf5ce53d13c3c917 Mon Sep 17 00:00:00 2001 From: Ashwin Gopalsamy <47941624+ashwingopalsamy@users.noreply.github.com> Date: Wed, 23 Sep 2026 17:21:54 +0530 Subject: [PATCH 06/20] feat(intelligence): map verification evidence to next actions --- internal/intelligence/focus.go | 30 +++- internal/intelligence/focus_action.go | 192 +++++++++++++++++++++ internal/intelligence/focus_action_test.go | 133 ++++++++++++++ internal/intelligence/focus_test.go | 29 ++++ 4 files changed, 379 insertions(+), 5 deletions(-) create mode 100644 internal/intelligence/focus_action.go create mode 100644 internal/intelligence/focus_action_test.go diff --git a/internal/intelligence/focus.go b/internal/intelligence/focus.go index 869a412..f7d9cd2 100644 --- a/internal/intelligence/focus.go +++ b/internal/intelligence/focus.go @@ -143,6 +143,11 @@ type verificationFocusMetadata struct { Request focusPolicyIdentity `json:"request"` } +type verificationAssessment struct { + Report *verification.Report + Applicability VerificationApplicability +} + type focusPolicyIdentity struct { MinChangedCoverage *float64 `json:"min_changed_coverage,omitempty"` Base string `json:"base"` @@ -243,7 +248,7 @@ func (c *Core) Focus(ctx context.Context, request FocusRequest) (FocusResult, er if analysis.Repository.BaseCommit != observation.snapshot.BaseCommit || analysis.Repository.MergeBaseCommit != observation.snapshot.MergeBaseCommit || analysis.Repository.HeadCommit != observation.snapshot.HeadCommit { return FocusResult{}, fmt.Errorf("%w while analyzing change consequences", ErrSnapshotChanged) } - applicability, err := c.verificationApplicability(ctx, observation.snapshot, focusIdentity(request)) + assessment, err := c.assessVerificationApplicability(ctx, observation.snapshot, focusIdentity(request)) if err != nil { return FocusResult{}, err } @@ -256,9 +261,10 @@ func (c *Core) Focus(ctx context.Context, request FocusRequest) (FocusResult, er result := FocusResult{ SchemaVersion: FocusSchemaVersion, Provider: c.provider(), Snapshot: observation.snapshot, Change: analysis.Change, Impact: analysis.Impact, Risks: nonNilRisks(analysis.Risks), - Uncertainties: nonNilVerificationUncertainties(analysis.Uncertainties), Verification: applicability, + Uncertainties: nonNilVerificationUncertainties(analysis.Uncertainties), Verification: assessment.Applicability, ObservedPackages: analysis.ObservedPackages, Complete: analysis.Complete, Context: focused, Refresh: refresh, } + applyFocusAction(&result, assessment.Report) if focused != nil { selection := focused.Selection if previous != nil { @@ -898,12 +904,20 @@ func focusIdentity(request FocusRequest) focusPolicyIdentity { } func (c *Core) verificationApplicability(ctx context.Context, snapshot SnapshotRef, requested focusPolicyIdentity) (VerificationApplicability, error) { + assessment, err := c.assessVerificationApplicability(ctx, snapshot, requested) + if err != nil { + return VerificationApplicability{}, err + } + return assessment.Applicability, nil +} + +func (c *Core) assessVerificationApplicability(ctx context.Context, snapshot SnapshotRef, requested focusPolicyIdentity) (verificationAssessment, error) { report, metadata, err := c.verifications.currentFocus(ctx, snapshot.RepositoryID) if errors.Is(err, ErrVerificationNotFound) { - return VerificationApplicability{Reasons: []string{"no stored verification report exists for this repository"}, NextAction: "request verification for the current snapshot and policy"}, nil + return verificationAssessment{Applicability: VerificationApplicability{Reasons: []string{"no stored verification report exists for this repository"}, NextAction: "request verification for the current snapshot and policy"}}, nil } if err != nil { - return VerificationApplicability{}, err + return verificationAssessment{}, err } reasons := make([]string, 0) if metadata == nil { @@ -915,6 +929,9 @@ func (c *Core) verificationApplicability(ctx context.Context, snapshot SnapshotR if report.Snapshot.CurrentID != snapshot.ID { reasons = append(reasons, "workspace snapshot differs") } + if !reflect.DeepEqual(metadata.Snapshot, snapshot) || metadata.Snapshot.ID != report.Snapshot.CurrentID { + reasons = append(reasons, "stored applicability snapshot differs") + } if metadata.Request.Scope != requested.Scope { reasons = append(reasons, "package scope differs") } @@ -933,7 +950,10 @@ func (c *Core) verificationApplicability(ctx context.Context, snapshot SnapshotR if !applicable { next = "request verification for the current snapshot and policy" } - return VerificationApplicability{ReportID: report.ID, Outcome: report.Result.Status, Reasons: reasons, NextAction: next, Present: true, Applicable: applicable}, nil + return verificationAssessment{ + Applicability: VerificationApplicability{ReportID: report.ID, Outcome: report.Result.Status, Reasons: reasons, NextAction: next, Present: true, Applicable: applicable}, + Report: &report, + }, nil } func nonNilRisks(items []verification.RiskArea) []verification.RiskArea { diff --git a/internal/intelligence/focus_action.go b/internal/intelligence/focus_action.go new file mode 100644 index 0000000..d81c5a2 --- /dev/null +++ b/internal/intelligence/focus_action.go @@ -0,0 +1,192 @@ +package intelligence + +import ( + "path" + "path/filepath" + "strings" + + "github.com/agentic-mcps/go/internal/verification" +) + +type focusActionKind string + +const ( + verificationNeeded focusActionKind = "verification_needed" + findingInspectionNeeded focusActionKind = "finding_inspection_needed" + evidenceUnavailable focusActionKind = "evidence_unavailable" + requestedChecksPassedWithLimits focusActionKind = "requested_checks_passed_with_limits" +) + +type focusAction struct { + Kind focusActionKind + NextAction string + Reasons []string +} + +func applyFocusAction(result *FocusResult, report *verification.Report) { + if result == nil { + return + } + action := projectFocusAction(*result, report) + result.Verification.NextAction = action.NextAction + result.Verification.Reasons = appendFocusActionReasons(result.Verification.Reasons, action.Reasons...) +} + +func appendFocusActionReasons(existing []string, additions ...string) []string { + result := append([]string{}, existing...) + for _, addition := range additions { + found := false + for _, current := range result { + if current == addition { + found = true + break + } + } + if !found { + result = append(result, addition) + } + } + return result +} + +func projectFocusAction(result FocusResult, report *verification.Report) focusAction { + if !result.Verification.Applicable || report == nil { + return focusAction{Kind: verificationNeeded, NextAction: "request verification for the current snapshot and policy", Reasons: []string{"applicable verification evidence is unavailable"}} + } + if incompleteFocusEvidence(result, report) { + return focusAction{Kind: evidenceUnavailable, NextAction: "complete or refresh the requested evidence", Reasons: []string{"verification evidence is incomplete or unavailable"}} + } + if report.Result.Status == verification.ResultFindings { + if len(report.Findings) == 0 { + return focusAction{Kind: evidenceUnavailable, NextAction: "complete or refresh the requested evidence", Reasons: []string{"findings outcome has no retained findings"}} + } + for _, finding := range report.Findings { + if finding.Location == nil || !validFocusLocation(*finding.Location) { + return focusAction{Kind: evidenceUnavailable, NextAction: "refresh finding evidence with usable locations", Reasons: []string{"findings lack valid workspace-relative locations"}} + } + } + return focusAction{Kind: findingInspectionNeeded, NextAction: "inspect the reported finding locations", Reasons: []string{"verification reported findings with usable locations"}} + } + if len(report.Findings) > 0 || report.Result.BlockingFindings > 0 { + return focusAction{Kind: evidenceUnavailable, NextAction: "complete or refresh the requested evidence", Reasons: []string{"report findings do not match its finalized outcome"}} + } + if report.Result.Status == verification.ResultPass { + return focusAction{Kind: requestedChecksPassedWithLimits, NextAction: "review the requested checks and their stated limits", Reasons: []string{"requested checks passed; this does not establish task completion"}} + } + return focusAction{Kind: evidenceUnavailable, NextAction: "complete or refresh the requested evidence", Reasons: []string{"report does not establish a complete finding or pass outcome"}} +} + +func incompleteFocusEvidence(result FocusResult, report *verification.Report) bool { + if !result.Complete || len(result.Uncertainties) > 0 || + result.Change.FilesTruncated || result.Change.DeclarationsTruncated || result.Impact.PackagesTruncated || + report.Result.Status == verification.ResultIncomplete || report.FindingsTruncated || + report.Change.FilesTruncated || report.Change.DeclarationsTruncated || report.Impact.PackagesTruncated { + return true + } + planned := map[string]verification.Check{} + for _, check := range report.Plan { + if check.ID == "" || check.TargetsTruncated { + return true + } + if _, exists := planned[check.ID]; exists { + return true + } + planned[check.ID] = check + } + if len(planned) == 0 { + return true + } + seen := map[string]int{} + for _, evidence := range report.Evidence { + check, exists := planned[evidence.CheckID] + if evidence.CheckID == "" || !exists || evidence.Kind != check.Kind { + return true + } + seen[evidence.CheckID]++ + if evidence.Status != verification.EvidencePassed && evidence.Status != verification.EvidenceFailed && evidence.Status != verification.EvidenceSkipped && evidence.Status != verification.EvidenceError { + return true + } + if evidence.Status == verification.EvidenceSkipped || evidence.Status == verification.EvidenceError { + return true + } + if report.Result.Status == verification.ResultPass && evidence.Status == verification.EvidenceFailed { + return true + } + if evidence.Analysis != nil && evidence.Analysis.Unknown > 0 { + return true + } + if evidenceIncompleteDetails(evidence) { + return true + } + } + for id := range planned { + if seen[id] != 1 { + return true + } + } + for _, count := range seen { + if count != 1 { + return true + } + } + if report.Result.Status != verification.ResultPass && report.Result.Status != verification.ResultFindings { + return true + } + if result.Context != nil { + context := result.Context + if context.Truncated || len(context.EvidenceStates) == 0 || (context.TypedEvidence != nil && (!context.TypedEvidence.Complete || context.TypedEvidence.Truncated)) { + return true + } + for _, state := range context.EvidenceStates { + switch state.State { + case "examined_and_absent", "unexamined": + case "unavailable", "gathered_but_omitted", "": + return true + default: + return true + } + } + for _, uncertainty := range context.Uncertainties { + if strings.Contains(uncertainty.Code, "unavailable") || strings.Contains(uncertainty.Code, "omitted") || strings.Contains(uncertainty.Code, "partial") || strings.Contains(uncertainty.Code, "truncated") { + return true + } + } + } + for _, uncertainty := range report.Uncertainties { + if uncertainty.LocationsTruncated || strings.Contains(uncertainty.Code, "unknown") || strings.Contains(uncertainty.Code, "unavailable") { + return true + } + } + return false +} + +func evidenceIncompleteDetails(evidence verification.Evidence) bool { + if evidence.Tests != nil && (evidence.Tests.PackagesTruncated || evidence.Tests.NonpassingTruncated) || + evidence.Coverage != nil && evidence.Coverage.UncoveredTruncated || + evidence.Diagnostics != nil && evidence.Diagnostics.Truncated || + evidence.Contract != nil && evidence.Contract.ViolationsTruncated { + return true + } + switch evidence.Kind { + case verification.CheckTests: + return evidence.Tests == nil + case verification.CheckCoverage: + return evidence.Coverage == nil + case verification.CheckRace: + return evidence.Race == nil + case verification.CheckConcurrency, verification.CheckErrors: + return evidence.Analysis == nil || evidence.Analysis.Unknown > 0 + case verification.CheckDiagnostics: + return evidence.Diagnostics == nil + case verification.CheckContract: + return evidence.Contract == nil + default: + return true + } +} + +func validFocusLocation(location verification.Location) bool { + file := strings.ReplaceAll(filepath.ToSlash(location.File), `\`, "/") + clean := path.Clean(file) + return file != "" && !strings.HasSuffix(file, "/") && !filepath.IsAbs(location.File) && !strings.HasPrefix(file, "/") && location.Line > 0 && location.Col >= 0 && clean != "." && clean != ".." && !strings.HasPrefix(clean, "../") +} diff --git a/internal/intelligence/focus_action_test.go b/internal/intelligence/focus_action_test.go new file mode 100644 index 0000000..58f87dc --- /dev/null +++ b/internal/intelligence/focus_action_test.go @@ -0,0 +1,133 @@ +package intelligence + +import ( + "testing" + + "github.com/agentic-mcps/go/internal/verification" +) + +func TestProjectFocusAction(t *testing.T) { + base := func() (*FocusResult, *verification.Report) { + result := &FocusResult{Verification: VerificationApplicability{Applicable: true}, Complete: true} + report := &verification.Report{ + Plan: []verification.Check{{ID: "tests", Kind: verification.CheckTests, Required: true}}, + Evidence: []verification.Evidence{{CheckID: "tests", Kind: verification.CheckTests, Status: verification.EvidencePassed, Tests: &verification.TestSummary{Packages: []verification.TestPackageSummary{}, Nonpassing: []verification.TestCaseSummary{}}}}, + Findings: []verification.Finding{}, Result: verification.PolicyResult{Status: verification.ResultPass}, + } + return result, report + } + validContext := func(result *FocusResult) { + result.Context = &FocusContext{EvidenceStates: []EvidenceState{{Facet: "related_tests", State: "examined_and_absent"}}, TypedEvidence: &TypedEvidence{Complete: true}} + } + + tests := []struct { + name string + setup func(*FocusResult, **verification.Report) + want focusActionKind + }{ + {name: "no report", setup: func(_ *FocusResult, report **verification.Report) { *report = nil }, want: verificationNeeded}, + {name: "non-applicable report", setup: func(result *FocusResult, _ **verification.Report) { result.Verification.Applicable = false }, want: verificationNeeded}, + {name: "legacy-like result", setup: func(_ *FocusResult, report **verification.Report) { + *report = &verification.Report{Result: verification.PolicyResult{Status: verification.ResultPass}} + }, want: evidenceUnavailable}, + {name: "incomplete report", setup: func(_ *FocusResult, report **verification.Report) { + (*report).Result.Status = verification.ResultIncomplete + }, want: evidenceUnavailable}, + {name: "incomplete focus", setup: func(result *FocusResult, _ **verification.Report) { result.Complete = false }, want: evidenceUnavailable}, + {name: "uncertain focus", setup: func(result *FocusResult, _ **verification.Report) { + result.Uncertainties = []verification.Uncertainty{{Code: "change.incomplete"}} + }, want: evidenceUnavailable}, + {name: "truncated changed files", setup: func(_ *FocusResult, report **verification.Report) { + (*report).Change.FilesTruncated = true + }, want: evidenceUnavailable}, + {name: "truncated current changed files", setup: func(result *FocusResult, _ **verification.Report) { + result.Change.FilesTruncated = true + }, want: evidenceUnavailable}, + {name: "truncated declarations", setup: func(_ *FocusResult, report **verification.Report) { + (*report).Change.DeclarationsTruncated = true + }, want: evidenceUnavailable}, + {name: "truncated current declarations", setup: func(result *FocusResult, _ **verification.Report) { + result.Change.DeclarationsTruncated = true + }, want: evidenceUnavailable}, + {name: "truncated impact", setup: func(_ *FocusResult, report **verification.Report) { + (*report).Impact.PackagesTruncated = true + }, want: evidenceUnavailable}, + {name: "truncated current impact", setup: func(result *FocusResult, _ **verification.Report) { + result.Impact.PackagesTruncated = true + }, want: evidenceUnavailable}, + {name: "missing required evidence", setup: func(_ *FocusResult, report **verification.Report) { (*report).Evidence = nil }, want: evidenceUnavailable}, + {name: "duplicate required evidence", setup: func(_ *FocusResult, report **verification.Report) { + (*report).Evidence = append((*report).Evidence, (*report).Evidence[0]) + }, want: evidenceUnavailable}, + {name: "missing optional evidence", setup: func(_ *FocusResult, report **verification.Report) { + (*report).Plan = append((*report).Plan, verification.Check{ID: "race", Kind: verification.CheckRace}) + }, want: evidenceUnavailable}, + {name: "truncated check targets", setup: func(_ *FocusResult, report **verification.Report) { + (*report).Plan[0].TargetsTruncated = true + }, want: evidenceUnavailable}, + {name: "skipped optional evidence", setup: func(_ *FocusResult, report **verification.Report) { + (*report).Plan = append((*report).Plan, verification.Check{ID: "race", Kind: verification.CheckRace}) + (*report).Evidence = append((*report).Evidence, verification.Evidence{CheckID: "race", Kind: verification.CheckRace, Status: verification.EvidenceSkipped}) + }, want: evidenceUnavailable}, + {name: "skipped evidence", setup: func(_ *FocusResult, report **verification.Report) { + (*report).Evidence[0].Status = verification.EvidenceSkipped + }, want: evidenceUnavailable}, + {name: "errored evidence", setup: func(_ *FocusResult, report **verification.Report) { + (*report).Evidence[0].Status = verification.EvidenceError + }, want: evidenceUnavailable}, + {name: "truncated evidence", setup: func(_ *FocusResult, report **verification.Report) { (*report).FindingsTruncated = true }, want: evidenceUnavailable}, + {name: "unknown analysis", setup: func(_ *FocusResult, report **verification.Report) { + (*report).Evidence[0].Kind = verification.CheckErrors + (*report).Evidence[0].Analysis = &verification.AnalysisSummary{Unknown: 1} + }, want: evidenceUnavailable}, + {name: "mismatched evidence kind", setup: func(_ *FocusResult, report **verification.Report) { + (*report).Evidence[0].Kind = verification.CheckErrors + (*report).Evidence[0].Analysis = &verification.AnalysisSummary{} + }, want: evidenceUnavailable}, + {name: "valid findings", setup: func(_ *FocusResult, report **verification.Report) { + (*report).Findings = []verification.Finding{{Location: &verification.Location{File: "pkg/file.go", Line: 3}}} + (*report).Result.Status = verification.ResultFindings + }, want: findingInspectionNeeded}, + {name: "invalid finding locations", setup: func(_ *FocusResult, report **verification.Report) { + (*report).Findings = []verification.Finding{{Location: &verification.Location{File: "../outside.go", Line: 3}}} + (*report).Result.Status = verification.ResultFindings + }, want: evidenceUnavailable}, + {name: "workspace directory finding location", setup: func(_ *FocusResult, report **verification.Report) { + (*report).Findings = []verification.Finding{{Location: &verification.Location{File: ".", Line: 3}}} + (*report).Result.Status = verification.ResultFindings + }, want: evidenceUnavailable}, + {name: "negative column finding location", setup: func(_ *FocusResult, report **verification.Report) { + (*report).Findings = []verification.Finding{{Location: &verification.Location{File: "pkg/file.go", Line: 3, Col: -1}}} + (*report).Result.Status = verification.ResultFindings + }, want: evidenceUnavailable}, + {name: "mixed finding locations", setup: func(_ *FocusResult, report **verification.Report) { + (*report).Findings = []verification.Finding{ + {Location: &verification.Location{File: "pkg/file.go", Line: 3}}, + {Location: &verification.Location{File: "../outside.go", Line: 4}}, + } + (*report).Result.Status = verification.ResultFindings + }, want: evidenceUnavailable}, + {name: "unknown context state", setup: func(result *FocusResult, _ **verification.Report) { + validContext(result) + result.Context.EvidenceStates[0].State = "unknown" + }, want: evidenceUnavailable}, + {name: "pass", setup: func(result *FocusResult, _ **verification.Report) { validContext(result) }, want: requestedChecksPassedWithLimits}, + {name: "valid typed and context evidence", setup: func(result *FocusResult, _ **verification.Report) { + validContext(result) + result.Context.TypedEvidence.Relationships = []GoRelationship{} + }, want: requestedChecksPassedWithLimits}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result, report := base() + tt.setup(result, &report) + got := projectFocusAction(*result, report) + if got.Kind != tt.want { + t.Fatalf("kind = %q, want %q", got.Kind, tt.want) + } + if got.NextAction == "" || got.Reasons == nil || len(got.Reasons) == 0 { + t.Fatalf("action lacks concise guidance: %#v", got) + } + }) + } +} diff --git a/internal/intelligence/focus_test.go b/internal/intelligence/focus_test.go index d498e9d..f033a16 100644 --- a/internal/intelligence/focus_test.go +++ b/internal/intelligence/focus_test.go @@ -244,6 +244,35 @@ func TestVerificationApplicabilityHandlesMissingAndLegacyMetadata(t *testing.T) } } +func TestVerificationApplicabilityRejectsMismatchedStoredSnapshotMetadata(t *testing.T) { + root := snapshotRepository(t) + snapshotter := newTestSnapshotter(t, root) + core := newTestCore(t, snapshotter, &fakeSemanticReader{}) + snapshot, err := core.capture(context.Background(), "HEAD", "./...", "") + if err != nil { + t.Fatal(err) + } + request := normalizedFocusRequest(t, FocusRequest{Base: "HEAD", Scope: "./..."}) + report := verification.NewReport("test", verification.Repository{SnapshotID: snapshot.ID}) + report.Snapshot.CurrentID = snapshot.ID + if err := report.Finalize(verification.Policy{}); err != nil { + t.Fatal(err) + } + metadata := verificationFocusMetadata{Snapshot: snapshot, Request: focusIdentity(request)} + metadata.Snapshot.ContentDigest = "sha256:mismatched" + if err := core.verifications.saveFocus(context.Background(), snapshot.RepositoryID, report, metadata); err != nil { + t.Fatal(err) + } + + got, err := core.verificationApplicability(context.Background(), snapshot, focusIdentity(request)) + if err != nil { + t.Fatal(err) + } + if got.Applicable || !containsReason(got.Reasons, "stored applicability snapshot") { + t.Fatalf("verificationApplicability() = %#v, want stored snapshot mismatch", got) + } +} + func TestFocusContextReturnsAmbiguousCandidatesBeforeExpansion(t *testing.T) { root := snapshotRepository(t) snapshotter := newTestSnapshotter(t, root) From 67f54b76d25938de1dd225e636bb366f3e088557 Mon Sep 17 00:00:00 2001 From: Ashwin Gopalsamy <47941624+ashwingopalsamy@users.noreply.github.com> Date: Wed, 23 Sep 2026 17:54:54 +0530 Subject: [PATCH 07/20] docs(release): prepare v1.2.1 --- CHANGELOG.md | 15 ++++++++++++++- README.md | 10 +++++----- assets/brand/pills/release.svg | 4 ++-- docs/module-migration.md | 6 +++--- llms.txt | 2 +- site/app.js | 16 ++++++++-------- site/assets/brand/pills/release.svg | 4 ++-- site/docs/contracts/index.html | 8 ++++---- site/docs/index.html | 2 +- site/index.html | 2 +- site/llms-full.txt | 4 ++-- site/llms.txt | 4 ++-- 12 files changed, 45 insertions(+), 32 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 01a20c6..fa581ed 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,18 @@ All notable changes to this project are documented here. The format follows ## [Unreleased] +## [1.2.1] - 2026-09-23 + +### Added + +- Deterministic `go_context` evidence-to-action guidance for verification, + finding inspection, unavailable evidence, and bounded passing checks. + +### Fixed + +- Stale, incomplete, truncated, mismatched, and invalid-location evidence now + fails closed before producing current action guidance. + ## [1.2.0] - 2026-09-23 ### Added @@ -92,7 +104,8 @@ All notable changes to this project are documented here. The format follows - Workspace containment, bounded execution, event-driven progress, and optional privacy-preserving local traces. -[Unreleased]: https://github.com/agentic-mcps/go/compare/v1.2.0...HEAD +[Unreleased]: https://github.com/agentic-mcps/go/compare/v1.2.1...HEAD +[1.2.1]: https://github.com/agentic-mcps/go/releases/tag/v1.2.1 [1.2.0]: https://github.com/agentic-mcps/go/releases/tag/v1.2.0 [1.1.0]: https://github.com/agentic-mcps/go/releases/tag/v1.1.0 [1.0.0]: https://github.com/agentic-mcps/go/releases/tag/v1.0.0 diff --git a/README.md b/README.md index 9c4102a..3a4c8cf 100644 --- a/README.md +++ b/README.md @@ -10,14 +10,14 @@ Install agentic-go Connect MCP Read docs - v1.2.0 release + v1.2.1 release

Website · Docs · Install · Connect · Workflow · Capabilities · FAQ

`agentic-go` is a local Go MCP server and CLI. It gives an external coding agent semantic context, change continuity, guarded refactoring, and executed verification without embedding an LLM or becoming an agent framework. -The v1.2.0 server exposes 15 MCP tools: the frozen v1 surface of 14 tools plus the additive `go_context` tool under `agentic.focus/v1`. This release hardens snapshot observation, focus applicability evidence, and bounded private tracing. The seven resources, resource template, six prompts, and frozen v1 schemas remain unchanged. +The v1.2.1 server exposes 15 MCP tools: the frozen v1 surface of 14 tools plus the additive `go_context` tool under `agentic.focus/v1`. This patch release makes the edit, refresh, verify, and inspect handoff explicit while preserving snapshot lineage and fail-closed evidence. The seven resources, resource template, six prompts, and frozen v1 schemas remain unchanged. ## Install @@ -28,7 +28,7 @@ brew install agentic-mcps/tap/agentic-go agentic-go --version ``` -That installs `agentic-go`, the pinned `agentic-go-gopls` companion, and `agentic-go-vet`. The Homebrew tap is maintained separately; the signed v1.2.0 release archive and checksum installer below are the canonical versioned distribution path. +That installs `agentic-go`, the pinned `agentic-go-gopls` companion, and `agentic-go-vet`. The Homebrew tap is maintained separately; the signed v1.2.1 release archive and checksum installer below are the canonical versioned distribution path. For an agent workflow, install the binary first, then print the client-native MCP entry for the current workspace: @@ -44,8 +44,8 @@ edit your client configuration. Install from the release archive instead ```sh -curl -fsSL https://raw.githubusercontent.com/agentic-mcps/go/v1.2.0/scripts/install.sh \ - | bash -s -- 1.2.0 +curl -fsSL https://raw.githubusercontent.com/agentic-mcps/go/v1.2.1/scripts/install.sh \ + | bash -s -- 1.2.1 ``` The installer places the binaries in `~/.local/bin` and verifies the release checksum before replacing them. diff --git a/assets/brand/pills/release.svg b/assets/brand/pills/release.svg index 76e6423..b063b4c 100644 --- a/assets/brand/pills/release.svg +++ b/assets/brand/pills/release.svg @@ -1,6 +1,6 @@ - + - v1.2.0 + v1.2.1 diff --git a/docs/module-migration.md b/docs/module-migration.md index edd19df..decc5a0 100644 --- a/docs/module-migration.md +++ b/docs/module-migration.md @@ -16,8 +16,8 @@ Install the organization release over the existing binary names: ```sh curl --fail --location --silent --show-error \ - https://raw.githubusercontent.com/agentic-mcps/go/v1.2.0/scripts/install.sh \ - | bash -s -- 1.2.0 + https://raw.githubusercontent.com/agentic-mcps/go/v1.2.1/scripts/install.sh \ + | bash -s -- 1.2.1 agentic-go --version agentic-go doctor @@ -26,7 +26,7 @@ agentic-go doctor Update GitHub Actions references to: ```yaml -- uses: agentic-mcps/go@v1.2.0 +- uses: agentic-mcps/go@v1.2.1 ``` Update source-checkout or advanced `go install` commands to use diff --git a/llms.txt b/llms.txt index c909613..ab0d82a 100644 --- a/llms.txt +++ b/llms.txt @@ -7,7 +7,7 @@ This is a convenience index for humans and coding agents. It is not a ranking or ## Start - [README](https://github.com/agentic-mcps/go/blob/main/README.md): installation, MCP setup, workflow, capabilities, compatibility, and trust boundary. -- [v1.2.0 release](https://github.com/agentic-mcps/go/releases/tag/v1.2.0): supported archives and checksums. +- [v1.2.1 release](https://github.com/agentic-mcps/go/releases/tag/v1.2.1): supported archives and checksums. - [Module migration](https://github.com/agentic-mcps/go/blob/main/docs/module-migration.md): separate personal and organization module identities. ## Contracts diff --git a/site/app.js b/site/app.js index b0fb4ea..383a819 100644 --- a/site/app.js +++ b/site/app.js @@ -516,7 +516,7 @@ const DOCS_SEARCH_INDEX = [ title: "Protocol Contracts & Schemas", path: "/go/docs/contracts/", relPath: "docs/contracts/", - summary: "Frozen v1 specification for 14 MCP tools, 7 resources, 6 prompts, and JSON schemas, plus the additive go_context tool in v1.2.0.", + summary: "Frozen v1 specification for 14 MCP tools, 7 resources, 6 prompts, and JSON schemas, plus the additive go_context tool in v1.2.1.", keywords: "contracts, schemas, protocol, tools, 14 tools, 15 tools, go_context, resources, prompts, agentic.context/v1, agentic.change/v1, agentic.verify/v1, agentic.focus/v1" }, { @@ -689,11 +689,11 @@ function generateInstallCommandPayload(params = {}) { command = 'brew install agentic-mcps/tap/agentic-go\nagentic-go --version'; instructions = 'Installs agentic-go, agentic-go-gopls companion, and agentic-go-vet via official Homebrew tap.'; } else if (method === 'curl') { - command = 'curl -fsSL https://raw.githubusercontent.com/agentic-mcps/go/v1.2.0/scripts/install.sh | bash -s -- 1.2.0'; - instructions = 'Downloads and installs the latest pinned v1.2.0 release binaries to /usr/local/bin or ~/.local/bin.'; + command = 'curl -fsSL https://raw.githubusercontent.com/agentic-mcps/go/v1.2.1/scripts/install.sh | bash -s -- 1.2.1'; + instructions = 'Downloads and installs the latest pinned v1.2.1 release binaries to /usr/local/bin or ~/.local/bin.'; } else { - const archiveName = `agentic-go_1.2.0_${os}_${arch}.tar.gz`; - command = `curl -LO https://github.com/agentic-mcps/go/releases/download/v1.2.0/${archiveName}\ntar -xzf ${archiveName}\nsudo mv agentic-go agentic-go-gopls agentic-go-vet /usr/local/bin/`; + const archiveName = `agentic-go_1.2.1_${os}_${arch}.tar.gz`; + command = `curl -LO https://github.com/agentic-mcps/go/releases/download/v1.2.1/${archiveName}\ntar -xzf ${archiveName}\nsudo mv agentic-go agentic-go-gopls agentic-go-vet /usr/local/bin/`; instructions = `Direct binary archive installation for ${os}/${arch}.`; } @@ -920,7 +920,7 @@ function initWebMCP() { execute: async () => { const data = { brew: "brew install agentic-mcps/tap/agentic-go", - curl: "curl -fsSL https://raw.githubusercontent.com/agentic-mcps/go/v1.2.0/scripts/install.sh | bash -s -- 1.2.0", + curl: "curl -fsSL https://raw.githubusercontent.com/agentic-mcps/go/v1.2.1/scripts/install.sh | bash -s -- 1.2.1", mcpConfig: { mcpServers: { "agentic-go": { @@ -1000,7 +1000,7 @@ function initWebMCP() { }, { name: "get_tool_catalog", - description: "Returns the current v1.2.0 surface of agentic-go: 15 tools (14 frozen v1 tools plus additive go_context), 7 resources, 1 resource template, and 6 prompts.", + description: "Returns the current v1.2.1 surface of agentic-go: 15 tools (14 frozen v1 tools plus additive go_context), 7 resources, 1 resource template, and 6 prompts.", inputSchema: { type: "object", properties: {}, @@ -1008,7 +1008,7 @@ function initWebMCP() { }, execute: async () => { const catalog = { - releaseVersion: "1.2.0", + releaseVersion: "1.2.1", toolsCount: 15, frozenToolsCount: 14, tools: [ diff --git a/site/assets/brand/pills/release.svg b/site/assets/brand/pills/release.svg index 76e6423..b063b4c 100644 --- a/site/assets/brand/pills/release.svg +++ b/site/assets/brand/pills/release.svg @@ -1,6 +1,6 @@ - + - v1.2.0 + v1.2.1 diff --git a/site/docs/contracts/index.html b/site/docs/contracts/index.html index 5bb054d..904c77c 100644 --- a/site/docs/contracts/index.html +++ b/site/docs/contracts/index.html @@ -13,12 +13,12 @@ - + - + @@ -132,7 +132,7 @@

Protocol Contracts & Schemas

- agentic-go v1.2.0 preserves a frozen protocol surface consisting of 14 MCP tools, 7 resources, 1 resource template, 6 prompts, and three durable JSON schemas. The current server also exposes additive go_context under agentic.focus/v1. + agentic-go v1.2.1 preserves a frozen protocol surface consisting of 14 MCP tools, 7 resources, 1 resource template, 6 prompts, and three durable JSON schemas. The current server also exposes additive go_context under agentic.focus/v1.

Core JSON Schemas

@@ -255,7 +255,7 @@

Frozen MCP Tools Inventory (14 Tools)

Current release addition

- v1.2.0 exposes 15 current tools: the 14-tool frozen v1 inventory above plus go_context, a snapshot-bound context and verification-applicability tool. The additive tool does not change the frozen v1 schemas or existing tool contracts. + v1.2.1 exposes 15 current tools: the 14-tool frozen v1 inventory above plus go_context, a snapshot-bound context and verification-applicability tool. The additive tool does not change the frozen v1 schemas or existing tool contracts.

diff --git a/site/docs/index.html b/site/docs/index.html index ece6dc3..155c858 100644 --- a/site/docs/index.html +++ b/site/docs/index.html @@ -177,7 +177,7 @@

Documentation Roadmap

Reference Protocol Contracts - Formal specifications for 14 frozen MCP tools, 7 resources, 6 prompts, and JSON schemas, plus the additive go_context tool in v1.2.0. + Formal specifications for 14 frozen MCP tools, 7 resources, 6 prompts, and JSON schemas, plus the additive go_context tool in v1.2.1. Frequently Asked Questions diff --git a/site/index.html b/site/index.html index ae0903a..43cd422 100644 --- a/site/index.html +++ b/site/index.html @@ -53,7 +53,7 @@ "name": "agentic-go", "applicationCategory": "DeveloperApplication", "operatingSystem": "macOS, Linux", - "softwareVersion": "1.2.0", + "softwareVersion": "1.2.1", "license": "https://www.apache.org/licenses/LICENSE-2.0", "description": "Source-grounded Go code intelligence, change continuity, guarded refactoring, and executed verification for coding agents via Model Context Protocol (MCP).", "offers": { diff --git a/site/llms-full.txt b/site/llms-full.txt index 8b1dabe..477856b 100644 --- a/site/llms-full.txt +++ b/site/llms-full.txt @@ -44,7 +44,7 @@ agentic-go --version ### Standalone Shell Installer ```sh -curl -fsSL https://raw.githubusercontent.com/agentic-mcps/go/v1.2.0/scripts/install.sh | bash -s -- 1.2.0 +curl -fsSL https://raw.githubusercontent.com/agentic-mcps/go/v1.2.1/scripts/install.sh | bash -s -- 1.2.1 ``` ### MCP Configuration (Claude Desktop, Cursor, Cline) @@ -61,7 +61,7 @@ curl -fsSL https://raw.githubusercontent.com/agentic-mcps/go/v1.2.0/scripts/inst --- -## 4. MCP Surface Inventory (v1.2.0) +## 4. MCP Surface Inventory (v1.2.1) ### 15 Current Tools 1. `go_workspace_brief` (read-only): Returns top-level module map, package list, and entry points. diff --git a/site/llms.txt b/site/llms.txt index f52d926..0fcdabe 100644 --- a/site/llms.txt +++ b/site/llms.txt @@ -21,10 +21,10 @@ agentic-go is a local Model Context Protocol (MCP) server and CLI for Go codebas - [Verification Engine](https://agentic-mcps.github.io/go/docs/verify/): Package-level test execution, race detection, coverage gaps, and verification reports. - [Safety & Containment](https://agentic-mcps.github.io/go/docs/safety/): Workspace boundary containment, non-destructive mutation guarantees, and execution permissions. - [GitHub Action in CI](https://agentic-mcps.github.io/go/docs/action/): Running agentic-go verification in GitHub Actions pull request workflows. -- [Protocol Contracts & Schemas](https://agentic-mcps.github.io/go/docs/contracts/): Frozen specification of 14 MCP tools, 7 resources, 6 prompts, and JSON schemas, plus the additive `go_context` tool in v1.2.0. +- [Protocol Contracts & Schemas](https://agentic-mcps.github.io/go/docs/contracts/): Frozen specification of 14 MCP tools, 7 resources, 6 prompts, and JSON schemas, plus the additive `go_context` tool in v1.2.1. - [Frequently Asked Questions](https://agentic-mcps.github.io/go/docs/faq/): Answers to common questions about gopls integration, sandboxing, and licensing. -## MCP Surface (v1.2.0) +## MCP Surface (v1.2.1) - **15 current tools:** `go_workspace_brief`, `go_search`, `go_symbol_context`, `go_begin_change`, `go_checkpoint_change`, `go_refactor`, `go_verify_change`, `go_test_structured`, `go_race_report`, `go_coverage_gaps`, `go_benchmark_diff`, `go_flake_finder`, `go_audit_concurrency`, `go_audit_errors`, and additive `go_context`. - **Frozen v1 baseline:** The first 14 tools remain the frozen v1 MCP contract. `go_context` is post-v1 functionality under `agentic.focus/v1` and does not change the frozen schemas. From 8e65ce10ffe41967293cb2f5c3f759f19a2c292c Mon Sep 17 00:00:00 2001 From: Ashwin Gopalsamy <47941624+ashwingopalsamy@users.noreply.github.com> Date: Wed, 23 Sep 2026 18:34:25 +0530 Subject: [PATCH 08/20] feat(mcp): expose context next actions in text --- internal/tools/intelligence_tools.go | 2 +- internal/tools/intelligence_tools_test.go | 22 +++--- internal/tools/mcp_surface_golden_test.go | 87 +++++++++++++++++++---- 3 files changed, 87 insertions(+), 24 deletions(-) diff --git a/internal/tools/intelligence_tools.go b/internal/tools/intelligence_tools.go index 3542911..4e4899d 100644 --- a/internal/tools/intelligence_tools.go +++ b/internal/tools/intelligence_tools.go @@ -190,7 +190,7 @@ func (r *Runtime) context(ctx context.Context, _ *mcp.CallToolRequest, input Con if err != nil { return nil, intelligence.FocusResult{}, fmt.Errorf("building change context: %w", err) } - text := intelligence.FocusSummary(result) + "; canonical evidence is in structuredContent" + text := intelligence.FocusSummary(result) + "; canonical evidence is in structuredContent; next_action: " + result.Verification.NextAction return &mcp.CallToolResult{Content: []mcp.Content{&mcp.TextContent{Text: text}}}, result, nil } diff --git a/internal/tools/intelligence_tools_test.go b/internal/tools/intelligence_tools_test.go index f93498e..d68e53c 100644 --- a/internal/tools/intelligence_tools_test.go +++ b/internal/tools/intelligence_tools_test.go @@ -16,15 +16,16 @@ import ( ) type fakeIntelligence struct { //nolint:govet // Test requests are grouped by operation. - brief intelligence.BriefRequest - search intelligence.SearchRequest - symbol intelligence.SymbolRequest - begin intelligence.BeginRequest - checkpoint intelligence.CheckpointRequest - refactor intelligence.RefactorRequest - verify verification.Request - focus intelligence.FocusRequest - focusErr error + brief intelligence.BriefRequest + search intelligence.SearchRequest + symbol intelligence.SymbolRequest + begin intelligence.BeginRequest + checkpoint intelligence.CheckpointRequest + refactor intelligence.RefactorRequest + verify verification.Request + focus intelligence.FocusRequest + focusResult *intelligence.FocusResult + focusErr error } func (f *fakeIntelligence) Brief(_ context.Context, request intelligence.BriefRequest) (intelligence.ContextPack, error) { @@ -47,6 +48,9 @@ func (f *fakeIntelligence) Focus(_ context.Context, request intelligence.FocusRe if f.focusErr != nil { return intelligence.FocusResult{}, f.focusErr } + if f.focusResult != nil { + return *f.focusResult, nil + } return intelligence.FocusResult{SchemaVersion: intelligence.FocusSchemaVersion, Snapshot: intelligence.SnapshotRef{ID: "snap-focus"}, Change: verification.Change{Files: []verification.ChangedFile{}, Declarations: []verification.ChangedDeclaration{}, FilesTotal: 1}, Impact: verification.Impact{Packages: []verification.ImpactedPackage{}, PackagesTotal: 2}, Risks: []verification.RiskArea{}, Uncertainties: []verification.Uncertainty{}, Verification: intelligence.VerificationApplicability{Reasons: []string{}}, PackID: strings.Repeat("c", 64), Refresh: &intelligence.FocusRefresh{Status: "replaced"}}, nil } diff --git a/internal/tools/mcp_surface_golden_test.go b/internal/tools/mcp_surface_golden_test.go index ed95924..aa5bdca 100644 --- a/internal/tools/mcp_surface_golden_test.go +++ b/internal/tools/mcp_surface_golden_test.go @@ -12,6 +12,8 @@ import ( "strings" "testing" + "github.com/agentic-mcps/go/internal/intelligence" + "github.com/agentic-mcps/go/internal/verification" "github.com/modelcontextprotocol/go-sdk/mcp" ) @@ -85,7 +87,8 @@ func TestPostV1FocusToolIsAdditiveAndDiscoverable(t *testing.T) { ctx := context.Background() server := mcp.NewServer(&mcp.Implementation{Name: "agentic-go-focus", Version: "dev"}, &mcp.ServerOptions{Capabilities: &mcp.ServerCapabilities{}}) runtime := newTestRuntime(t) - runtime.intelligence = &fakeIntelligence{} + fake := &fakeIntelligence{} + runtime.intelligence = fake RegisterAll(server, runtime) RegisterContext(server, runtime) clientTransport, serverTransport := mcp.NewInMemoryTransports() @@ -125,20 +128,76 @@ func TestPostV1FocusToolIsAdditiveAndDiscoverable(t *testing.T) { t.Fatalf("go_context input schema lacks %q: %s", field, encoded) } } - result, callErr := clientSession.CallTool(ctx, &mcp.CallToolParams{Name: "go_context", Arguments: map[string]any{"base": "HEAD", "query": "Worker"}}) - if callErr != nil || result.IsError { - t.Fatalf("go_context call error=%v result=%#v", callErr, result) - } - payload, marshalErr := json.Marshal(result.StructuredContent) - if marshalErr != nil { - t.Fatal(marshalErr) + actions := []struct { + name string + outcome verification.ResultStatus + nextAction string + present bool + applicable bool + }{ + {name: "non-applicable", nextAction: "request verification for the current snapshot and policy"}, + {name: "findings", outcome: verification.ResultFindings, present: true, applicable: true, nextAction: "inspect the reported finding locations"}, + {name: "unavailable", outcome: verification.ResultIncomplete, present: true, applicable: true, nextAction: "complete or refresh the requested evidence"}, + {name: "bounded-pass", outcome: verification.ResultPass, present: true, applicable: true, nextAction: "review the requested checks and their stated limits"}, } - var focus struct { - SchemaVersion string `json:"schema_version"` - PackID string `json:"pack_id"` - } - if err := json.Unmarshal(payload, &focus); err != nil || focus.SchemaVersion != "agentic.focus/v1" || focus.PackID == "" { - t.Fatalf("go_context structured output = %s, err %v", payload, err) + for _, action := range actions { + t.Run(action.name, func(t *testing.T) { + privateReason := strings.Repeat("private verification reason ", 64) + fake.focusResult = &intelligence.FocusResult{ + SchemaVersion: intelligence.FocusSchemaVersion, + Snapshot: intelligence.SnapshotRef{ID: "snap-focus"}, + Change: verification.Change{Files: []verification.ChangedFile{}, Declarations: []verification.ChangedDeclaration{}, FilesTotal: 1}, + Impact: verification.Impact{Packages: []verification.ImpactedPackage{}, PackagesTotal: 2}, + Risks: []verification.RiskArea{}, + Uncertainties: []verification.Uncertainty{}, + Verification: intelligence.VerificationApplicability{ + Outcome: action.outcome, Reasons: []string{privateReason}, NextAction: action.nextAction, + Present: action.present, Applicable: action.applicable, + }, + PackID: strings.Repeat("c", 64), + } + + result, callErr := clientSession.CallTool(ctx, &mcp.CallToolParams{Name: "go_context", Arguments: map[string]any{"base": "HEAD", "query": "Worker"}}) + if callErr != nil || result.IsError { + t.Fatalf("go_context call error=%v result=%#v", callErr, result) + } + if len(result.Content) != 1 { + t.Fatalf("go_context text content count = %d, want 1", len(result.Content)) + } + content, ok := result.Content[0].(*mcp.TextContent) + if !ok { + t.Fatalf("go_context content = %T, want text", result.Content[0]) + } + if !strings.Contains(content.Text, "next_action: "+action.nextAction) { + t.Errorf("go_context text %q missing next action %q", content.Text, action.nextAction) + } + if strings.Contains(content.Text, "private verification reason") || len(content.Text) > 512 { + t.Errorf("go_context text is not concise or includes reasons: length=%d", len(content.Text)) + } + + payload, marshalErr := json.Marshal(result.StructuredContent) + if marshalErr != nil { + t.Fatal(marshalErr) + } + var focus struct { + SchemaVersion string `json:"schema_version"` + PackID string `json:"pack_id"` + Verification struct { + Outcome verification.ResultStatus `json:"outcome"` + NextAction string `json:"next_action"` + Reasons []string `json:"reasons"` + Present bool `json:"present"` + Applicable bool `json:"applicable"` + } `json:"verification"` + } + if err := json.Unmarshal(payload, &focus); err != nil || focus.SchemaVersion != "agentic.focus/v1" || focus.PackID == "" { + t.Fatalf("go_context structured output = %s, err %v", payload, err) + } + if focus.Verification.Outcome != action.outcome || focus.Verification.Present != action.present || focus.Verification.Applicable != action.applicable || + focus.Verification.NextAction != action.nextAction || len(focus.Verification.Reasons) != 1 || focus.Verification.Reasons[0] != privateReason { + t.Fatalf("go_context structured verification = %#v, want action %#v and original reason", focus.Verification, action) + } + }) } return } From 71099dcbcb24831cc60644437239b35c2972e868 Mon Sep 17 00:00:00 2001 From: Ashwin Gopalsamy <47941624+ashwingopalsamy@users.noreply.github.com> Date: Wed, 23 Sep 2026 19:01:12 +0530 Subject: [PATCH 09/20] fix(intelligence): clarify truncated evidence guidance --- internal/intelligence/focus_action.go | 276 +++++++++++++++++++-- internal/intelligence/focus_action_test.go | 233 ++++++++++++++++- internal/intelligence/focus_test.go | 4 + 3 files changed, 488 insertions(+), 25 deletions(-) diff --git a/internal/intelligence/focus_action.go b/internal/intelligence/focus_action.go index d81c5a2..dce90c8 100644 --- a/internal/intelligence/focus_action.go +++ b/internal/intelligence/focus_action.go @@ -17,6 +17,14 @@ const ( requestedChecksPassedWithLimits focusActionKind = "requested_checks_passed_with_limits" ) +const ( + verifyCurrentSnapshotAction = "request verification for the current snapshot and matching policy" + freshSelectionAndVerifyAction = "make a fresh selection against the current snapshot, then request verification for that snapshot and matching policy" + focusBudgetAction = "make a fresh narrower selection or use a larger max_bytes on a new selection; replacement refresh inherits the prior budget and cannot fix the omission" + reportTruncationAction = "treat omitted verification report detail as unavailable; refreshing cannot recover the omitted detail" + unavailableEvidenceAction = "treat affected evidence as unavailable; retry recovery is not established" +) + type focusAction struct { Kind focusActionKind NextAction string @@ -51,36 +59,276 @@ func appendFocusActionReasons(existing []string, additions ...string) []string { func projectFocusAction(result FocusResult, report *verification.Report) focusAction { if !result.Verification.Applicable || report == nil { - return focusAction{Kind: verificationNeeded, NextAction: "request verification for the current snapshot and policy", Reasons: []string{"applicable verification evidence is unavailable"}} + return verificationNeededAction(result) } - if incompleteFocusEvidence(result, report) { - return focusAction{Kind: evidenceUnavailable, NextAction: "complete or refresh the requested evidence", Reasons: []string{"verification evidence is incomplete or unavailable"}} + if focusSelectionRequired(result) || incompleteFocusEvidence(result, report) { + return unavailableFocusAction(result, report) } - if report.Result.Status == verification.ResultFindings { - if len(report.Findings) == 0 { - return focusAction{Kind: evidenceUnavailable, NextAction: "complete or refresh the requested evidence", Reasons: []string{"findings outcome has no retained findings"}} - } + if report.Result.Status == verification.ResultPass && report.Result.BlockingFindings > 0 { + return unavailableFocusAction(result, report, "pass outcome contradicts blocking findings") + } + if len(report.Findings) > 0 { for _, finding := range report.Findings { if finding.Location == nil || !validFocusLocation(*finding.Location) { - return focusAction{Kind: evidenceUnavailable, NextAction: "refresh finding evidence with usable locations", Reasons: []string{"findings lack valid workspace-relative locations"}} + return unavailableFocusAction(result, report, "findings lack valid workspace-relative locations") } } return focusAction{Kind: findingInspectionNeeded, NextAction: "inspect the reported finding locations", Reasons: []string{"verification reported findings with usable locations"}} } - if len(report.Findings) > 0 || report.Result.BlockingFindings > 0 { - return focusAction{Kind: evidenceUnavailable, NextAction: "complete or refresh the requested evidence", Reasons: []string{"report findings do not match its finalized outcome"}} + if report.Result.Status == verification.ResultFindings { + return unavailableFocusAction(result, report, "findings outcome has no retained findings") } if report.Result.Status == verification.ResultPass { return focusAction{Kind: requestedChecksPassedWithLimits, NextAction: "review the requested checks and their stated limits", Reasons: []string{"requested checks passed; this does not establish task completion"}} } - return focusAction{Kind: evidenceUnavailable, NextAction: "complete or refresh the requested evidence", Reasons: []string{"report does not establish a complete finding or pass outcome"}} + return unavailableFocusAction(result, report, "report does not establish a complete finding or pass outcome") +} + +func verificationNeededAction(result FocusResult) focusAction { + parts := []string{verifyCurrentSnapshotAction} + reasons := []string{"applicable verification evidence for the current snapshot and matching policy is unavailable"} + if focusSelectionRequired(result) { + parts[0] = freshSelectionAndVerifyAction + reasons = append(reasons, "the previous focus selection could not be resolved and requires a fresh selection") + } + if focusScopeMismatch(result) { + parts[0] += "; a different package scope answers a different question and does not support the requested scope" + reasons = append(reasons, "a different package scope answers a different question and does not support the requested scope") + } + if focusContextBudgetOmitted(result.Context) { + parts = append(parts, focusBudgetAction) + reasons = append(reasons, "focused context detail was omitted by the requested byte budget") + } + if unavailableFocusContext(result) { + parts = append(parts, unavailableEvidenceAction) + reasons = append(reasons, "affected focus evidence is unavailable, unknown, or internally inconsistent") + } + return focusAction{Kind: verificationNeeded, NextAction: strings.Join(parts, "; "), Reasons: reasons} +} + +func unavailableFocusAction(result FocusResult, report *verification.Report, cause ...string) focusAction { + parts := []string{} + reasons := []string{} + if focusSelectionRequired(result) { + parts = append(parts, freshSelectionAndVerifyAction) + reasons = append(reasons, "the previous focus selection could not be resolved and requires a fresh selection") + } + if focusContextBudgetOmitted(result.Context) { + parts = append(parts, focusBudgetAction) + reasons = append(reasons, "focused context detail was omitted by the requested byte budget") + } + if reportDetailsTruncated(report) { + parts = append(parts, reportTruncationAction) + reasons = append(reasons, "verification report detail was truncated; refreshing cannot recover the omitted detail") + } + if unavailableFocusEvidence(result, report) { + parts = append(parts, unavailableEvidenceAction) + reasons = append(reasons, "affected verification or focus evidence is unavailable, unknown, or internally inconsistent") + } + reasons = append(reasons, cause...) + if len(parts) == 0 { + parts = append(parts, unavailableEvidenceAction) + reasons = append(reasons, "verification evidence is incomplete or unavailable") + } + return focusAction{Kind: evidenceUnavailable, NextAction: strings.Join(parts, "; "), Reasons: reasons} +} + +func focusSelectionRequired(result FocusResult) bool { + return result.Refresh != nil && result.Refresh.Status == "selection_required" +} + +func focusScopeMismatch(result FocusResult) bool { + for _, reason := range result.Verification.Reasons { + if reason == "package scope differs" { + return true + } + } + return false +} + +func focusContextBudgetOmitted(context *FocusContext) bool { + return context != nil && (context.Truncated || context.Symbol != nil && context.Symbol.Truncated) +} + +func reportDetailsTruncated(report *verification.Report) bool { + if report == nil { + return false + } + if report.FindingsTruncated || report.Change.FilesTruncated || report.Change.DeclarationsTruncated || report.Impact.PackagesTruncated { + return true + } + for _, file := range report.Change.Files { + if file.BaseRangesTruncated || file.CurrentRangesTruncated { + return true + } + } + for _, check := range report.Plan { + if check.TargetsTruncated { + return true + } + } + for _, evidence := range report.Evidence { + if evidenceDetailsTruncated(evidence) { + return true + } + } + for _, risk := range report.Risks { + if risk.LocationsTruncated { + return true + } + } + for _, uncertainty := range report.Uncertainties { + if uncertainty.LocationsTruncated { + return true + } + } + return false +} + +func evidenceDetailsTruncated(evidence verification.Evidence) bool { + if evidence.Tests != nil && (evidence.Tests.PackagesTruncated || evidence.Tests.NonpassingTruncated) || + evidence.Coverage != nil && evidence.Coverage.UncoveredTruncated || + evidence.Diagnostics != nil && evidence.Diagnostics.Truncated || + evidence.Contract != nil && evidence.Contract.ViolationsTruncated { + return true + } + if evidence.Contract != nil { + for _, violation := range evidence.Contract.Violations { + if violation.LocationsTruncated { + return true + } + } + } + return false +} + +func unavailableFocusEvidence(result FocusResult, report *verification.Report) bool { + if unavailableFocusContext(result) { + return true + } + if report == nil { + return false + } + if report.Result.Status == verification.ResultIncomplete { + return true + } + if report.Result.Status != verification.ResultPass && report.Result.Status != verification.ResultFindings { + return true + } + planned := map[string]verification.Check{} + for _, check := range report.Plan { + if check.ID == "" { + return true + } + if _, exists := planned[check.ID]; exists { + return true + } + planned[check.ID] = check + } + if len(planned) == 0 { + return true + } + seen := map[string]int{} + for _, evidence := range report.Evidence { + check, exists := planned[evidence.CheckID] + if evidence.CheckID == "" || !exists || evidence.Kind != check.Kind { + return true + } + seen[evidence.CheckID]++ + if evidence.Status == verification.EvidenceSkipped || evidence.Status == verification.EvidenceError || + evidence.Status != verification.EvidencePassed && evidence.Status != verification.EvidenceFailed || + evidence.Analysis != nil && evidence.Analysis.Unknown > 0 || evidencePayloadMissing(evidence) { + return true + } + if report.Result.Status == verification.ResultPass && evidence.Status == verification.EvidenceFailed { + return true + } + } + for id := range planned { + if seen[id] != 1 { + return true + } + } + for _, count := range seen { + if count != 1 { + return true + } + } + for _, finding := range report.Findings { + if finding.Location == nil || !validFocusLocation(*finding.Location) { + return true + } + } + if report.Result.Status == verification.ResultFindings { + if len(report.Findings) == 0 { + return true + } + } else if report.Result.Status == verification.ResultPass && report.Result.BlockingFindings > 0 { + return true + } + for _, uncertainty := range report.Uncertainties { + if strings.Contains(uncertainty.Code, "unknown") || strings.Contains(uncertainty.Code, "unavailable") { + return true + } + } + return false +} + +func unavailableFocusContext(result FocusResult) bool { + if !result.Complete || len(result.Uncertainties) > 0 || result.Change.FilesTruncated || result.Change.DeclarationsTruncated || result.Impact.PackagesTruncated { + return true + } + if result.Context == nil { + return false + } + context := result.Context + if len(context.EvidenceStates) == 0 || context.TypedEvidence != nil && (!context.TypedEvidence.Complete || context.TypedEvidence.Truncated) { + return true + } + for _, state := range context.EvidenceStates { + switch state.State { + case "examined_and_absent", "unexamined": + case "gathered_but_omitted": + if state.Facet != "budgeted_evidence" { + return true + } + case "unavailable", "": + return true + default: + return true + } + } + for _, uncertainty := range context.Uncertainties { + if strings.Contains(uncertainty.Code, "unavailable") || strings.Contains(uncertainty.Code, "omitted") || strings.Contains(uncertainty.Code, "partial") || strings.Contains(uncertainty.Code, "truncated") { + return true + } + } + return false +} + +func evidencePayloadMissing(evidence verification.Evidence) bool { + switch evidence.Kind { + case verification.CheckTests: + return evidence.Tests == nil + case verification.CheckCoverage: + return evidence.Coverage == nil + case verification.CheckRace: + return evidence.Race == nil + case verification.CheckConcurrency, verification.CheckErrors: + return evidence.Analysis == nil + case verification.CheckDiagnostics: + return evidence.Diagnostics == nil + case verification.CheckContract: + return evidence.Contract == nil + default: + return true + } } func incompleteFocusEvidence(result FocusResult, report *verification.Report) bool { if !result.Complete || len(result.Uncertainties) > 0 || result.Change.FilesTruncated || result.Change.DeclarationsTruncated || result.Impact.PackagesTruncated || - report.Result.Status == verification.ResultIncomplete || report.FindingsTruncated || - report.Change.FilesTruncated || report.Change.DeclarationsTruncated || report.Impact.PackagesTruncated { + report.Result.Status == verification.ResultIncomplete || reportDetailsTruncated(report) { return true } planned := map[string]verification.Check{} @@ -134,7 +382,7 @@ func incompleteFocusEvidence(result FocusResult, report *verification.Report) bo } if result.Context != nil { context := result.Context - if context.Truncated || len(context.EvidenceStates) == 0 || (context.TypedEvidence != nil && (!context.TypedEvidence.Complete || context.TypedEvidence.Truncated)) { + if context.Truncated || context.Symbol != nil && context.Symbol.Truncated || len(context.EvidenceStates) == 0 || (context.TypedEvidence != nil && (!context.TypedEvidence.Complete || context.TypedEvidence.Truncated)) { return true } for _, state := range context.EvidenceStates { diff --git a/internal/intelligence/focus_action_test.go b/internal/intelligence/focus_action_test.go index 58f87dc..7f21689 100644 --- a/internal/intelligence/focus_action_test.go +++ b/internal/intelligence/focus_action_test.go @@ -1,21 +1,14 @@ package intelligence import ( + "strings" "testing" "github.com/agentic-mcps/go/internal/verification" ) func TestProjectFocusAction(t *testing.T) { - base := func() (*FocusResult, *verification.Report) { - result := &FocusResult{Verification: VerificationApplicability{Applicable: true}, Complete: true} - report := &verification.Report{ - Plan: []verification.Check{{ID: "tests", Kind: verification.CheckTests, Required: true}}, - Evidence: []verification.Evidence{{CheckID: "tests", Kind: verification.CheckTests, Status: verification.EvidencePassed, Tests: &verification.TestSummary{Packages: []verification.TestPackageSummary{}, Nonpassing: []verification.TestCaseSummary{}}}}, - Findings: []verification.Finding{}, Result: verification.PolicyResult{Status: verification.ResultPass}, - } - return result, report - } + base := focusActionFixture validContext := func(result *FocusResult) { result.Context = &FocusContext{EvidenceStates: []EvidenceState{{Facet: "related_tests", State: "examined_and_absent"}}, TypedEvidence: &TypedEvidence{Complete: true}} } @@ -27,9 +20,10 @@ func TestProjectFocusAction(t *testing.T) { }{ {name: "no report", setup: func(_ *FocusResult, report **verification.Report) { *report = nil }, want: verificationNeeded}, {name: "non-applicable report", setup: func(result *FocusResult, _ **verification.Report) { result.Verification.Applicable = false }, want: verificationNeeded}, - {name: "legacy-like result", setup: func(_ *FocusResult, report **verification.Report) { + {name: "legacy-like result", setup: func(result *FocusResult, report **verification.Report) { + result.Verification.Applicable = false *report = &verification.Report{Result: verification.PolicyResult{Status: verification.ResultPass}} - }, want: evidenceUnavailable}, + }, want: verificationNeeded}, {name: "incomplete report", setup: func(_ *FocusResult, report **verification.Report) { (*report).Result.Status = verification.ResultIncomplete }, want: evidenceUnavailable}, @@ -131,3 +125,220 @@ func TestProjectFocusAction(t *testing.T) { }) } } + +func TestProjectFocusActionCauseSpecificGuidance(t *testing.T) { + tests := []struct { + name string + setup func(*FocusResult, **verification.Report) + wantKind focusActionKind + wantNext string + wantReason []string + }{ + { + name: "missing verification", + setup: func(result *FocusResult, report **verification.Report) { + result.Verification.Applicable = false + *report = nil + }, + wantKind: verificationNeeded, + wantNext: "request verification for the current snapshot and matching policy", + wantReason: []string{ + "applicable verification evidence for the current snapshot and matching policy is unavailable", + }, + }, + { + name: "missing verification with focus byte budget omission", + setup: func(result *FocusResult, report **verification.Report) { + result.Verification.Applicable = false + result.Context = &FocusContext{Truncated: true, EvidenceStates: []EvidenceState{{Facet: "related_tests", State: "examined_and_absent"}}} + *report = nil + }, + wantKind: verificationNeeded, + wantNext: "request verification for the current snapshot and matching policy; make a fresh narrower selection or use a larger max_bytes on a new selection; replacement refresh inherits the prior budget and cannot fix the omission", + wantReason: []string{ + "applicable verification evidence for the current snapshot and matching policy is unavailable", + "focused context detail was omitted by the requested byte budget", + }, + }, + { + name: "missing verification with unavailable focus evidence", + setup: func(result *FocusResult, report **verification.Report) { + result.Verification.Applicable = false + result.Context = &FocusContext{EvidenceStates: []EvidenceState{{Facet: "declaration_relationships", State: "unavailable"}}} + *report = nil + }, + wantKind: verificationNeeded, + wantNext: "request verification for the current snapshot and matching policy; treat affected evidence as unavailable; retry recovery is not established", + wantReason: []string{ + "applicable verification evidence for the current snapshot and matching policy is unavailable", + "affected focus evidence is unavailable, unknown, or internally inconsistent", + }, + }, + { + name: "snapshot mismatch requests verification without reselection", + setup: func(result *FocusResult, _ **verification.Report) { + result.Verification.Applicable = false + result.Verification.Reasons = []string{"workspace snapshot differs"} + }, + wantKind: verificationNeeded, + wantNext: "request verification for the current snapshot and matching policy", + wantReason: []string{ + "applicable verification evidence for the current snapshot and matching policy is unavailable", + }, + }, + { + name: "legacy report without applicability metadata", + setup: func(result *FocusResult, report **verification.Report) { + result.Verification.Applicable = false + result.Verification.Reasons = []string{"stored report has no applicability metadata"} + (*report).Result.Status = verification.ResultPass + }, + wantKind: verificationNeeded, + wantNext: "request verification for the current snapshot and matching policy", + wantReason: []string{ + "applicable verification evidence for the current snapshot and matching policy is unavailable", + }, + }, + { + name: "policy mismatch", + setup: func(result *FocusResult, _ **verification.Report) { + result.Verification.Applicable = false + result.Verification.Reasons = []string{"requested verification check policy differs"} + }, + wantKind: verificationNeeded, + wantNext: "request verification for the current snapshot and matching policy", + wantReason: []string{ + "applicable verification evidence for the current snapshot and matching policy is unavailable", + }, + }, + { + name: "narrower verification scope", + setup: func(result *FocusResult, _ **verification.Report) { + result.Verification.Applicable = false + result.Verification.Reasons = []string{"package scope differs"} + }, + wantKind: verificationNeeded, + wantNext: "request verification for the current snapshot and matching policy; a different package scope answers a different question and does not support the requested scope", + wantReason: []string{ + "applicable verification evidence for the current snapshot and matching policy is unavailable", + "a different package scope answers a different question and does not support the requested scope", + }, + }, + { + name: "focus byte budget omission", + setup: func(result *FocusResult, _ **verification.Report) { + result.Context = &FocusContext{Truncated: true, EvidenceStates: []EvidenceState{{Facet: "related_tests", State: "examined_and_absent"}}} + }, + wantKind: evidenceUnavailable, + wantNext: "make a fresh narrower selection or use a larger max_bytes on a new selection; replacement refresh inherits the prior budget and cannot fix the omission", + wantReason: []string{ + "focused context detail was omitted by the requested byte budget", + }, + }, + { + name: "verification report detail truncation", + setup: func(_ *FocusResult, report **verification.Report) { + (*report).Change.FilesTruncated = true + }, + wantKind: evidenceUnavailable, + wantNext: "treat omitted verification report detail as unavailable; refreshing cannot recover the omitted detail", + wantReason: []string{ + "verification report detail was truncated; refreshing cannot recover the omitted detail", + }, + }, + { + name: "pass with advisory finding", + setup: func(_ *FocusResult, report **verification.Report) { + (*report).Findings = []verification.Finding{{Location: &verification.Location{File: "pkg/file.go", Line: 3}}} + }, + wantKind: findingInspectionNeeded, + wantNext: "inspect the reported finding locations", + wantReason: []string{ + "verification reported findings with usable locations", + }, + }, + { + name: "pass outcome contradicts blocking findings", + setup: func(_ *FocusResult, report **verification.Report) { + (*report).Result.BlockingFindings = 1 + }, + wantKind: evidenceUnavailable, + wantNext: "treat affected evidence as unavailable; retry recovery is not established", + wantReason: []string{ + "affected verification or focus evidence is unavailable, unknown, or internally inconsistent", + "pass outcome contradicts blocking findings", + }, + }, + { + name: "findings outcome without retained findings", + setup: func(_ *FocusResult, report **verification.Report) { + (*report).Result.Status = verification.ResultFindings + }, + wantKind: evidenceUnavailable, + wantNext: "treat affected evidence as unavailable; retry recovery is not established", + wantReason: []string{ + "affected verification or focus evidence is unavailable, unknown, or internally inconsistent", + "findings outcome has no retained findings", + }, + }, + { + name: "mixed truncation and unavailable evidence", + setup: func(result *FocusResult, report **verification.Report) { + result.Context = &FocusContext{Truncated: true, EvidenceStates: []EvidenceState{{Facet: "budgeted_evidence", State: "gathered_but_omitted"}, {Facet: "declaration_relationships", State: "unavailable"}}} + (*report).FindingsTruncated = true + }, + wantKind: evidenceUnavailable, + wantNext: "make a fresh narrower selection or use a larger max_bytes on a new selection; replacement refresh inherits the prior budget and cannot fix the omission; treat omitted verification report detail as unavailable; refreshing cannot recover the omitted detail; treat affected evidence as unavailable; retry recovery is not established", + wantReason: []string{ + "focused context detail was omitted by the requested byte budget", + "verification report detail was truncated; refreshing cannot recover the omitted detail", + "affected verification or focus evidence is unavailable, unknown, or internally inconsistent", + }, + }, + { + name: "stale focus selection", + setup: func(result *FocusResult, _ **verification.Report) { + result.Refresh = &FocusRefresh{Status: "selection_required"} + result.Context = &FocusContext{EvidenceStates: []EvidenceState{{Facet: "previous_declaration", State: "unavailable"}}} + }, + wantKind: evidenceUnavailable, + wantNext: "make a fresh selection against the current snapshot, then request verification for that snapshot and matching policy; treat affected evidence as unavailable; retry recovery is not established", + wantReason: []string{ + "the previous focus selection could not be resolved and requires a fresh selection", + "affected verification or focus evidence is unavailable, unknown, or internally inconsistent", + }, + }, + { + name: "passing checks do not complete the task", + setup: func(result *FocusResult, _ **verification.Report) { + result.Context = &FocusContext{EvidenceStates: []EvidenceState{{Facet: "related_tests", State: "examined_and_absent"}}} + }, + wantKind: requestedChecksPassedWithLimits, + wantNext: "review the requested checks and their stated limits", + wantReason: []string{ + "requested checks passed; this does not establish task completion", + }, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + result, report := focusActionFixture() + tt.setup(result, &report) + got := projectFocusAction(*result, report) + if got.Kind != tt.wantKind || got.NextAction != tt.wantNext || strings.Join(got.Reasons, "\n") != strings.Join(tt.wantReason, "\n") { + t.Fatalf("action = %#v, want kind %q, next action %q, reasons %q", got, tt.wantKind, tt.wantNext, strings.Join(tt.wantReason, "\n")) + } + }) + } +} + +func focusActionFixture() (*FocusResult, *verification.Report) { + result := &FocusResult{Verification: VerificationApplicability{Applicable: true}, Complete: true} + report := &verification.Report{ + Plan: []verification.Check{{ID: "tests", Kind: verification.CheckTests, Required: true}}, + Evidence: []verification.Evidence{{CheckID: "tests", Kind: verification.CheckTests, Status: verification.EvidencePassed, Tests: &verification.TestSummary{Packages: []verification.TestPackageSummary{}, Nonpassing: []verification.TestCaseSummary{}}}}, + Findings: []verification.Finding{}, Result: verification.PolicyResult{Status: verification.ResultPass}, + } + return result, report +} diff --git a/internal/intelligence/focus_test.go b/internal/intelligence/focus_test.go index f033a16..164b1d3 100644 --- a/internal/intelligence/focus_test.go +++ b/internal/intelligence/focus_test.go @@ -149,6 +149,7 @@ func TestFocusWorkflowBeforeEditRefreshAndVerificationStaleness(t *testing.T) { symbol: SymbolMatch{Name: "Worker", Qualified: "fixture.Worker", Kind: "go.type", Package: "fixture", Location: Location{File: "worker.go", Line: 3, Column: 6}}, } core := newTestCore(t, snapshotter, reader) + core.semantic.(*fakeSemanticProvider).identity.Capabilities.CallHierarchy = true seed, err := core.capture(context.Background(), "HEAD", "./...", "") if err != nil { t.Fatal(err) @@ -192,6 +193,9 @@ func TestFocusWorkflowBeforeEditRefreshAndVerificationStaleness(t *testing.T) { if after.Verification.Applicable || !containsReason(after.Verification.Reasons, "workspace snapshot") { t.Fatalf("verification applicability = %#v", after.Verification) } + if after.Verification.NextAction != "request verification for the current snapshot and matching policy" { + t.Fatalf("stale verification next action = %q", after.Verification.NextAction) + } if _, err := core.Symbol(context.Background(), SymbolRequest{Ref: oldRef, MaxBytes: DefaultSymbolBytes}); !errors.Is(err, ErrSnapshotChanged) { t.Fatalf("old Symbol Ref error = %v", err) } From 56c2199418454c6e0371023e042c61d2062cdf88 Mon Sep 17 00:00:00 2001 From: Ashwin Gopalsamy <47941624+ashwingopalsamy@users.noreply.github.com> Date: Wed, 23 Sep 2026 22:38:35 +0530 Subject: [PATCH 10/20] docs(intelligence): reconcile reliability guidance --- docs/README.md | 23 +++-- docs/continuation/go-intelligence.md | 144 ++++++++++++++------------- docs/go-intelligence-north-star.md | 75 ++++++++------ 3 files changed, 133 insertions(+), 109 deletions(-) diff --git a/docs/README.md b/docs/README.md index bfa7c2e..c26645b 100644 --- a/docs/README.md +++ b/docs/README.md @@ -21,14 +21,21 @@ investigation. The [continuation handoff](continuation/go-intelligence.md) owns implementation status and the next action; the [Go intelligence north star](go-intelligence-north-star.md) owns the approved architectural direction and acceptance criteria. -The additive post-v1 `agentic.focus/v1` evidence layer is implemented and -locally qualified. The initial private 20-run Luna feasibility pilot found no -treatment use because `go_context` was unused in all 10 focus runs. The later -27-run adoption follow-up found 0/6 use with description-only discoverability, -6/6 with generic prompt guidance, and 6/6 with the shipped skill; it still -establishes no causal engineering benefit. See the -[adoption results](../validation/v1.0.0/adoption-results.md). Delta refresh is -deferred. These documents do not override frozen v1 contracts. +`v1.2.1` (tag `67f54b7`) is the latest released baseline. The current +`codex/v1.2-reliability` branch contains unreleased post-v1.2.1 work. Its +product direction is a deterministic, snapshot-bound evidence compiler that +supports the edit, refresh, verify, inspect loop. Slice 1A adds next-action +guidance to existing MCP text, and Slice 1B refines private evidence projection +guidance. Both preserve the frozen public MCP inventory and +`agentic.focus/v1` schema. + +The v0.8/v1.0 pilot and adoption reports are historical instruction-use and +workflow evidence, not proof of causal engineering benefit. The recent +deterministic evaluation dry run validated harness inputs and replay records; +it did not compare live model runs with and without agentic-go. See the +[historical adoption results](../validation/v1.0.0/adoption-results.md). +Delta refresh remains deferred. These documents do not override frozen v1 +contracts. ## Verification report contracts diff --git a/docs/continuation/go-intelligence.md b/docs/continuation/go-intelligence.md index cd73911..1f7b726 100644 --- a/docs/continuation/go-intelligence.md +++ b/docs/continuation/go-intelligence.md @@ -6,13 +6,12 @@ This document preserves the current product understanding for a future agent working in another account or session. It is independent of conversation history, account identity, private memory, and previous tool output. -The user’s end goal is a flagship Go MCP and language-server/code-navigator -experience usable through any coding agent. It should materially remove the -repeated investigation normally required to understand and change unfamiliar -Go code: the agent should receive compact, source-grounded context for the -next change, understand implementation obligations and existing examples, -retain orientation after edits, and see what changed. “1000x” and quality -rankings express ambition, not measured or promised results. +The product direction is a deterministic, snapshot-bound evidence compiler for +Go coding agents. Its central workflow is edit, refresh, verify, inspect +evidence, then reconsider what the evidence requires. The goal is to help an +agent understand what to revisit before treating a Go change as complete. +Effectiveness claims require comparative evidence and are not established by +the historical pilots or the recent deterministic dry run. The full approved architectural direction is in the canonical [Go intelligence north-star plan](../go-intelligence-north-star.md). Approval @@ -35,19 +34,24 @@ become stale. ## Current status -As of 2026-09-22, this branch prepares the signed `v1.2.0` release from the -`v1.1.0` baseline. **Stage -2, coherent observation**, and the additive focus slice are shipped in that -release. -The observation-correctness follow-up closed the source-confirmed gaps recorded -by the Astra review. -The change preserves the existing exact content-based Snapshot Ref and v1 -interfaces while threading a private request-scoped observation through the -intelligence paths. The implemented additive slice includes `go_context`, -`agentic-go context --format text|json`, and `agentic.focus/v1`; these remain -post-v1 capabilities and are not frozen v1 interfaces. The broader roadmap -stages remain partially implemented or pending and must not be inferred as -complete from this slice. +`v1.2.1` (tag `67f54b7`) is the latest released baseline. This branch, +`codex/v1.2-reliability`, contains unreleased work after that release, including +commits `8e65ce1` and `71099dc`. Do not describe this branch as a release +candidate or assign a new release label before a reliability milestone passes. + +The current product wedge is to make the edit, refresh, verify, and inspect +loop dependable enough that a coding agent knows what to reconsider before +declaring a Go change complete. The system provides deterministic, +snapshot-bound context and verification evidence; it does not decide that the +engineering task itself is complete. + +The public MCP inventory and `agentic.focus/v1` schema remain frozen. Slice 1A +adds `next_action` to existing MCP text using the existing structured +`Verification.NextAction`. Slice 1B refines the private projection with +cause-specific guidance for stale or unavailable evidence, truncation, budget +omission, advisory findings, and passing checks with limits. Slice 1A changes +existing MCP text rendering; Slice 1B changes private projection guidance. +Neither changes inventory, structured fields, or schemas. Implemented in v1.1.0: @@ -83,30 +87,29 @@ Focused correctness tests cover captured-source precedence, same-size and A→B→A rewrites on a source-cap miss, guidance identity and mismatch rejection, Brief observation forwarding, Symbol position-error lease release, active manifest protection, fail-closed admission, and replacement byte accounting. -Those historical checks qualify the v1.1.0 release work; they do not qualify -later changes. The v1.2.0 candidate separately passed the full -repository gates (`go test ./...`, `go test -race ./...`, `go vet ./...`, -`go build ./...`, and `git diff --check`). The v0.8 task, adoption, and pilot -definitions validate, and two available private server replays pass. No -three-tier model evaluation or productivity claim is included. - -Stage 2 does not introduce a public interface or a general derived cache. -Derived parsing/semantic caching, useful-context selection, richer -relationships, refresh, and verification lineage remain later-stage work. - -## v1.2.0 reliability release - -The `codex/v1.2-reliability` release narrows focus semantic expansion -by the selected declaration kind. Function and method selections do not request -type definitions, and non-callable declarations do not request call hierarchy. -Skipped facets are reported as unexamined rather than as examined-and-absent -evidence; incomplete provider evidence is not reported as absent. This is a -bounded reliability fix for observed provider failures; the public MCP -inventory and `agentic.focus/v1` schema remain unchanged. Focused package -validation and full repository gates have passed. The prerequisite snapshot -input fix is committed as `5ff6902`, the focus follow-up as `4381456`, and -bounded outcome tracing as `7cb2117`. The release metadata preserves the -existing 15-tool current surface and frozen v1 contracts. +Those checks qualify the historical v1.1.0 work; they do not qualify later +changes. The v0.8 task, adoption, and pilot records below are historical. A +recent deterministic evaluation dry run validated local harness inputs and +replay records; it was not a live baseline-versus-agentic-go comparison and +supports no model, speed, token, quality, adoption, or causal improvement +claim. + +In the v1.1.0 implementation, Stage 2 added no public interface or general +derived cache. The post-v1 focus and refresh capabilities described below were +added later; the frozen v1 contracts remain unchanged. + +## Current reliability work + +The post-v1.2.1 branch work improves how agents interpret existing evidence. +The four private projection outcomes are `verification_needed`, +`finding_inspection_needed`, `evidence_unavailable`, and +`requested_checks_passed_with_limits`. Stale, mismatched, legacy, missing, or +policy-incompatible reports do not yield current repair targets. Findings are +actionable for inspection only when their source locations remain valid. +Incomplete, cancelled, truncated, provider-failed, and unknown evidence stays +unavailable. A passing requested check is not a declaration that the task is +complete. The follow-up guidance is bounded, provenance-linked, and +non-mutating. ## Findings recorded from the documentation and targeted source inspection @@ -150,17 +153,17 @@ replay from model outcomes. The reviewed [v1 release evidence](../../validation/ does not establish a paid model pilot or comparative agent advantage. These are historical document statements, not checks rerun in this session. -The initial review recommended a comparative pilot. The user then explicitly -redirected the strategy toward material product behavior, not eval/benchmark -milestones. The resulting plan follows that correction: context selection, -coherent reads, Go-specific relationships, and explicit refresh. The user chose -additive interface evolution and focused correctness checks when asked. +Comparative evaluation is not a product milestone for this reliability work. +Prioritize material product behavior: context selection, coherent reads, +Go-specific relationships, explicit refresh, and trustworthy verification +applicability. Additive post-v1 interfaces are documented separately from the +frozen v1 contracts. These are repository observations and design opportunities, not claims that the current runtime is fast, relevant, or superior to other tools. Those properties have not been verified in this documentation task. -## Important limits and exact next action +## Important limits The approved plan is architectural direction, not an already-frozen wire specification. Implementation must resolve and record these local facts in the @@ -225,7 +228,7 @@ replacement with current locations and Symbol Refs. Expiry requires a fresh request; changed selectors are rejected; ambiguous moves or renames require a current candidate selection. Failed resolution and budget omission are marked unavailable and never reported as confirmed deletion. Delta refresh remains -deferred. The next separately authorized extension is delta refresh. +deferred. The four planned Go relationship families are now implemented sequentially in the focused evidence layer. Declaration, file, and package selection can return @@ -262,19 +265,23 @@ implementation request is sufficient to begin the next pending stage without asking again for the same authorization. The private adoption harness links the existing client-go and grpc-go tasks, validates sanitized records, hashes transcripts, and produces deterministic condition summaries without changing -the frozen v0.8 corpus. The 27-run adoption follow-up is complete: description -alone produced 0/6 focus use, generic prompt guidance produced 6/6, the shipped -skill produced 6/6, initial integrated safety was 5/6, and the scope-wording -rerun was 3/3 acceptance-pass, qualifying, and scope-safe. Provider failures -remain the next reliability issue. See the [tracked adoption -results](../../validation/v1.0.0/adoption-results.md) for exact gates, metrics, -identities, and limitations. Retain focus and full replacement; defer delta -refresh. The next slice is release hardening, instruction-surface -discoverability, and provider-failure investigation. Raw artifacts remain -private and ignored. -The v1.2.0 publication workflow is separately authorized. Public publication -preserves existing tags and history and does not claim that the paid model -comparison ran. +the frozen v0.8 corpus. The historical 27-run adoption follow-up is complete: +description alone produced 0/6 focus use, generic prompt guidance produced +6/6, the shipped skill produced 6/6, initial integrated safety was 5/6, and the +scope-wording rerun was 3/3 acceptance-pass, qualifying, and scope-safe. See +the [tracked adoption results](../../validation/v1.0.0/adoption-results.md) +for exact gates, metrics, +identities, and limitations. These historical results do not establish +comparative engineering benefit. Retain focus and full replacement; defer delta +refresh. Raw artifacts remain private and ignored. + +## Current next action + +Finish this documentation reconciliation, then run the agreed repository +gates and scope checks. After that, decide whether to run a separate Luna-only +live pilot using the validated source checkouts. No live with/without +comparison has yet been completed. Defer any new release label until a +reliability milestone passes. ## Standing continuation instruction @@ -293,9 +300,8 @@ Use this prompt only when the user separately authorizes further work: ```text Read docs/continuation/astra-understanding.md and this handoff first. Observation, verification applicability, declaration selection, and full-replacement refresh -and focus-v1 stabilization are complete. Inspect the current diff and choose a -new explicitly authorized objective. Delta refresh, general derived caches, -expanded refactoring, and speculative test selection remain outside the -completed scope. Preserve existing tags and public history when continuing the -reliability work. +are implemented. Continue the current reliability milestone from the current +next action in this handoff. Delta refresh, general derived caches, expanded +refactoring, and speculative test selection remain deferred. Preserve existing +tags and public history. ``` diff --git a/docs/go-intelligence-north-star.md b/docs/go-intelligence-north-star.md index 7ffb0e6..a667fbb 100644 --- a/docs/go-intelligence-north-star.md +++ b/docs/go-intelligence-north-star.md @@ -1,11 +1,19 @@ # Go intelligence for navigating, generating, and changing code -Status: approved architectural direction, revised 2026-09-06. Stage 2 focus is -implemented as an additive post-v1 capability; the frozen v1 inventory and -contracts are unchanged. The [Astra analysis](continuation/astra-understanding.md) -narrows the next work around release hardening and tool discoverability. -`go_context`, `agentic-go context`, and `agentic.focus/v1` are implemented -capabilities, but are not part of the frozen v1 contract. +Status: approved product direction, revised 2026-09-23. `v1.2.1` (tag +`67f54b7`) is the latest released baseline. The current +`codex/v1.2-reliability` branch contains unreleased post-v1.2.1 work. The +product is a deterministic, snapshot-bound evidence compiler for Go coding +agents. Its wedge is a dependable edit, refresh, verify, inspect loop that +helps an agent know what to reconsider before treating a change as complete. + +The `go_context`, `agentic-go context`, and `agentic.focus/v1` capabilities are +post-v1 additions. Slice 1A adds next-action guidance to existing MCP text +using existing structured verification data. Slice 1B refines the private +evidence projection with cause-specific guidance. Both preserve the frozen +public MCP inventory and `agentic.focus/v1` schema. See the +[continuation handoff](continuation/go-intelligence.md) for current milestone +status and next action. Start with the [continuation handoff](continuation/go-intelligence.md) for implementation status, source pointers, and the exact next step. Existing @@ -15,11 +23,10 @@ behavior. ## End goal and observable workflows -Develop agentic-go around one complete workflow: +The core workflow is: ```text -Locate relevant code -> understand obligations and existing patterns - -> external agent edits -> refresh consequences -> request verification +context -> edit -> refresh -> verify -> inspect evidence -> reconsider or continue ``` A coding agent should be able to answer four questions from compact, @@ -35,13 +42,14 @@ what needs reconsideration after an edit. | Continue after an edit | Refresh the previous selection with current locations and evidence differences, without mandatory Change Contract setup. | | Verify a change | Request executed verification explicitly and identify the snapshot to which that evidence belongs. | -The ambition is exceptional usefulness across coding agents. "1000x" and -"top 0.0001%" describe ambition, not measured performance, quality rankings, -or acceptance criteria. No comparative evaluation or benchmark campaign is a -product milestone. The private 20-run Luna paired feasibility pilot and its -27-run adoption follow-up are complete; their results are evidence for the next -engineering decision only, not product claims. The tracked adoption report is -in [validation/v1.0.0/adoption-results.md](../validation/v1.0.0/adoption-results.md). +The historical private 20-run Luna feasibility pilot and 27-run adoption +follow-up record workflow and instruction-use observations. They do not show +causal improvement in engineering outcomes. A later deterministic dry run +validated evaluation harness inputs and replay records, but did not compare +live model runs with and without agentic-go. No model, speed, token, quality, +adoption, or causal improvement claim is supported by those results. The +historical adoption report is in +[validation/v1.0.0/adoption-results.md](../validation/v1.0.0/adoption-results.md). Keep Go-only, local, deterministic operation; pinned gopls; source provenance; explicit uncertainty; and existing containment and guarded-refactor guarantees. @@ -278,21 +286,23 @@ evidence with its observed snapshot so later edits cannot make an older passing report appear current. Preserve verification result semantics, conservative package selection, and the distinction between context and executed evidence. -## Adoption result and next decision +## Historical adoption evidence and next decision The 27-run Luna/max adoption follow-up contains an 18-run canonical three-arm matrix, six integrated-skill diagnostics, and three scope-wording reruns. Description-only discoverability produced 0/6 focus use; generic prompt guidance produced 6/6; and the shipped skill produced 6/6. Initial integrated safety was 5/6, while the scope-wording rerun was 3/3 acceptance-pass, -qualifying, and scope-safe. Provider failures remain the next reliability issue. +qualifying, and scope-safe. These are historical observations, not proof that +the workflow improves task outcomes. These records establish observed instruction-surface use and safety only. They do not support causal speed, token, reliability, adoption, performance, or -generalization claims. Retain focus and full-replacement refresh, defer delta -refresh, and prioritize release hardening, discoverability, instruction-surface -placement, and provider-failure investigation. Raw artifacts remain private and -ignored. +generalization claims. Keep focus and full-replacement refresh; defer delta +refresh. The current next step is to finish documentation reconciliation and +run repository gates. Then decide whether to conduct a separate Luna-only live +pilot. Defer any new release label until a reliability milestone passes. Raw +artifacts remain private and ignored. ## Delivery order and completion criteria @@ -328,10 +338,9 @@ decisions and the evidence actually gathered in the handoff. - Equivalent evidence across CLI text, CLI JSON, and MCP, accounting for rendered budgets and deliberate compatibility checks for existing v1 tools. -These cases specify future focused correctness work. Add or run checks only -within the implementation request's authorization and report only checks -actually performed. No test, build, benchmark, or validation command is part -of this documentation stage. +These cases specify future focused correctness work. This documentation change +introduces no new tests or benchmarks. Complete the agreed repository gates and +scope checks before committing. ## Boundaries and implementation decisions @@ -347,12 +356,14 @@ of this documentation stage. introduced by this work. - No broad comparative evaluation, paid model run, or benchmark campaign is a release gate. The private Luna focus and adoption follow-up are recorded - evidence only and do not establish causal engineering improvement. Automatic - commits, tags, and releases remain outside this implementation direction; - separately authorized publication must preserve existing public history. -- Existing uncommitted documentation work is preserved. The accepted immediate - deliverable is stage 1; product stages require a later implementation - instruction. + evidence only and do not establish causal engineering improvement. Local + Conventional Commits are authorized after the agreed gates and scope checks + pass. Pushes, tags, and releases require separate authorization and must + preserve existing public history. +- The documentation stage and the implemented observation, context, focus, and + verification slices are recorded above. Continue only from the current + handoff next action with an explicit bounded implementation slice; later + roadmap rows are not complete merely because they are listed here. The behavioral decisions above govern future implementation. Exact new wire fields, cache capacities, serialized budget accounting, and typed extraction From cbbdffc049ec5cd5d4ad4ddd86f2b241364ccc3a Mon Sep 17 00:00:00 2001 From: Ashwin Gopalsamy <47941624+ashwingopalsamy@users.noreply.github.com> Date: Thu, 24 Sep 2026 00:43:33 +0530 Subject: [PATCH 11/20] docs(eval): record Luna adoption evidence --- docs/README.md | 25 +++++++++++++++------- docs/continuation/go-intelligence.md | 31 +++++++++++++++++----------- docs/go-intelligence-north-star.md | 25 ++++++++++++++-------- 3 files changed, 53 insertions(+), 28 deletions(-) diff --git a/docs/README.md b/docs/README.md index c26645b..231fc1b 100644 --- a/docs/README.md +++ b/docs/README.md @@ -29,13 +29,24 @@ guidance to existing MCP text, and Slice 1B refines private evidence projection guidance. Both preserve the frozen public MCP inventory and `agentic.focus/v1` schema. -The v0.8/v1.0 pilot and adoption reports are historical instruction-use and -workflow evidence, not proof of causal engineering benefit. The recent -deterministic evaluation dry run validated harness inputs and replay records; -it did not compare live model runs with and without agentic-go. See the -[historical adoption results](../validation/v1.0.0/adoption-results.md). -Delta refresh remains deferred. These documents do not override frozen v1 -contracts. +Private local Luna evaluations now include a 20-run discoverability pilot and +a 12-run integrated adoption follow-up. In the focus arm, capability delivery +was healthy in 10/10 runs, but no run called `go_context`. All six integrated +runs discovered the shipped skill, called `go_context`, refreshed after +editing, and used the evidence. Integrated median duration was 431,641 ms +versus 223,758 ms for baseline, about 93% higher; median tool calls were 24 +versus 26. Astra judged workflow adoption locally demonstrated for the +combined skill and MCP surface, with comparative +product value still unproven. They do not establish MCP-alone causality, +improved correctness, productivity, token efficiency, or speed, statistical +significance, generalization, or production readiness. Both studies used two +scenarios and `gpt-5.6-luna` at max reasoning. The integrated follow-up used +three repetitions per scenario per arm. Reports remain private and are not +tracked. See the +[historical adoption results](../validation/v1.0.0/adoption-results.md) for +older v0.8/v1.0 evidence. The two current studies are regression evidence, not +fresh proof of product superiority. Delta refresh remains deferred. These +documents do not override frozen v1 contracts. ## Verification report contracts diff --git a/docs/continuation/go-intelligence.md b/docs/continuation/go-intelligence.md index 1f7b726..4c66354 100644 --- a/docs/continuation/go-intelligence.md +++ b/docs/continuation/go-intelligence.md @@ -10,8 +10,8 @@ The product direction is a deterministic, snapshot-bound evidence compiler for Go coding agents. Its central workflow is edit, refresh, verify, inspect evidence, then reconsider what the evidence requires. The goal is to help an agent understand what to revisit before treating a Go change as complete. -Effectiveness claims require comparative evidence and are not established by -the historical pilots or the recent deterministic dry run. +Effectiveness claims require comparative evidence. The private Luna studies +record workflow adoption, but do not establish comparative product value. The full approved architectural direction is in the canonical [Go intelligence north-star plan](../go-intelligence-north-star.md). Approval @@ -88,11 +88,9 @@ A→B→A rewrites on a source-cap miss, guidance identity and mismatch rejectio Brief observation forwarding, Symbol position-error lease release, active manifest protection, fail-closed admission, and replacement byte accounting. Those checks qualify the historical v1.1.0 work; they do not qualify later -changes. The v0.8 task, adoption, and pilot records below are historical. A -recent deterministic evaluation dry run validated local harness inputs and -replay records; it was not a live baseline-versus-agentic-go comparison and -supports no model, speed, token, quality, adoption, or causal improvement -claim. +changes. The v0.8 task, adoption, and pilot records below are historical. Two +private local Luna studies provide bounded instruction-use and workflow +adoption observations; comparative product value remains unproven. In the v1.1.0 implementation, Stage 2 added no public interface or general derived cache. The post-v1 focus and refresh capabilities described below were @@ -277,11 +275,20 @@ refresh. Raw artifacts remain private and ignored. ## Current next action -Finish this documentation reconciliation, then run the agreed repository -gates and scope checks. After that, decide whether to run a separate Luna-only -live pilot using the validated source checkouts. No live with/without -comparison has yet been completed. Defer any new release label until a -reliability milestone passes. +The local evaluation milestone is partial. The 20-run discoverability pilot +qualified all runs and kept all runs scope-safe, but its 10 focus runs made no +`go_context` calls. The 12-run integrated adoption follow-up also qualified +all runs and kept them scope-safe; all six integrated runs discovered the +shipped skill, used `go_context`, refreshed after editing, and used the +evidence. The integrated arm combines the skill and MCP surface, so it does +not establish MCP-alone causality or comparative product value. + +Inspect the six integrated transcripts to determine whether context or refresh +evidence changed a necessary edit or verification decision and to identify +avoidable repeated work behind the longer integrated duration. Use that +inspection to justify at most one bounded guidance improvement. Keep both +studies as regression evidence, not fresh proof of product superiority. Defer +any new release label until a reliability milestone passes. ## Standing continuation instruction diff --git a/docs/go-intelligence-north-star.md b/docs/go-intelligence-north-star.md index a667fbb..44528c6 100644 --- a/docs/go-intelligence-north-star.md +++ b/docs/go-intelligence-north-star.md @@ -43,11 +43,15 @@ what needs reconsideration after an edit. | Verify a change | Request executed verification explicitly and identify the snapshot to which that evidence belongs. | The historical private 20-run Luna feasibility pilot and 27-run adoption -follow-up record workflow and instruction-use observations. They do not show -causal improvement in engineering outcomes. A later deterministic dry run -validated evaluation harness inputs and replay records, but did not compare -live model runs with and without agentic-go. No model, speed, token, quality, -adoption, or causal improvement claim is supported by those results. The +follow-up record workflow and instruction-use observations. A new private +20-run discoverability pilot qualified all runs and kept them scope-safe, but +its 10 focus runs made no `go_context` calls. A 12-run integrated adoption +follow-up also qualified all runs and kept them scope-safe; all six integrated +runs discovered the shipped skill, used `go_context`, refreshed after editing, +and used the evidence. The integrated arm combines the skill and MCP surface. +These results do not establish MCP-alone causality, improved correctness, +productivity, token efficiency, speed, statistical significance, or broad +model generalization. External and multi-model evaluation remain pending. The historical adoption report is in [validation/v1.0.0/adoption-results.md](../validation/v1.0.0/adoption-results.md). @@ -299,10 +303,13 @@ the workflow improves task outcomes. These records establish observed instruction-surface use and safety only. They do not support causal speed, token, reliability, adoption, performance, or generalization claims. Keep focus and full-replacement refresh; defer delta -refresh. The current next step is to finish documentation reconciliation and -run repository gates. Then decide whether to conduct a separate Luna-only live -pilot. Defer any new release label until a reliability milestone passes. Raw -artifacts remain private and ignored. +refresh. The current next step is to inspect the six integrated transcripts +for whether context or refresh evidence changed a necessary edit or +verification decision, and for avoidable repeated work behind the longer +integrated duration. Use that inspection to justify at most one bounded +guidance improvement. Keep the studies as regression evidence, not fresh proof +of product superiority. Defer any new release label until a reliability +milestone passes. Raw artifacts remain private and ignored. ## Delivery order and completion criteria From 6d7b1e928801695918fa23d0ea95e6c3b0ec9ee8 Mon Sep 17 00:00:00 2001 From: Ashwin Gopalsamy <47941624+ashwingopalsamy@users.noreply.github.com> Date: Thu, 24 Sep 2026 00:56:30 +0530 Subject: [PATCH 12/20] docs(guidance): refine context refresh workflow --- .agents/skills/agentic-go-context/SKILL.md | 15 ++++++--- docs/README.md | 14 ++++++-- docs/continuation/go-intelligence.md | 38 +++++++++++++--------- docs/go-intelligence-north-star.md | 27 ++++++++++----- 4 files changed, 63 insertions(+), 31 deletions(-) diff --git a/.agents/skills/agentic-go-context/SKILL.md b/.agents/skills/agentic-go-context/SKILL.md index 481d2f5..2e1cbb2 100644 --- a/.agents/skills/agentic-go-context/SKILL.md +++ b/.agents/skills/agentic-go-context/SKILL.md @@ -11,12 +11,17 @@ refresh after edits. 1. Start with `base` and one selector: `query`, `symbol_ref`, source `file`+`line`+`column`, or `focus_file`/`focus_package`. Selectors are mutually exclusive. -2. After editing, refresh with `base` and `previous_pack_id` only. Omit old - selectors and Symbol Refs. -3. Stale rejection is expected: stop and select current evidence; never bypass +2. If a result is ambiguous, narrow it using the returned current candidates + instead of repeating the same broad query. +3. Finish each batch of edits and formatting before refreshing. Refresh with + `base` and `previous_pack_id` only; omit old selectors and Symbol Refs. + Refresh again after further edits or stale-snapshot rejection; avoid + redundant refreshes while the snapshot is unchanged. +4. Stale rejection is expected: stop and select current evidence; never bypass it. -4. Use the result for impact and verification applicability. Treat impact and +5. Use the result for impact and verification applicability. Treat impact and reverse-dependency evidence as planning guidance, not authorization to edit affected packages; edit only the smallest task owner within supplied path/package scope. Skip trivial edits and preserve uncertainty when - evidence is unavailable. + evidence is unavailable. Context does not execute checks or prove + correctness. diff --git a/docs/README.md b/docs/README.md index 231fc1b..54e702e 100644 --- a/docs/README.md +++ b/docs/README.md @@ -45,8 +45,18 @@ three repetitions per scenario per arm. Reports remain private and are not tracked. See the [historical adoption results](../validation/v1.0.0/adoption-results.md) for older v0.8/v1.0 evidence. The two current studies are regression evidence, not -fresh proof of product superiority. Delta refresh remains deferred. These -documents do not override frozen v1 contracts. +fresh proof of product superiority. Review of the six integrated traces is +complete: no transcript establishes that context improved the necessary code +edit. In gRPC run 3, refreshed context prompted broader `./...` verification, +which hit the output cap; focused verification later passed after a stale +snapshot rejection. Client-go runs made 3-4 context calls each, and gRPC runs +made 4-5, including ambiguous or unhelpful selections. The bounded guidance +improvement is to narrow ambiguity using returned candidates, finish each batch +of edits and formatting before refreshing, and refresh again after further +edits or stale-snapshot rejection while avoiding redundant refreshes when the +snapshot is unchanged. This observation does not establish causal edit-quality or product +value. External and multi-model evaluation remain pending. Delta refresh +remains deferred. These documents do not override frozen v1 contracts. ## Verification report contracts diff --git a/docs/continuation/go-intelligence.md b/docs/continuation/go-intelligence.md index 4c66354..41b8701 100644 --- a/docs/continuation/go-intelligence.md +++ b/docs/continuation/go-intelligence.md @@ -275,20 +275,27 @@ refresh. Raw artifacts remain private and ignored. ## Current next action -The local evaluation milestone is partial. The 20-run discoverability pilot -qualified all runs and kept all runs scope-safe, but its 10 focus runs made no -`go_context` calls. The 12-run integrated adoption follow-up also qualified -all runs and kept them scope-safe; all six integrated runs discovered the -shipped skill, used `go_context`, refreshed after editing, and used the -evidence. The integrated arm combines the skill and MCP surface, so it does -not establish MCP-alone causality or comparative product value. - -Inspect the six integrated transcripts to determine whether context or refresh -evidence changed a necessary edit or verification decision and to identify -avoidable repeated work behind the longer integrated duration. Use that -inspection to justify at most one bounded guidance improvement. Keep both -studies as regression evidence, not fresh proof of product superiority. Defer -any new release label until a reliability milestone passes. +The six integrated traces have been reviewed. No transcript proves that +`go_context` improved the necessary code edit. In gRPC run 3, refreshed context +showed broader affected scope and the agent explicitly said this prompted +`./...` verification. That attempt was incomplete due to the output cap; +focused verification later passed after a stale-snapshot rejection. Client-go +runs made 3-4 context calls each, including ambiguous or unhelpful initial +selections. gRPC runs made 4-5 calls each; runs 1 and 3 repeated verification +after incomplete or stale results, and run 2 had a failing root test run and a +failing narrowed observability rerun. + +The bounded guidance improvement is recorded in the +[`agentic-go-context` skill](../../.agents/skills/agentic-go-context/SKILL.md): +narrow an ambiguous result using returned current candidates instead of +repeating the broad query; finish each batch of edits and formatting before +refreshing; and refresh again after further edits or stale-snapshot rejection +while avoiding redundant refreshes when the snapshot is unchanged. This evidence does not +establish causal edit-quality or product value. The combined skill and MCP +workflow observations do not establish MCP-alone causality. Keep the studies +as regression evidence, not fresh proof of product superiority. External and +multi-model evaluation remain pending. Defer any new release label until a +reliability milestone passes. ## Standing continuation instruction @@ -298,7 +305,8 @@ approved direction, and unverified proposals visibly distinct. Replace stale status rather than appending a conversation transcript. Record relevant source paths, decisions, evidence actually gathered, limitations, and the exact next step. Do not add benchmark or evaluation claims unless a future request -explicitly includes them. +explicitly includes them. Context gathering does not execute checks or prove +correctness. ## Copy-paste continuation prompt diff --git a/docs/go-intelligence-north-star.md b/docs/go-intelligence-north-star.md index 44528c6..c915a09 100644 --- a/docs/go-intelligence-north-star.md +++ b/docs/go-intelligence-north-star.md @@ -51,8 +51,19 @@ runs discovered the shipped skill, used `go_context`, refreshed after editing, and used the evidence. The integrated arm combines the skill and MCP surface. These results do not establish MCP-alone causality, improved correctness, productivity, token efficiency, speed, statistical significance, or broad -model generalization. External and multi-model evaluation remain pending. The -historical adoption report is in +model generalization. External and multi-model evaluation remain pending. A +review of all six integrated traces found no transcript proving that context +improved the necessary code edit. In gRPC run 3, refreshed context prompted +broader `./...` verification, which was incomplete due to the output cap; +focused verification later passed after a stale-snapshot rejection. Client-go +runs made 3-4 context calls each and gRPC runs made 4-5, including ambiguous +or unhelpful initial selections and repeated verification after incomplete or +stale results. The bounded guidance improvement is to narrow ambiguity with +returned current candidates, finish each batch of edits and formatting before +refreshing, and refresh again after further edits or stale-snapshot rejection +while avoiding redundant refreshes when the snapshot is unchanged. This trace review does +not establish causal edit-quality or product value. The historical adoption +report is in [validation/v1.0.0/adoption-results.md](../validation/v1.0.0/adoption-results.md). Keep Go-only, local, deterministic operation; pinned gopls; source provenance; @@ -303,13 +314,11 @@ the workflow improves task outcomes. These records establish observed instruction-surface use and safety only. They do not support causal speed, token, reliability, adoption, performance, or generalization claims. Keep focus and full-replacement refresh; defer delta -refresh. The current next step is to inspect the six integrated transcripts -for whether context or refresh evidence changed a necessary edit or -verification decision, and for avoidable repeated work behind the longer -integrated duration. Use that inspection to justify at most one bounded -guidance improvement. Keep the studies as regression evidence, not fresh proof -of product superiority. Defer any new release label until a reliability -milestone passes. Raw artifacts remain private and ignored. +refresh. Trace review is complete and its bounded guidance improvement is +recorded in the project skill. Keep the studies as regression evidence, not +fresh proof of product superiority. External and multi-model evaluation remain +pending. Defer any new release label until a reliability milestone passes. +Raw artifacts remain private and ignored. ## Delivery order and completion criteria From 9e7e4b8b7a1c5be49ca6fd091c64aded9742048e Mon Sep 17 00:00:00 2001 From: Ashwin Gopalsamy <47941624+ashwingopalsamy@users.noreply.github.com> Date: Thu, 24 Sep 2026 01:41:44 +0530 Subject: [PATCH 13/20] docs(eval): record guidance regression signals --- docs/README.md | 14 ++++++++++++++ docs/continuation/go-intelligence.md | 20 ++++++++++++++++++++ docs/go-intelligence-north-star.md | 15 +++++++++++++++ 3 files changed, 49 insertions(+) diff --git a/docs/README.md b/docs/README.md index 54e702e..10b2bcf 100644 --- a/docs/README.md +++ b/docs/README.md @@ -58,6 +58,20 @@ snapshot is unchanged. This observation does not establish causal edit-quality o value. External and multi-model evaluation remain pending. Delta refresh remains deferred. These documents do not override frozen v1 contracts. +A post-guidance regression used six integrated Luna runs, with three +repetitions on each of two scenarios. All six qualified, passed acceptance, +stayed within scope, and required no operator intervention. All six called +`go_context`, refreshed after edits, and recorded evidence use. Five focus +calls failed: three client-go calls returned `invalid_input` for invalid +symbol references, and two calls in grpc-go run 1 returned `stale_snapshot` +because an observed semantic location was absent from the snapshot manifest. +grpc-go runs 2 and 3 had no failed focus calls. This shows workflow adoption +continued while exposing failure categories; it does not establish improved +quality, correctness, productivity, speed, or product value. Before changing +provider behavior, classify these failures as expected strict rejection, agent +misuse, or a reproducible provider defect. External and multi-model evaluation +remain pending. + ## Verification report contracts - [`v0.2.0-release-scope.md`](v0.2.0-release-scope.md) — change-aware diff --git a/docs/continuation/go-intelligence.md b/docs/continuation/go-intelligence.md index 41b8701..a593bd9 100644 --- a/docs/continuation/go-intelligence.md +++ b/docs/continuation/go-intelligence.md @@ -275,6 +275,26 @@ refresh. Raw artifacts remain private and ignored. ## Current next action +The post-guidance regression used six integrated Luna runs, with three +repetitions on each of two scenarios. All six qualified, passed acceptance, +stayed within scope, and required no operator intervention. All six used +`go_context`, refreshed after edits, and recorded evidence use. Five focus +calls failed: client-go had three `invalid_input` failures reporting invalid +symbol references; grpc-go run 1 had two `stale_snapshot` failures reporting +an observed semantic location absent from the snapshot manifest. grpc-go runs +2 and 3 had no failed focus calls. This shows continued workflow adoption +while exposing failure categories. It does not establish improved quality, +correctness, productivity, speed, or product value. External and multi-model +evaluation remain pending. + +Before changing provider behavior, classify the invalid-symbol-reference and +stale-snapshot failures as expected strict rejection, agent misuse, or a +reproducible provider defect. Inspect the run inputs, selectors, snapshot +identities, and provider responses to distinguish these cases, while preserving +strict stale rejection and the frozen public contracts. + +## Earlier trace review + The six integrated traces have been reviewed. No transcript proves that `go_context` improved the necessary code edit. In gRPC run 3, refreshed context showed broader affected scope and the agent explicitly said this prompted diff --git a/docs/go-intelligence-north-star.md b/docs/go-intelligence-north-star.md index c915a09..ca4e7c4 100644 --- a/docs/go-intelligence-north-star.md +++ b/docs/go-intelligence-north-star.md @@ -66,6 +66,21 @@ not establish causal edit-quality or product value. The historical adoption report is in [validation/v1.0.0/adoption-results.md](../validation/v1.0.0/adoption-results.md). +A post-guidance regression comprised six integrated Luna runs, with three +repetitions on each of two scenarios. All six qualified, passed acceptance, +had zero scope violations and zero operator interventions, and used +`go_context`, refreshed after edits, and recorded evidence use. Five focus +calls failed: three client-go calls reported `invalid_input` for invalid +symbol references, and two calls in grpc-go run 1 reported `stale_snapshot` +because an observed semantic location was absent from the snapshot manifest; +grpc-go runs 2 and 3 had no failed focus calls. Workflow adoption remained +possible while the failures exposed categories to investigate. The result does +not establish that guidance improved quality, correctness, productivity, +speed, or product value. Before changing provider behavior, classify the +invalid-symbol-reference and stale-snapshot failures as expected strict +rejection, agent misuse, or a reproducible provider defect. External and +multi-model evaluation remain pending. + Keep Go-only, local, deterministic operation; pinned gopls; source provenance; explicit uncertainty; and existing containment and guarded-refactor guarantees. The intelligence implementation owns context selection. gopls supplies semantic From 7b5111c6365a2a12806a561d86745d5bc50d7c9e Mon Sep 17 00:00:00 2001 From: Ashwin Gopalsamy <47941624+ashwingopalsamy@users.noreply.github.com> Date: Thu, 24 Sep 2026 14:58:07 +0530 Subject: [PATCH 14/20] fix(intelligence): preserve bounded focus evidence --- .agents/skills/agentic-go-context/SKILL.md | 6 ++-- docs/README.md | 14 ++++++-- docs/continuation/go-intelligence.md | 26 +++++++++++--- docs/go-intelligence-north-star.md | 16 ++++++--- internal/intelligence/focus.go | 6 ++++ internal/intelligence/focus_test.go | 25 +++++++++++++ internal/intelligence/semantic_gopls.go | 6 ++++ internal/intelligence/semantic_gopls_test.go | 37 ++++++++++++++++++++ internal/tools/intelligence_tools.go | 2 +- internal/tools/runtime.go | 2 +- validation/internal/adoption/types.go | 2 +- 11 files changed, 125 insertions(+), 17 deletions(-) diff --git a/.agents/skills/agentic-go-context/SKILL.md b/.agents/skills/agentic-go-context/SKILL.md index 2e1cbb2..675b63a 100644 --- a/.agents/skills/agentic-go-context/SKILL.md +++ b/.agents/skills/agentic-go-context/SKILL.md @@ -17,9 +17,11 @@ refresh after edits. `base` and `previous_pack_id` only; omit old selectors and Symbol Refs. Refresh again after further edits or stale-snapshot rejection; avoid redundant refreshes while the snapshot is unchanged. -4. Stale rejection is expected: stop and select current evidence; never bypass +4. Never reuse a Symbol Ref returned before an edit. After a stale rejection, + refresh or select a current candidate; do not retry the stale selector. +5. Stale rejection is expected: stop and select current evidence; never bypass it. -5. Use the result for impact and verification applicability. Treat impact and +6. Use the result for impact and verification applicability. Treat impact and reverse-dependency evidence as planning guidance, not authorization to edit affected packages; edit only the smallest task owner within supplied path/package scope. Skip trivial edits and preserve uncertainty when diff --git a/docs/README.md b/docs/README.md index 10b2bcf..e6d495a 100644 --- a/docs/README.md +++ b/docs/README.md @@ -68,9 +68,17 @@ because an observed semantic location was absent from the snapshot manifest. grpc-go runs 2 and 3 had no failed focus calls. This shows workflow adoption continued while exposing failure categories; it does not establish improved quality, correctness, productivity, speed, or product value. Before changing -provider behavior, classify these failures as expected strict rejection, agent -misuse, or a reproducible provider defect. External and multi-model evaluation -remain pending. +provider behavior, the audit classified the client-go failures as malformed or +reconstructed refs and the gRPC failures as unbound workspace locations. The +provider now omits unbound locations with bounded uncertainty while preserving +strict stale rejection. External and multi-model evaluation remain pending. + +After the failure remediation, six integrated Luna reruns completed with 6/6 +qualification, 6/6 acceptance, zero scope violations, zero operator +interventions, zero failed focus calls, and complete refresh and evidence-use +signals. The median duration was 411,537 ms and the median tool-call count was +31. This is diagnostic regression evidence only and does not establish product +value or comparative engineering benefit. ## Verification report contracts diff --git a/docs/continuation/go-intelligence.md b/docs/continuation/go-intelligence.md index a593bd9..d12d50c 100644 --- a/docs/continuation/go-intelligence.md +++ b/docs/continuation/go-intelligence.md @@ -287,11 +287,27 @@ while exposing failure categories. It does not establish improved quality, correctness, productivity, speed, or product value. External and multi-model evaluation remain pending. -Before changing provider behavior, classify the invalid-symbol-reference and -stale-snapshot failures as expected strict rejection, agent misuse, or a -reproducible provider defect. Inspect the run inputs, selectors, snapshot -identities, and provider responses to distinguish these cases, while preserving -strict stale rejection and the frozen public contracts. +The failure audit classified the three client-go invalid-input calls as agent +misuse or ref reconstruction: each failed ref had an invalid version field or +inconsistent identity, while valid query-issued refs succeeded. The two gRPC +stale-snapshot calls were repeated without an intervening edit; workspace-symbol +search returned a location inside the workspace but outside the active snapshot +manifest. The provider now omits such unbound workspace locations as bounded +uncertainty while continuing to propagate stale errors for manifest entries that +changed or disappeared. Strict stale rejection and the frozen public contracts +remain unchanged. + +Focused tests cover malformed refs, current observation membership, and omitted +workspace-symbol uncertainty. The private audit remains at +`/Users/ashwin/agentic-go-eval-audits/20260924-failure-verdicts.md`. + +The six-run integrated Luna rerun completed with 6/6 qualification, 6/6 +acceptance, zero scope violations, zero operator interventions, zero failed +focus calls, and complete refresh and evidence-use signals. Skill discovery was +6/6. Median duration was 411,537 ms, median tool calls were 31, and median +transcript evidence was 450,293 bytes. These are diagnostic regression values, +not evidence of improved speed, correctness, productivity, or product value. +External and multi-model evaluation remain pending. ## Earlier trace review diff --git a/docs/go-intelligence-north-star.md b/docs/go-intelligence-north-star.md index ca4e7c4..b7ac025 100644 --- a/docs/go-intelligence-north-star.md +++ b/docs/go-intelligence-north-star.md @@ -76,10 +76,18 @@ because an observed semantic location was absent from the snapshot manifest; grpc-go runs 2 and 3 had no failed focus calls. Workflow adoption remained possible while the failures exposed categories to investigate. The result does not establish that guidance improved quality, correctness, productivity, -speed, or product value. Before changing provider behavior, classify the -invalid-symbol-reference and stale-snapshot failures as expected strict -rejection, agent misuse, or a reproducible provider defect. External and -multi-model evaluation remain pending. +speed, or product value. The failure audit classified the three client-go +invalid-input calls as malformed or reconstructed Symbol Refs and the two +gRPC stale calls as workspace-symbol locations outside the active observation +manifest. Such locations are now omitted with bounded uncertainty; changed or +missing manifest entries still fail closed as stale. External and multi-model +evaluation remain pending. + +The follow-up six-run integrated Luna rerun passed qualification and acceptance +in every run, with zero scope violations, zero operator interventions, zero +failed focus calls, and complete refresh and evidence-use signals. Median +duration was 411,537 ms and median tool calls were 31. This remains diagnostic +regression evidence, not a claim of improved quality, speed, or product value. Keep Go-only, local, deterministic operation; pinned gopls; source provenance; explicit uncertainty; and existing containment and guarded-refactor guarantees. diff --git a/internal/intelligence/focus.go b/internal/intelligence/focus.go index f7d9cd2..0156ae6 100644 --- a/internal/intelligence/focus.go +++ b/internal/intelligence/focus.go @@ -415,6 +415,12 @@ func (c *Core) focusContext(ctx context.Context, observation *snapshotObservatio } else { result.Candidates = append(result.Candidates, normalized...) } + if matches.Omitted > 0 { + result.Uncertainties = append(result.Uncertainties, Uncertainty{Code: "semantic.external_locations", Message: fmt.Sprintf("%d workspace symbol locations were omitted because they were outside the active snapshot observation", matches.Omitted), Locations: []Location{}}) + result.EvidenceStates = append(result.EvidenceStates, EvidenceState{Facet: "declarations", State: "gathered_but_omitted", Reason: "some workspace symbol locations were outside the active snapshot observation"}) + result.Reasons = append(result.Reasons, "query returned incomplete workspace candidates; select a retained current candidate or use a narrower file or package selector") + break + } switch len(normalized) { case 0: result.Reasons = append(result.Reasons, "query returned no declarations") diff --git a/internal/intelligence/focus_test.go b/internal/intelligence/focus_test.go index 164b1d3..33dd43b 100644 --- a/internal/intelligence/focus_test.go +++ b/internal/intelligence/focus_test.go @@ -300,6 +300,31 @@ func TestFocusContextReturnsAmbiguousCandidatesBeforeExpansion(t *testing.T) { } } +func TestFocusContextPreservesUncertaintyForOmittedWorkspaceSymbols(t *testing.T) { + root := snapshotRepository(t) + snapshotter := newTestSnapshotter(t, root) + reader := &fakeSemanticReader{search: semanticSymbols{ + Items: []SymbolMatch{{Name: "Value", Qualified: "fixture.Value", Kind: "go.function", Package: "fixture", Location: Location{File: "main.go", Line: 3, Column: 1}}}, + Omitted: 1, + }} + core := newTestCore(t, snapshotter, reader) + observation, err := core.observe(context.Background(), "HEAD", "./...", "") + if err != nil { + t.Fatal(err) + } + defer observation.release() + result, err := core.focusContext(context.Background(), &observation, FocusRequest{Query: "value", MaxBytes: DefaultBriefBytes}) + if err != nil { + t.Fatal(err) + } + if result.Symbol != nil || !hasEvidenceState(result.EvidenceStates, "declarations", "gathered_but_omitted") || hasEvidenceState(result.EvidenceStates, "declarations", "examined_and_absent") { + t.Fatalf("focusContext(omitted query results) = %#v", result) + } + if !containsUncertainty(result.Uncertainties, "semantic.external_locations") { + t.Fatalf("focusContext(omitted query results) uncertainties = %#v", result.Uncertainties) + } +} + func hasEvidenceState(states []EvidenceState, facet, state string) bool { for _, candidate := range states { if candidate.Facet == facet && candidate.State == state { diff --git a/internal/intelligence/semantic_gopls.go b/internal/intelligence/semantic_gopls.go index 9eb9f85..abe5579 100644 --- a/internal/intelligence/semantic_gopls.go +++ b/internal/intelligence/semantic_gopls.go @@ -343,6 +343,12 @@ func (r *goplsReader) Search(ctx context.Context, query string) (semanticSymbols result.Omitted++ continue } + if r.observation != nil { + if _, found := observationRecord(r.observation.records, file); !found { + result.Omitted++ + continue + } + } match, err := r.symbol(candidate.Name, candidate.Kind, candidate.ContainerName, file, sourceRange) if err != nil { return semanticSymbols{}, err diff --git a/internal/intelligence/semantic_gopls_test.go b/internal/intelligence/semantic_gopls_test.go index a65a31b..d8e2626 100644 --- a/internal/intelligence/semantic_gopls_test.go +++ b/internal/intelligence/semantic_gopls_test.go @@ -291,6 +291,43 @@ func TestGoplsReaderNormalizesSymbolsLocationsAndDiagnostics(t *testing.T) { } } +func TestGoplsReaderOmitsWorkspaceSymbolsOutsideObservationManifest(t *testing.T) { + root := snapshotRepository(t) + snapshots := newTestSnapshotter(t, root) + root = snapshots.workspace.Root() + writeSnapshotFile(t, root, "value.go", "package sample\n\nfunc Value() {}\n") + writeSnapshotFile(t, root, "dependency.go", "package sample\n\nfunc Dependency() {}\n") + rpc := &fakeGoplsRPC{responses: map[string]json.RawMessage{ + "workspace/symbol": json.RawMessage(`[{ + "name":"Value","kind":12,"containerName":"example.test/sample", + "location":{"uri":"` + fileURI(root, "value.go") + `","range":{"start":{"line":2,"character":5},"end":{"line":2,"character":10}}} + },{ + "name":"Dependency","kind":12,"containerName":"example.test/dependency", + "location":{"uri":"` + fileURI(root, "dependency.go") + `","range":{"start":{"line":2,"character":5},"end":{"line":2,"character":15}}} + }]`), + }} + reader := &goplsReader{ + p: &goplsProvider{manager: rpc, root: root, workspace: snapshots.workspace}, + snapshot: SnapshotRef{ID: "snapshot", Scope: "./internal/xds/rbac"}, + observation: &snapshotObservation{ + lease: &manifestLease{snapshotter: snapshots, id: "snapshot"}, + records: []contentRecord{{Path: "value.go", Kind: "file", Digest: "value"}}, + sources: map[string][]byte{"value.go": []byte("package sample\n\nfunc Value() {}\n")}, + }, + } + search, err := reader.Search(context.Background(), "Value") + if err != nil { + t.Fatal(err) + } + if len(search.Items) != 1 || search.Omitted != 1 || search.Items[0].Name != "Value" { + t.Fatalf("search = %#v, want one retained symbol and one omitted location", search) + } + reader.observation.records = []contentRecord{{Path: "dependency.go", Kind: "file", Digest: "sha256:stale"}} + if _, err := reader.Search(context.Background(), "Dependency"); !errors.Is(err, ErrSnapshotChanged) { + t.Fatalf("changed observed workspace symbol error = %v, want ErrSnapshotChanged", err) + } +} + func TestGoplsReaderConvertsUTF16RangesToUTF8ByteColumns(t *testing.T) { root := snapshotRepository(t) snapshots := newTestSnapshotter(t, root) diff --git a/internal/tools/intelligence_tools.go b/internal/tools/intelligence_tools.go index 4e4899d..332a6d2 100644 --- a/internal/tools/intelligence_tools.go +++ b/internal/tools/intelligence_tools.go @@ -85,7 +85,7 @@ func RegisterSymbolContext(server *mcp.Server, runtime *Runtime) { // RegisterContext adds the post-v1 focus tool after the frozen registry. func RegisterContext(server *mcp.Server, runtime *Runtime) { - mcp.AddTool(server, &mcp.Tool{Name: "go_context", Description: "Before editing unfamiliar or cross-package Go code, call with base and one selector group (query, symbol_ref, file+line+column, or focus_file/focus_package) to map impact and verification applicability. After editing, refresh with base and previous_pack_id only; stale selectors and refs are expected to be rejected, so select current evidence.", Annotations: intelligenceAnnotations()}, runtime.context) + mcp.AddTool(server, &mcp.Tool{Name: "go_context", Description: "Before editing unfamiliar or cross-package Go code, call with base and one selector group (query, symbol_ref, file+line+column, or focus_file/focus_package) to map impact and verification applicability. After editing, refresh with base and previous_pack_id only; never reuse a Symbol Ref from before an edit; stale selectors and refs are expected to be rejected, so select current evidence.", Annotations: intelligenceAnnotations()}, runtime.context) } func (r *Runtime) requireIntelligence() (IntelligenceService, error) { diff --git a/internal/tools/runtime.go b/internal/tools/runtime.go index 567227c..da7b976 100644 --- a/internal/tools/runtime.go +++ b/internal/tools/runtime.go @@ -28,7 +28,7 @@ type Runtime struct { // ServerInstructions is the concise decision rule surfaced during MCP // initialize so clients can discover the focused context workflow. -const ServerInstructions = "For unfamiliar or cross-package Go changes, call go_context before editing with base and one selector (query, symbol_ref, file+line+column, or focus_file/focus_package). After editing, refresh with base and previous_pack_id only; stale selectors or refs mean select current evidence. Use it for impact and verification applicability, but treat reverse-dependency evidence as planning guidance: edit only the smallest task owner within supplied path/package scope; skip trivial edits." +const ServerInstructions = "For unfamiliar or cross-package Go changes, call go_context before editing with base and one selector (query, symbol_ref, file+line+column, or focus_file/focus_package). After editing, refresh with base and previous_pack_id only; stale selectors and refs mean select current evidence, and never reuse a Symbol Ref after an edit. Use it for impact and verification applicability; treat dependency evidence as planning guidance, edit only the smallest task owner within path/package scope, and skip trivial edits." // NewProductionServer creates the configured MCP server used by the binary. func NewProductionServer(implementation *mcp.Implementation) *mcp.Server { diff --git a/validation/internal/adoption/types.go b/validation/internal/adoption/types.go index f31fb4c..e4641aa 100644 --- a/validation/internal/adoption/types.go +++ b/validation/internal/adoption/types.go @@ -32,7 +32,7 @@ const ( ) // Guidance is the generic treatment instruction. -const Guidance = "For an unfamiliar Go change, call go_context with base and one selector: query, symbol_ref, or file with line and column. After editing, refresh with base and previous_pack_id only; stale selectors and refs are expected to be rejected." +const Guidance = "For an unfamiliar Go change, call go_context with base and one selector: query, symbol_ref, or file with line and column. After editing, refresh with base and previous_pack_id only; never reuse a Symbol Ref from before an edit. Stale selectors and refs are expected to be rejected, so select current evidence." // Scenario describes an adoption task. type Scenario struct { From 09b908f865611706336e057189c5b86705942c81 Mon Sep 17 00:00:00 2001 From: Ashwin Gopalsamy <47941624+ashwingopalsamy@users.noreply.github.com> Date: Sun, 27 Sep 2026 12:33:02 +0530 Subject: [PATCH 15/20] feat(sourceview): add exact branch worktree preview --- README.md | 28 + cmd/agentic-go/main.go | 2 + cmd/agentic-go/main_test.go | 71 +++ cmd/agentic-go/source_view.go | 89 ++++ internal/sourceview/sourceview.go | 682 +++++++++++++++++++++++++ internal/sourceview/sourceview_test.go | 250 +++++++++ 6 files changed, 1122 insertions(+) create mode 100644 cmd/agentic-go/source_view.go create mode 100644 internal/sourceview/sourceview.go create mode 100644 internal/sourceview/sourceview_test.go diff --git a/README.md b/README.md index 3a4c8cf..6969c83 100644 --- a/README.md +++ b/README.md @@ -17,6 +17,12 @@ `agentic-go` is a local Go MCP server and CLI. It gives an external coding agent semantic context, change continuity, guarded refactoring, and executed verification without embedding an LLM or becoming an agent framework. +The [product north star](docs/go-intelligence-north-star.md) targets Go engineers +using agents throughout understanding, editing, debugging, verification, and +review. It distinguishes current capabilities from planned workflow work and +unproven benefits across models. See the +[continuation handoff](docs/continuation/go-intelligence.md) for current status. + The v1.2.1 server exposes 15 MCP tools: the frozen v1 surface of 14 tools plus the additive `go_context` tool under `agentic.focus/v1`. This patch release makes the edit, refresh, verify, and inspect handoff explicit while preserving snapshot lineage and fail-closed evidence. The seven resources, resource template, six prompts, and frozen v1 schemas remain unchanged. ## Install @@ -106,6 +112,28 @@ For focused context before an edit, an MCP client can call `go_context` with the agentic-go context --base origin/main --query Worker --format text ``` +### Branch source-view preview + +The development branch also includes an explicit source-view command for +checking a branch in its own exact Git worktree: + +```sh +agentic-go source-view --workspace "$PWD" --branch feature/example \ + --output ../agentic-go-feature-example --format json +agentic-go mcp-config --client codex --workspace ../agentic-go-feature-example +``` + +Without `--branch`, it selects local `main`, or the configured +`origin/HEAD` when `main` is absent. Add `--include-dirty` only when the +source checkout's `HEAD` is exactly the selected commit; it carries staged, +unstaged, and regular untracked changes into the view. The output names the +branch ref, commit, tree, and any checkout limitations. Configure the agent +with the exact view root so snapshot checks can reject a branch that moved. +The source view is a visible detached worktree. This preview does not create a +persistent index or automatically switch a running MCP server, and it reports +uninitialized submodules as incomplete. External `go.work` or local +`replace` inputs are not captured. + ## Capabilities | Area | What the agent gets | diff --git a/cmd/agentic-go/main.go b/cmd/agentic-go/main.go index c7c484c..15d5e16 100644 --- a/cmd/agentic-go/main.go +++ b/cmd/agentic-go/main.go @@ -40,6 +40,8 @@ func run(args []string) int { return runVerify(args[1:], os.Stdout, os.Stderr) case "context": return runContext(args[1:], os.Stdout, os.Stderr) + case "source-view": + return runSourceView(args[1:], os.Stdout, os.Stderr) case "doctor": return runDoctor(args[1:], os.Stdout, os.Stderr, defaultDoctorDependencies()) case "mcp-config": diff --git a/cmd/agentic-go/main_test.go b/cmd/agentic-go/main_test.go index 0327c21..71909af 100644 --- a/cmd/agentic-go/main_test.go +++ b/cmd/agentic-go/main_test.go @@ -14,6 +14,7 @@ import ( "github.com/agentic-mcps/go/internal/changeimpact" "github.com/agentic-mcps/go/internal/execution" "github.com/agentic-mcps/go/internal/intelligence" + "github.com/agentic-mcps/go/internal/sourceview" "github.com/agentic-mcps/go/internal/verification" "github.com/agentic-mcps/go/internal/workspace" ) @@ -52,6 +53,76 @@ func TestRunContextValidatesArgumentsBeforeWorkspaceSetup(t *testing.T) { } } +func TestRunSourceViewSelectsDefaultAndExplicitBranches(t *testing.T) { + parent := t.TempDir() + repository := filepath.Join(parent, "repo") + if err := os.Mkdir(repository, 0o755); err != nil { + t.Fatal(err) + } + cliGit(t, repository, "init", "-b", "main") + cliGit(t, repository, "config", "user.name", "Fixture") + cliGit(t, repository, "config", "user.email", "fixture@example.test") + cliWrite(t, repository, "go.mod", "module example.test/sourceviewcli\n\ngo 1.25.0\n") + cliWrite(t, repository, "main.go", "package fixture\n\nfunc Branch() string { return \"main\" }\n") + cliGit(t, repository, "add", ".") + cliGit(t, repository, "-c", "commit.gpgsign=false", "commit", "-m", "main") + mainCommit := cliGit(t, repository, "rev-parse", "HEAD") + cliGit(t, repository, "checkout", "-b", "feature") + cliWrite(t, repository, "main.go", "package fixture\n\nfunc Branch() string { return \"feature\" }\n") + cliGit(t, repository, "add", ".") + cliGit(t, repository, "-c", "commit.gpgsign=false", "commit", "-m", "feature") + featureCommit := cliGit(t, repository, "rev-parse", "HEAD") + cliGit(t, repository, "checkout", "main") + + mainPath := filepath.Join(parent, "main-view") + var stdout, stderr bytes.Buffer + if exit := runSourceView([]string{"--workspace", repository, "--output", mainPath, "--format", "json"}, &stdout, &stderr); exit != 0 { + t.Fatalf("default source-view exit=%d stderr=%q stdout=%q", exit, stderr.String(), stdout.String()) + } + var mainView map[string]any + if err := json.Unmarshal(stdout.Bytes(), &mainView); err != nil { + t.Fatalf("decoding default source view: %v\n%s", err, stdout.String()) + } + if mainView["ref"] != "refs/heads/main" || mainView["commit"] != mainCommit || mainView["schema_version"] != sourceview.SchemaVersion { + t.Fatalf("default source view = %#v", mainView) + } + if got := cliGit(t, repository, "branch", "--show-current"); got != "main" { + t.Fatalf("source checkout branch = %q, want main", got) + } + + featurePath := filepath.Join(parent, "feature-view") + stdout.Reset() + stderr.Reset() + if exit := runSourceView([]string{"--workspace", repository, "--branch", "feature", "--output", featurePath, "--format", "json"}, &stdout, &stderr); exit != 0 { + t.Fatalf("feature source-view exit=%d stderr=%q stdout=%q", exit, stderr.String(), stdout.String()) + } + var featureView map[string]any + if err := json.Unmarshal(stdout.Bytes(), &featureView); err != nil { + t.Fatalf("decoding feature source view: %v\n%s", err, stdout.String()) + } + resolvedParent, err := filepath.EvalSymlinks(parent) + if err != nil { + t.Fatal(err) + } + if featureView["ref"] != "refs/heads/feature" || featureView["commit"] != featureCommit || featureView["view_path"] != filepath.Join(resolvedParent, "feature-view") { + t.Fatalf("feature source view = %#v", featureView) + } + contents, err := os.ReadFile(filepath.Join(featurePath, "main.go")) + if err != nil || !strings.Contains(string(contents), `return "feature"`) { + t.Fatalf("feature worktree contents = %q, error = %v", contents, err) + } +} + +func TestRunSourceViewRequiresExplicitNewOutputPath(t *testing.T) { + var stdout, stderr bytes.Buffer + if exit := runSourceView(nil, &stdout, &stderr); exit != 2 { + t.Fatalf("exit = %d, want 2", exit) + } + if stdout.Len() != 0 || !strings.Contains(stderr.String(), "--output is required") { + t.Fatalf("stdout=%q stderr=%q", stdout.String(), stderr.String()) + } +} + func TestContextJSONAndTextRenderSameFocusIdentity(t *testing.T) { result := intelligence.FocusResult{SchemaVersion: intelligence.FocusSchemaVersion, Snapshot: intelligence.SnapshotRef{ID: "snapshot-current"}, Change: verification.Change{Files: []verification.ChangedFile{}, Declarations: []verification.ChangedDeclaration{}, FilesTotal: 1}, Impact: verification.Impact{Packages: []verification.ImpactedPackage{}, PackagesTotal: 2}, Risks: []verification.RiskArea{}, Uncertainties: []verification.Uncertainty{}, Verification: intelligence.VerificationApplicability{Reasons: []string{"snapshot differs"}, NextAction: "request verification", Present: true}, Complete: true, PackID: strings.Repeat("a", 64), Refresh: &intelligence.FocusRefresh{PreviousPackID: strings.Repeat("b", 64), Status: "replaced", Message: "refreshed"}} var jsonOutput bytes.Buffer diff --git a/cmd/agentic-go/source_view.go b/cmd/agentic-go/source_view.go new file mode 100644 index 0000000..d4986ac --- /dev/null +++ b/cmd/agentic-go/source_view.go @@ -0,0 +1,89 @@ +package main + +import ( + "context" + "encoding/json" + "flag" + "fmt" + "io" + "os" + "os/signal" + "strings" + "syscall" + "time" + + "github.com/agentic-mcps/go/internal/execution" + "github.com/agentic-mcps/go/internal/sourceview" + "github.com/agentic-mcps/go/internal/workspace" +) + +func runSourceView(args []string, stdout, stderr io.Writer) int { + flags := flag.NewFlagSet("agentic-go source-view", flag.ContinueOnError) + flags.SetOutput(stderr) + workspacePath := flags.String("workspace", ".", "Git repository root containing the Go workspace") + branch := flags.String("branch", "", "local branch or full refs/heads or refs/remotes ref; defaults to main") + outputPath := flags.String("output", "", "new visible worktree directory outside the source repository") + includeDirty := flags.Bool("include-dirty", false, "overlay staged, unstaged, and regular untracked files when source HEAD exactly matches the selected branch") + format := flags.String("format", "text", "output format: text or json") + if err := flags.Parse(args); err != nil { + return 2 + } + if flags.NArg() != 0 || strings.TrimSpace(*outputPath) == "" || (*format != "text" && *format != "json") { + _, _ = fmt.Fprintln(stderr, "agentic-go source-view: --output is required; --format must be text or json") + return 2 + } + if invalidCLIArgument(*branch) || invalidCLIArgument(*workspacePath) { + _, _ = fmt.Fprintln(stderr, "agentic-go source-view: branch and workspace must each be one local argument") + return 2 + } + + ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM) + defer stop() + ctx, cancel := context.WithTimeout(ctx, 5*time.Minute) + defer cancel() + ws, err := workspace.Open(ctx, *workspacePath) + if err != nil { + _, _ = fmt.Fprintf(stderr, "agentic-go source-view: validating workspace: %v\n", err) + return 1 + } + runner, err := execution.New(ws, execution.Config{MaxConcurrent: 2, Timeout: 5 * time.Minute}) + if err != nil { + _, _ = fmt.Fprintf(stderr, "agentic-go source-view: creating process runner: %v\n", err) + return 1 + } + result, err := sourceview.Create(ctx, runner, ws, sourceview.Request{ + Branch: *branch, OutputPath: *outputPath, IncludeDirty: *includeDirty, + }) + if err != nil { + _, _ = fmt.Fprintf(stderr, "agentic-go source-view: %v\n", err) + return 1 + } + if err := writeSourceViewResult(stdout, *format, result); err != nil { + _, _ = fmt.Fprintf(stderr, "agentic-go source-view: writing result: %v\n", err) + return 1 + } + return 0 +} + +func writeSourceViewResult(writer io.Writer, format string, result sourceview.Result) error { + switch format { + case "json": + return json.NewEncoder(writer).Encode(result) + case "text": + coverage := "complete Git tree checkout" + if !result.CheckoutComplete { + coverage = "partial Git tree checkout" + } + if _, err := fmt.Fprintf(writer, "Branch view: %s\nCommit: %s\nTree: %s\nPath: %s\nDirty overlay: %t\nCheckout: %s\n", result.Ref, result.Commit, result.Tree, result.ViewPath, result.OverlayIncluded, coverage); err != nil { + return err + } + for _, limitation := range result.CheckoutLimitations { + if _, err := fmt.Fprintf(writer, "Checkout limitation: %s\n", limitation); err != nil { + return err + } + } + return nil + default: + return fmt.Errorf("unsupported output format %q", format) + } +} diff --git a/internal/sourceview/sourceview.go b/internal/sourceview/sourceview.go new file mode 100644 index 0000000..3c34ea0 --- /dev/null +++ b/internal/sourceview/sourceview.go @@ -0,0 +1,682 @@ +// Package sourceview creates exact, visible Git worktree views for a selected +// branch without changing the source checkout's branch or worktree. +package sourceview + +import ( + "bytes" + "context" + "crypto/sha256" + "encoding/base64" + "encoding/binary" + "encoding/hex" + "encoding/json" + "errors" + "fmt" + "io" + "io/fs" + "os" + "path" + "path/filepath" + "sort" + "strings" + "time" + "unicode/utf8" + + "github.com/agentic-mcps/go/internal/execution" + "github.com/agentic-mcps/go/internal/workspace" +) + +const ( + SchemaVersion = "agentic.branch-view/v1" + metadataConfigKey = "agentic-go.branch-view" + metadataRef = "refs/worktree/agentic-go/source-view" + maximumOverlaySize = 8 << 20 +) + +var ( + ErrStale = errors.New("branch source view is stale") +) + +// Marker is private per-worktree provenance used to reject a moved branch ref. +// It is stored in the linked worktree's Git metadata, not in repository files. +// +//nolint:govet // Field order matches the serialized source-view identity. +type Marker struct { + SchemaVersion string `json:"schema_version"` + Ref string `json:"ref"` + Commit string `json:"commit"` + Tree string `json:"tree"` + OverlayIncluded bool `json:"overlay_included"` + OverlayDigest string `json:"overlay_digest,omitempty"` +} + +// Request selects a branch source view and an optional exact dirty overlay. +type Request struct { + Branch string + OutputPath string + IncludeDirty bool +} + +// Result identifies the materialized source tree and any known checkout gaps. +// +//nolint:govet // Field order matches the CLI JSON response. +type Result struct { + SchemaVersion string `json:"schema_version"` + RequestedBranch string `json:"requested_branch"` + Branch string `json:"branch"` + Ref string `json:"ref"` + Commit string `json:"commit"` + Tree string `json:"tree"` + ViewPath string `json:"view_path"` + OverlayIncluded bool `json:"overlay_included"` + OverlayDigest string `json:"overlay_digest,omitempty"` + CheckoutComplete bool `json:"checkout_complete"` + CheckoutLimitations []string `json:"checkout_limitations"` +} + +type selectedBranch struct { + requested string + name string + ref string + commit string + tree string +} + +type overlayFile struct { + path string + mode fs.FileMode + content []byte +} + +type overlay struct { + patch []byte + files []overlayFile + digest string +} + +// Create resolves a branch to an exact commit, creates a detached worktree at +// the requested new path, and records the ref so later observations can reject +// the view if the branch moves. +func Create(ctx context.Context, runner *execution.Runner, source *workspace.Workspace, request Request) (result Result, returnErr error) { + if runner == nil { + return Result{}, errors.New("source-view runner is nil") + } + if source == nil { + return Result{}, errors.New("source-view workspace is nil") + } + repositoryRoot, err := gitText(ctx, runner, "rev-parse", "--show-toplevel") + if err != nil { + return Result{}, fmt.Errorf("resolving repository root: %w", err) + } + repositoryRoot, err = filepath.EvalSymlinks(repositoryRoot) + if err != nil { + return Result{}, fmt.Errorf("resolving repository root: %w", err) + } + if repositoryRoot != source.Root() { + return Result{}, errors.New("source-view workspace must be the Git repository root") + } + + outputPath, err := validateOutputPath(repositoryRoot, request.OutputPath) + if err != nil { + return Result{}, err + } + branch, err := resolveBranch(ctx, runner, request.Branch) + if err != nil { + return Result{}, err + } + limitations, err := checkoutLimitations(ctx, runner, branch.commit) + if err != nil { + limitations = []string{"Git could not determine whether the selected tree contains submodules"} + } + + var captured overlay + if request.IncludeDirty { + captured, err = captureOverlay(ctx, runner, source.Root(), branch.commit) + if err != nil { + return Result{}, fmt.Errorf("capturing exact dirty overlay: %w", err) + } + } + + created := false + complete := false + defer func() { + if complete || !created { + return + } + cleanupCtx, cancel := context.WithTimeout(context.Background(), 10*time.Second) + defer cancel() + if cleanupErr := removeWorktree(cleanupCtx, runner, outputPath); cleanupErr != nil { + returnErr = errors.Join(returnErr, fmt.Errorf("cleaning incomplete source view: %w", cleanupErr)) + } + }() + + if _, err := gitBytes(ctx, runner, "worktree", "add", "--detach", "--", outputPath, branch.commit); err != nil { + return Result{}, fmt.Errorf("creating detached branch view: %w", err) + } + created = true + + viewWorkspace, err := workspace.Open(ctx, outputPath) + if err != nil { + return Result{}, fmt.Errorf("validating selected branch workspace: %w", err) + } + viewRunner, err := runner.ForWorkspace(viewWorkspace) + if err != nil { + return Result{}, fmt.Errorf("creating selected branch runner: %w", err) + } + if request.IncludeDirty { + if len(captured.patch) > 0 { + if err := applyPatch(ctx, viewRunner, outputPath, captured.patch); err != nil { + return Result{}, fmt.Errorf("applying exact tracked overlay: %w", err) + } + } + if err := copyOverlayFiles(outputPath, captured.files); err != nil { + return Result{}, fmt.Errorf("copying untracked overlay: %w", err) + } + current, err := captureOverlay(ctx, runner, source.Root(), branch.commit) + if err != nil { + return Result{}, fmt.Errorf("rechecking source overlay: %w", err) + } + if current.digest != captured.digest { + return Result{}, errors.New("source checkout changed while the dirty overlay was captured") + } + } + + if err := writeMarker(ctx, viewRunner, Marker{ + SchemaVersion: SchemaVersion, Ref: branch.ref, Commit: branch.commit, + Tree: branch.tree, OverlayIncluded: request.IncludeDirty, + OverlayDigest: captured.digest, + }); err != nil { + return Result{}, fmt.Errorf("recording selected branch identity: %w", err) + } + if err := Validate(ctx, viewRunner, branch.commit); err != nil { + return Result{}, fmt.Errorf("validating selected branch identity: %w", err) + } + + result = Result{ + SchemaVersion: SchemaVersion, RequestedBranch: branch.requested, + Branch: branch.name, Ref: branch.ref, Commit: branch.commit, Tree: branch.tree, + ViewPath: outputPath, OverlayIncluded: request.IncludeDirty, + OverlayDigest: captured.digest, CheckoutComplete: len(limitations) == 0, + CheckoutLimitations: limitations, + } + complete = true + return result, nil +} + +// Validate rejects an invalid or moved branch-view marker. An ordinary +// checkout has no marker and is left unchanged. +func Validate(ctx context.Context, runner *execution.Runner, currentHead string) error { + marker, found, err := ReadMarker(ctx, runner) + if err != nil { + return err + } + if !found { + return nil + } + if !validObjectID(marker.Commit) || !validObjectID(marker.Tree) || !validBranchRef(marker.Ref) || marker.OverlayIncluded && !validDigest(marker.OverlayDigest) || !marker.OverlayIncluded && marker.OverlayDigest != "" { + return errors.New("branch source view metadata is malformed") + } + if marker.SchemaVersion != SchemaVersion { + return fmt.Errorf("unsupported branch source view schema %q", marker.SchemaVersion) + } + if currentHead != marker.Commit { + return fmt.Errorf("%w: workspace HEAD is %s, expected %s", ErrStale, currentHead, marker.Commit) + } + commit, found, err := resolveCommit(ctx, runner, marker.Ref) + if err != nil { + return fmt.Errorf("checking branch source view ref: %w", err) + } + if !found { + return fmt.Errorf("%w: ref %s no longer resolves", ErrStale, marker.Ref) + } + if commit != marker.Commit { + return fmt.Errorf("%w: ref %s moved from %s to %s", ErrStale, marker.Ref, marker.Commit, commit) + } + tree, err := gitText(ctx, runner, "rev-parse", "--verify", "--end-of-options", commit+"^{tree}") + if err != nil { + return fmt.Errorf("checking branch source view tree: %w", err) + } + if tree != marker.Tree { + return fmt.Errorf("%w: ref %s now identifies a different tree", ErrStale, marker.Ref) + } + return nil +} + +// ReadMarker returns per-worktree provenance, if present. +func ReadMarker(ctx context.Context, runner *execution.Runner) (Marker, bool, error) { + if runner == nil { + return Marker{}, false, errors.New("branch-view runner is nil") + } + gitDir, err := gitText(ctx, runner, "rev-parse", "--absolute-git-dir") + if err != nil { + return Marker{}, false, fmt.Errorf("resolving worktree Git directory: %w", err) + } + if !filepath.IsAbs(gitDir) { + return Marker{}, false, errors.New("Git returned a non-absolute worktree directory") + } + configPath := filepath.Join(gitDir, "config.worktree") + value, exitCode, err := gitRun(ctx, runner, "config", "--file", configPath, "--get", metadataConfigKey) + if err != nil { + return Marker{}, false, fmt.Errorf("reading branch source view metadata: %w", err) + } + if exitCode == 1 { + markerCommit, found, markerErr := resolveCommit(ctx, runner, metadataRef) + if markerErr != nil { + return Marker{}, false, fmt.Errorf("checking branch source view sentinel: %w", markerErr) + } + if found { + return Marker{}, false, fmt.Errorf("branch source view marker is missing for private ref %s", markerCommit) + } + return Marker{}, false, nil + } + if exitCode != 0 { + return Marker{}, false, fmt.Errorf("reading branch source view metadata: git exited with status %d", exitCode) + } + encoded := strings.TrimSpace(string(value)) + data, err := base64.RawURLEncoding.DecodeString(encoded) + if err != nil { + return Marker{}, false, fmt.Errorf("decoding branch source view metadata: %w", err) + } + var marker Marker + if err := json.Unmarshal(data, &marker); err != nil { + return Marker{}, false, fmt.Errorf("parsing branch source view metadata: %w", err) + } + markerCommit, found, err := resolveCommit(ctx, runner, metadataRef) + if err != nil { + return Marker{}, false, fmt.Errorf("checking branch source view sentinel: %w", err) + } + if !found || markerCommit != marker.Commit { + return Marker{}, false, errors.New("branch source view metadata and private sentinel disagree") + } + return marker, true, nil +} + +func resolveBranch(ctx context.Context, runner *execution.Runner, requested string) (selectedBranch, error) { + branchName := strings.TrimSpace(requested) + if branchName != requested { + return selectedBranch{}, errors.New("branch name must not contain leading or trailing whitespace") + } + ref := "" + if branchName == "" { + branchName = "main" + ref = "refs/heads/main" + if _, found, err := resolveCommit(ctx, runner, ref); err != nil { + return selectedBranch{}, err + } else if !found { + remoteHead, exitCode, symbolicErr := gitRun(ctx, runner, "symbolic-ref", "--quiet", "refs/remotes/origin/HEAD") + if symbolicErr != nil { + return selectedBranch{}, fmt.Errorf("resolving configured default branch: %w", symbolicErr) + } + if exitCode != 0 || !strings.HasPrefix(strings.TrimSpace(string(remoteHead)), "refs/remotes/") { + return selectedBranch{}, errors.New("branch main does not exist and origin/HEAD is not configured; select a branch explicitly") + } + ref = strings.TrimSpace(string(remoteHead)) + branchName = strings.TrimPrefix(ref, "refs/remotes/") + } + } else if strings.HasPrefix(branchName, "refs/heads/") || strings.HasPrefix(branchName, "refs/remotes/") { + ref = branchName + } else if strings.HasPrefix(branchName, "refs/") || strings.HasPrefix(branchName, "-") || strings.ContainsRune(branchName, 0) { + return selectedBranch{}, errors.New("branch must be a local branch name or a full refs/heads or refs/remotes ref") + } else { + ref = "refs/heads/" + branchName + } + if err := checkBranchRef(ctx, runner, ref); err != nil { + return selectedBranch{}, err + } + commit, found, err := resolveCommit(ctx, runner, ref) + if err != nil { + return selectedBranch{}, err + } + if !found { + return selectedBranch{}, fmt.Errorf("branch ref %s does not resolve to a commit", ref) + } + tree, err := gitText(ctx, runner, "rev-parse", "--verify", "--end-of-options", commit+"^{tree}") + if err != nil || !validObjectID(tree) { + if err == nil { + err = errors.New("Git returned an invalid tree object ID") + } + return selectedBranch{}, fmt.Errorf("resolving branch tree: %w", err) + } + canonicalName := strings.TrimPrefix(strings.TrimPrefix(ref, "refs/heads/"), "refs/remotes/") + return selectedBranch{requested: branchName, name: canonicalName, ref: ref, commit: commit, tree: tree}, nil +} + +func checkBranchRef(ctx context.Context, runner *execution.Runner, ref string) error { + if !validBranchRef(ref) { + return fmt.Errorf("branch ref %q is invalid", ref) + } + _, exitCode, err := gitRun(ctx, runner, "check-ref-format", ref) + if err != nil { + return fmt.Errorf("validating branch ref: %w", err) + } + if exitCode != 0 { + return fmt.Errorf("branch ref %q is invalid", ref) + } + return nil +} + +func validBranchRef(ref string) bool { + return strings.HasPrefix(ref, "refs/heads/") || strings.HasPrefix(ref, "refs/remotes/") +} + +func resolveCommit(ctx context.Context, runner *execution.Runner, ref string) (string, bool, error) { + output, exitCode, err := gitRun(ctx, runner, "rev-parse", "--verify", "--end-of-options", ref+"^{commit}") + if err != nil { + return "", false, err + } + if exitCode != 0 { + return "", false, nil + } + commit := strings.TrimSpace(string(output)) + if !validObjectID(commit) { + return "", false, errors.New("Git returned an invalid commit object ID") + } + return commit, true, nil +} + +func validateOutputPath(repositoryRoot, requested string) (string, error) { + if strings.TrimSpace(requested) == "" { + return "", errors.New("--output is required") + } + absolute, err := filepath.Abs(requested) + if err != nil { + return "", fmt.Errorf("resolving output path: %w", err) + } + absolute = filepath.Clean(absolute) + parent, err := filepath.EvalSymlinks(filepath.Dir(absolute)) + if err != nil { + return "", fmt.Errorf("resolving output parent: %w", err) + } + parentInfo, err := os.Stat(parent) + if err != nil || !parentInfo.IsDir() { + return "", errors.New("output parent must be an existing directory") + } + absolute = filepath.Join(parent, filepath.Base(absolute)) + if _, err := os.Lstat(absolute); err == nil { + return "", fmt.Errorf("output path %q already exists", absolute) + } else if !errors.Is(err, fs.ErrNotExist) { + return "", fmt.Errorf("checking output path: %w", err) + } + if inside, err := pathWithin(repositoryRoot, absolute); err != nil || inside { + return "", errors.New("output path must be outside the source repository") + } + if contains, err := pathWithin(absolute, repositoryRoot); err != nil || contains { + return "", errors.New("output path must not contain the source repository") + } + return absolute, nil +} + +func pathWithin(root, candidate string) (bool, error) { + relative, err := filepath.Rel(root, candidate) + if err != nil { + return false, err + } + return relative == "." || relative != ".." && !strings.HasPrefix(relative, ".."+string(filepath.Separator)), nil +} + +func checkoutLimitations(ctx context.Context, runner *execution.Runner, commit string) ([]string, error) { + output, err := gitText(ctx, runner, "ls-tree", "-r", "--format=%(objectmode)", commit) + if err != nil { + return nil, err + } + modules := 0 + for _, mode := range strings.Fields(output) { + if mode == "160000" { + modules++ + } + } + if modules == 0 { + return []string{}, nil + } + return []string{fmt.Sprintf("%d Git submodule entries are not initialized in this branch view", modules)}, nil +} + +func captureOverlay(ctx context.Context, runner *execution.Runner, sourceRoot, commit string) (overlay, error) { + head, err := gitText(ctx, runner, "rev-parse", "--verify", "HEAD^{commit}") + if err != nil { + return overlay{}, fmt.Errorf("resolving source checkout HEAD: %w", err) + } + if head != commit { + return overlay{}, fmt.Errorf("source checkout HEAD %s does not match selected branch commit %s", head, commit) + } + unmerged, err := gitText(ctx, runner, "ls-files", "-u", "-z", "--") + if err != nil { + return overlay{}, fmt.Errorf("checking unresolved source merges: %w", err) + } + if unmerged != "" { + return overlay{}, errors.New("source checkout has unresolved merge entries") + } + patch, err := gitBytes(ctx, runner, "diff", "--binary", "--full-index", "--no-ext-diff", "--no-textconv", "--no-renames", commit, "--") + if err != nil { + return overlay{}, fmt.Errorf("capturing tracked source changes: %w", err) + } + listed, err := gitBytes(ctx, runner, "ls-files", "--others", "--exclude-standard", "--full-name", "-z", "--") + if err != nil { + return overlay{}, fmt.Errorf("listing untracked source files: %w", err) + } + paths := make([]string, 0) + for _, raw := range bytes.Split(listed, []byte{0}) { + if len(raw) == 0 { + continue + } + path := filepath.ToSlash(string(raw)) + if !fs.ValidPath(path) || !utf8.ValidString(path) || strings.HasPrefix(path, ".git/") || path == ".git" { + return overlay{}, fmt.Errorf("untracked source path %q is unsupported", path) + } + paths = append(paths, path) + } + sort.Strings(paths) + files := make([]overlayFile, 0, len(paths)) + contentBytes := len(patch) + sourceDirectory, err := os.OpenRoot(sourceRoot) + if err != nil { + return overlay{}, fmt.Errorf("opening source workspace root: %w", err) + } + defer sourceDirectory.Close() + for _, path := range paths { + if err := ctx.Err(); err != nil { + return overlay{}, err + } + info, err := sourceDirectory.Lstat(path) + if err != nil { + return overlay{}, fmt.Errorf("inspecting untracked file %q: %w", path, err) + } + if !info.Mode().IsRegular() { + return overlay{}, fmt.Errorf("untracked path %q is not a regular file; symlink and special-file overlays are unsupported", path) + } + if info.Size() < 0 || info.Size() > int64(maximumOverlaySize-contentBytes) { + return overlay{}, fmt.Errorf("dirty overlay exceeds the %d-byte limit", maximumOverlaySize) + } + opened, err := sourceDirectory.Open(path) + if err != nil { + return overlay{}, fmt.Errorf("reading untracked file %q: %w", path, err) + } + content, readErr := io.ReadAll(io.LimitReader(opened, int64(maximumOverlaySize-contentBytes)+1)) + closeFileErr := opened.Close() + if readErr != nil { + return overlay{}, fmt.Errorf("reading untracked file %q: %w", path, readErr) + } + if closeFileErr != nil { + return overlay{}, fmt.Errorf("closing untracked file %q: %w", path, closeFileErr) + } + contentBytes += len(content) + if contentBytes > maximumOverlaySize { + return overlay{}, fmt.Errorf("dirty overlay exceeds the %d-byte limit", maximumOverlaySize) + } + mode := fs.FileMode(0o644) + if info.Mode().Perm()&0o111 != 0 { + mode = 0o755 + } + files = append(files, overlayFile{path: path, mode: mode, content: content}) + } + digest := digestOverlay(patch, files) + return overlay{patch: patch, files: files, digest: digest}, nil +} + +func digestOverlay(patch []byte, files []overlayFile) string { + hash := sha256.New() + _, _ = hash.Write([]byte("agentic-go/branch-overlay/v1\x00")) + writeHashField(hash, patch) + for _, file := range files { + writeHashField(hash, []byte(file.path)) + var mode [4]byte + binary.BigEndian.PutUint32(mode[:], uint32(file.mode.Perm())) + _, _ = hash.Write(mode[:]) + writeHashField(hash, file.content) + } + return hex.EncodeToString(hash.Sum(nil)) +} + +func writeHashField(hash interface{ Write([]byte) (int, error) }, value []byte) { + var size [8]byte + binary.BigEndian.PutUint64(size[:], uint64(len(value))) + _, _ = hash.Write(size[:]) + _, _ = hash.Write(value) +} + +func applyPatch(ctx context.Context, runner *execution.Runner, viewPath string, patch []byte) error { + file, err := os.CreateTemp(viewPath, ".agentic-go-overlay-*.patch") + if err != nil { + return fmt.Errorf("creating temporary overlay patch: %w", err) + } + patchPath := file.Name() + defer func() { _ = os.Remove(patchPath) }() + if _, err := file.Write(patch); err != nil { + _ = file.Close() + return fmt.Errorf("writing temporary overlay patch: %w", err) + } + if err := file.Close(); err != nil { + return fmt.Errorf("closing temporary overlay patch: %w", err) + } + _, err = gitBytes(ctx, runner, "apply", "--binary", "--whitespace=nowarn", patchPath) + return err +} + +func copyOverlayFiles(root string, files []overlayFile) error { + viewRoot, err := os.OpenRoot(root) + if err != nil { + return fmt.Errorf("opening selected worktree root: %w", err) + } + defer viewRoot.Close() + for _, file := range files { + if err := makeContainedParents(viewRoot, file.path); err != nil { + return err + } + if _, err := viewRoot.Lstat(file.path); err == nil { + return fmt.Errorf("untracked path %q collides with selected branch content", file.path) + } else if !errors.Is(err, fs.ErrNotExist) { + return fmt.Errorf("checking overlay destination %q: %w", file.path, err) + } + created, err := viewRoot.OpenFile(file.path, os.O_WRONLY|os.O_CREATE|os.O_EXCL, file.mode) + if err != nil { + return fmt.Errorf("creating overlay file %q: %w", file.path, err) + } + _, writeErr := created.Write(file.content) + closeErr := created.Close() + if writeErr != nil { + return fmt.Errorf("writing overlay file %q: %w", file.path, writeErr) + } + if closeErr != nil { + return fmt.Errorf("closing overlay file %q: %w", file.path, closeErr) + } + } + return nil +} + +func makeContainedParents(root *os.Root, relative string) error { + if !fs.ValidPath(relative) { + return fmt.Errorf("overlay path %q is invalid", relative) + } + parts := strings.Split(path.Dir(relative), "/") + current := "" + for _, part := range parts { + if part == "." || part == "" { + continue + } + if current == "" { + current = part + } else { + current += "/" + part + } + info, err := root.Lstat(current) + if errors.Is(err, fs.ErrNotExist) { + if err := root.Mkdir(current, 0o755); err != nil && !errors.Is(err, fs.ErrExist) { + return fmt.Errorf("creating overlay directory: %w", err) + } + info, err = root.Lstat(current) + } + if err != nil { + return fmt.Errorf("inspecting overlay directory: %w", err) + } + if !info.IsDir() || info.Mode()&os.ModeSymlink != 0 { + return fmt.Errorf("overlay path %q traverses a non-directory or symlink", relative) + } + } + return nil +} + +func writeMarker(ctx context.Context, runner *execution.Runner, marker Marker) error { + data, err := json.Marshal(marker) + if err != nil { + return err + } + gitDir, err := gitText(ctx, runner, "rev-parse", "--absolute-git-dir") + if err != nil { + return err + } + if !filepath.IsAbs(gitDir) { + return errors.New("Git returned a non-absolute worktree directory") + } + encoded := base64.RawURLEncoding.EncodeToString(data) + if _, err := gitBytes(ctx, runner, "config", "--file", filepath.Join(gitDir, "config.worktree"), "--replace-all", metadataConfigKey, encoded); err != nil { + return err + } + _, err = gitBytes(ctx, runner, "update-ref", metadataRef, marker.Commit) + return err +} + +func removeWorktree(ctx context.Context, runner *execution.Runner, path string) error { + _, err := gitBytes(ctx, runner, "worktree", "remove", "--force", "--", path) + return err +} + +func gitText(ctx context.Context, runner *execution.Runner, args ...string) (string, error) { + output, err := gitBytes(ctx, runner, args...) + return strings.TrimSpace(string(output)), err +} + +func gitBytes(ctx context.Context, runner *execution.Runner, args ...string) ([]byte, error) { + output, exitCode, err := gitRun(ctx, runner, args...) + if err != nil { + return nil, err + } + if exitCode != 0 { + return nil, fmt.Errorf("git %s exited with status %d", args[0], exitCode) + } + return output, nil +} + +func gitRun(ctx context.Context, runner *execution.Runner, args ...string) ([]byte, int, error) { + var stdout, stderr bytes.Buffer + result, err := runner.Run(ctx, execution.Command{Name: "git", Args: args}, execution.Streams{Stdout: &stdout, Stderr: &stderr}) + if err != nil { + return stdout.Bytes(), 0, err + } + return stdout.Bytes(), result.ExitCode, nil +} + +func validObjectID(value string) bool { + if len(value) != 40 && len(value) != 64 { + return false + } + _, err := hex.DecodeString(value) + return err == nil +} + +func validDigest(value string) bool { + if len(value) != 64 { + return false + } + _, err := hex.DecodeString(value) + return err == nil +} diff --git a/internal/sourceview/sourceview_test.go b/internal/sourceview/sourceview_test.go new file mode 100644 index 0000000..2676783 --- /dev/null +++ b/internal/sourceview/sourceview_test.go @@ -0,0 +1,250 @@ +package sourceview + +import ( + "context" + "errors" + "os" + "os/exec" + "path/filepath" + "strings" + "testing" + + "github.com/agentic-mcps/go/internal/execution" + "github.com/agentic-mcps/go/internal/workspace" +) + +func TestCreateDefaultsToMainAndSelectsExplicitFeatureBranch(t *testing.T) { + repository, parent := sourceViewRepository(t) + mainCommit := sourceViewGit(t, repository, "rev-parse", "HEAD") + source, runner := sourceViewRunner(t, repository) + + mainPath := filepath.Join(parent, "main-view") + mainView, err := Create(context.Background(), runner, source, Request{OutputPath: mainPath}) + if err != nil { + t.Fatal(err) + } + if mainView.Branch != "main" || mainView.Ref != "refs/heads/main" || mainView.Commit != mainCommit { + t.Fatalf("default view identity = %#v", mainView) + } + if got := sourceViewRead(t, mainPath, "branch.go"); !strings.Contains(got, `return "main"`) { + t.Fatalf("default branch contents = %q", got) + } + if got := sourceViewGit(t, repository, "branch", "--show-current"); got != "main" { + t.Fatalf("source checkout branch = %q, want main", got) + } + + featurePath := filepath.Join(parent, "feature-view") + featureView, err := Create(context.Background(), runner, source, Request{Branch: "feature", OutputPath: featurePath}) + if err != nil { + t.Fatal(err) + } + if featureView.Branch != "feature" || featureView.Ref != "refs/heads/feature" || featureView.Commit == mainView.Commit { + t.Fatalf("feature view identity = %#v", featureView) + } + if got := sourceViewRead(t, featurePath, "branch.go"); !strings.Contains(got, `return "feature"`) { + t.Fatalf("feature branch contents = %q", got) + } + if got := sourceViewRead(t, featurePath, "branch_test.go"); !strings.Contains(got, "TestFeatureBehavior") { + t.Fatalf("feature branch test was not checked out: %q", got) + } + if got := sourceViewRead(t, featurePath, "README.md"); !strings.Contains(got, "feature behavior") { + t.Fatalf("feature branch documentation was not checked out: %q", got) + } + + if err := os.WriteFile(filepath.Join(featurePath, "branch.go"), []byte("package fixture\n\nfunc Branch() string { return \"edited view\" }\n"), 0o644); err != nil { + t.Fatal(err) + } + if got := sourceViewRead(t, repository, "branch.go"); !strings.Contains(got, `return "main"`) { + t.Fatalf("editing the returned view changed source checkout: %q", got) + } +} + +func TestCreateOverlaysStagedUnstagedUntrackedAddedAndDeletedFiles(t *testing.T) { + repository, parent := sourceViewRepository(t) + source, runner := sourceViewRunner(t, repository) + sourceViewWrite(t, repository, "branch.go", "package fixture\n\nfunc Branch() string { return \"staged\" }\n") + sourceViewGit(t, repository, "add", "branch.go") + sourceViewWrite(t, repository, "branch.go", "package fixture\n\nfunc Branch() string { return \"unstaged final\" }\n") + sourceViewWrite(t, repository, "staged-added.go", "package fixture\n\nfunc StagedAdded() {}\n") + sourceViewGit(t, repository, "add", "staged-added.go") + sourceViewWrite(t, repository, "untracked.go", "package fixture\n\nfunc Untracked() {}\n") + if err := os.Remove(filepath.Join(repository, "remove.go")); err != nil { + t.Fatal(err) + } + + viewPath := filepath.Join(parent, "overlay-view") + view, err := Create(context.Background(), runner, source, Request{OutputPath: viewPath, IncludeDirty: true}) + if err != nil { + t.Fatal(err) + } + if !view.OverlayIncluded || len(view.OverlayDigest) != 64 { + t.Fatalf("overlay identity = %#v", view) + } + if got := sourceViewRead(t, viewPath, "branch.go"); !strings.Contains(got, `return "unstaged final"`) { + t.Fatalf("staged and unstaged overlay = %q", got) + } + if got := sourceViewRead(t, viewPath, "staged-added.go"); !strings.Contains(got, "StagedAdded") { + t.Fatalf("staged addition = %q", got) + } + if got := sourceViewRead(t, viewPath, "untracked.go"); !strings.Contains(got, "Untracked") { + t.Fatalf("untracked addition = %q", got) + } + if _, err := os.Stat(filepath.Join(viewPath, "remove.go")); !errors.Is(err, os.ErrNotExist) { + t.Fatalf("deleted path stat error = %v, want not-exist", err) + } + if got := sourceViewRead(t, repository, "branch.go"); !strings.Contains(got, `return "unstaged final"`) { + t.Fatalf("source overlay changed while materializing view: %q", got) + } +} + +func TestCreateRejectsOverlayAgainstDifferentBranch(t *testing.T) { + repository, parent := sourceViewRepository(t) + source, runner := sourceViewRunner(t, repository) + viewPath := filepath.Join(parent, "wrong-base-view") + _, err := Create(context.Background(), runner, source, Request{Branch: "feature", OutputPath: viewPath, IncludeDirty: true}) + if err == nil || !strings.Contains(err.Error(), "does not match selected branch commit") { + t.Fatalf("Create() error = %v, want exact-base rejection", err) + } + if _, err := os.Lstat(viewPath); !errors.Is(err, os.ErrNotExist) { + t.Fatalf("mismatched overlay output stat error = %v, want not-exist", err) + } +} + +func TestValidateRejectsMovedBranchAndCorruptMetadata(t *testing.T) { + repository, parent := sourceViewRepository(t) + source, runner := sourceViewRunner(t, repository) + viewPath := filepath.Join(parent, "stale-view") + view, err := Create(context.Background(), runner, source, Request{Branch: "feature", OutputPath: viewPath}) + if err != nil { + t.Fatal(err) + } + viewWorkspace, viewRunner := sourceViewRunner(t, viewPath) + if err := Validate(context.Background(), viewRunner, view.Commit); err != nil { + t.Fatalf("Validate() before ref movement = %v", err) + } + mainCommit := sourceViewGit(t, repository, "rev-parse", "refs/heads/main") + sourceViewGit(t, repository, "update-ref", "refs/heads/feature", mainCommit) + if err := Validate(context.Background(), viewRunner, view.Commit); !errors.Is(err, ErrStale) { + t.Fatalf("Validate() after ref movement = %v, want ErrStale", err) + } + + gitDir := sourceViewGit(t, viewWorkspace.Root(), "rev-parse", "--absolute-git-dir") + configPath := filepath.Join(gitDir, "config.worktree") + sourceViewGit(t, viewPath, "config", "--file", configPath, "--replace-all", metadataConfigKey, "not-base64") + if err := Validate(context.Background(), viewRunner, view.Commit); err == nil || !strings.Contains(err.Error(), "branch source view metadata") { + t.Fatalf("Validate() with corrupt metadata = %v, want metadata error", err) + } + sourceViewGit(t, viewPath, "config", "--file", configPath, "--unset-all", metadataConfigKey) + if err := Validate(context.Background(), viewRunner, view.Commit); err == nil || !strings.Contains(err.Error(), "branch source view marker is missing") { + t.Fatalf("Validate() with missing metadata = %v, want fail-closed marker error", err) + } +} + +func TestCreateCleansPartialWorktreeWhenSelectedBranchIsNotGoWorkspace(t *testing.T) { + repository, parent := sourceViewRepository(t) + source, runner := sourceViewRunner(t, repository) + sourceViewGit(t, repository, "checkout", "-b", "no-go-workspace") + if err := os.Remove(filepath.Join(repository, "go.mod")); err != nil { + t.Fatal(err) + } + sourceViewGit(t, repository, "add", "go.mod") + sourceViewGit(t, repository, "-c", "commit.gpgsign=false", "commit", "-m", "remove Go workspace") + sourceViewGit(t, repository, "checkout", "main") + viewPath := filepath.Join(parent, "invalid-view") + _, err := Create(context.Background(), runner, source, Request{Branch: "no-go-workspace", OutputPath: viewPath}) + if err == nil || !strings.Contains(err.Error(), "validating selected branch workspace") { + t.Fatalf("Create() error = %v, want workspace validation error", err) + } + if _, err := os.Lstat(viewPath); !errors.Is(err, os.ErrNotExist) { + t.Fatalf("partial view stat error = %v, want not-exist", err) + } + worktrees := sourceViewGit(t, repository, "worktree", "list", "--porcelain") + if strings.Contains(worktrees, viewPath) { + t.Fatalf("failed source view remains registered:\n%s", worktrees) + } +} + +func TestCreateRequiresNewOutputOutsideSourceRepository(t *testing.T) { + repository, _ := sourceViewRepository(t) + source, runner := sourceViewRunner(t, repository) + for name, output := range map[string]string{ + "inside repository": filepath.Join(repository, "nested-view"), + "existing path": repository, + } { + t.Run(name, func(t *testing.T) { + if _, err := Create(context.Background(), runner, source, Request{OutputPath: output}); err == nil { + t.Fatal("Create() succeeded for unsafe output path") + } + }) + } +} + +func sourceViewRepository(t *testing.T) (string, string) { + t.Helper() + parent := t.TempDir() + repository := filepath.Join(parent, "repo") + if err := os.Mkdir(repository, 0o755); err != nil { + t.Fatal(err) + } + sourceViewGit(t, repository, "init", "-b", "main") + sourceViewGit(t, repository, "config", "user.name", "Fixture") + sourceViewGit(t, repository, "config", "user.email", "fixture@example.test") + sourceViewWrite(t, repository, "go.mod", "module example.test/sourceview\n\ngo 1.25.0\n") + sourceViewWrite(t, repository, "branch.go", "package fixture\n\nfunc Branch() string { return \"main\" }\n") + sourceViewWrite(t, repository, "remove.go", "package fixture\n\nfunc RemoveMe() {}\n") + sourceViewWrite(t, repository, "README.md", "main behavior\n") + sourceViewGit(t, repository, "add", ".") + sourceViewGit(t, repository, "-c", "commit.gpgsign=false", "commit", "-m", "main source") + sourceViewGit(t, repository, "checkout", "-b", "feature") + sourceViewWrite(t, repository, "branch.go", "package fixture\n\nfunc Branch() string { return \"feature\" }\n") + sourceViewWrite(t, repository, "branch_test.go", "package fixture\n\nfunc TestFeatureBehavior() {}\n") + sourceViewWrite(t, repository, "README.md", "feature behavior\n") + sourceViewGit(t, repository, "add", ".") + sourceViewGit(t, repository, "-c", "commit.gpgsign=false", "commit", "-m", "feature source") + sourceViewGit(t, repository, "checkout", "main") + return repository, parent +} + +func sourceViewRunner(t *testing.T, root string) (*workspace.Workspace, *execution.Runner) { + t.Helper() + ws, err := workspace.Open(context.Background(), root) + if err != nil { + t.Fatal(err) + } + runner, err := execution.New(ws, execution.Config{}) + if err != nil { + t.Fatal(err) + } + return ws, runner +} + +func sourceViewRead(t *testing.T, root, name string) string { + t.Helper() + contents, err := os.ReadFile(filepath.Join(root, name)) + if err != nil { + t.Fatal(err) + } + return string(contents) +} + +func sourceViewWrite(t *testing.T, root, name, contents string) { + t.Helper() + path := filepath.Join(root, name) + if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(path, []byte(contents), 0o644); err != nil { + t.Fatal(err) + } +} + +func sourceViewGit(t *testing.T, root string, args ...string) string { + t.Helper() + command := exec.Command("git", args...) + command.Dir = root + output, err := command.CombinedOutput() + if err != nil { + t.Fatalf("git %v: %v\n%s", args, err, output) + } + return strings.TrimSpace(string(output)) +} From 298454f9c4b2fb22f9b29ae392992b747fecd53e Mon Sep 17 00:00:00 2001 From: Ashwin Gopalsamy <47941624+ashwingopalsamy@users.noreply.github.com> Date: Sun, 27 Sep 2026 12:33:14 +0530 Subject: [PATCH 16/20] feat(context): add focused retrieval and delivery evidence --- .agents/skills/agentic-go-context/SKILL.md | 24 ++- internal/intelligence/core.go | 5 +- internal/intelligence/focus.go | 132 +++++++++++- internal/intelligence/focus_test.go | 30 +++ internal/intelligence/snapshot.go | 15 ++ internal/intelligence/snapshot_test.go | 52 +++++ internal/tools/intelligence_tools.go | 2 +- internal/tools/mcp_surface_golden_test.go | 4 +- internal/tools/runtime.go | 2 +- validation/cmd/eval/main.go | 6 +- validation/internal/adoption/types.go | 115 ++++++----- validation/internal/adoption/types_test.go | 22 ++ validation/internal/pilot/runner.go | 221 ++++++++++++++++++++- validation/internal/pilot/runner_test.go | 125 ++++++++++++ validation/internal/pilot/types.go | 78 ++++---- 15 files changed, 727 insertions(+), 106 deletions(-) diff --git a/.agents/skills/agentic-go-context/SKILL.md b/.agents/skills/agentic-go-context/SKILL.md index 675b63a..30755a3 100644 --- a/.agents/skills/agentic-go-context/SKILL.md +++ b/.agents/skills/agentic-go-context/SKILL.md @@ -6,22 +6,26 @@ description: Use agentic-go context before unfamiliar or cross-package Go edits # Agentic Go Context For unfamiliar or cross-package Go edits, call `go_context` before editing and -refresh after edits. +refresh after edits. The workflow is snapshot-bound evidence, not a correctness +claim. -1. Start with `base` and one selector: `query`, `symbol_ref`, source +1. Start with `base` and exactly one selector: `query`, `symbol_ref`, source `file`+`line`+`column`, or `focus_file`/`focus_package`. Selectors are mutually exclusive. -2. If a result is ambiguous, narrow it using the returned current candidates - instead of repeating the same broad query. -3. Finish each batch of edits and formatting before refreshing. Refresh with +2. Symbol Refs are opaque byte strings. Copy them exactly. Never decode, edit, + shorten, reconstruct, or re-encode one. +3. File coordinates must point to the declaration identifier itself, never + whitespace, a line start, a comment, or a local variable. +4. If a result is ambiguous, copy one returned current candidate or issue a + fresh query/file selector. Do not repeat the same broad query. +5. A selector failure produces no evidence. Do not retry the same selector; + recover with fresh evidence. Stale rejection is expected and must not be + bypassed. +6. Finish each batch of edits and formatting before refreshing. Refresh with `base` and `previous_pack_id` only; omit old selectors and Symbol Refs. Refresh again after further edits or stale-snapshot rejection; avoid redundant refreshes while the snapshot is unchanged. -4. Never reuse a Symbol Ref returned before an edit. After a stale rejection, - refresh or select a current candidate; do not retry the stale selector. -5. Stale rejection is expected: stop and select current evidence; never bypass - it. -6. Use the result for impact and verification applicability. Treat impact and +7. Use the result for impact and verification applicability. Treat impact and reverse-dependency evidence as planning guidance, not authorization to edit affected packages; edit only the smallest task owner within supplied path/package scope. Skip trivial edits and preserve uncertainty when diff --git a/internal/intelligence/core.go b/internal/intelligence/core.go index c8042f1..4a654cd 100644 --- a/internal/intelligence/core.go +++ b/internal/intelligence/core.go @@ -15,6 +15,7 @@ import ( "github.com/agentic-mcps/go/internal/execution" "github.com/agentic-mcps/go/internal/gopls" + "github.com/agentic-mcps/go/internal/intelligence/retrieval" "github.com/agentic-mcps/go/internal/verification" "github.com/agentic-mcps/go/internal/workspace" ) @@ -50,6 +51,7 @@ type Core struct { contracts *ContractStore workspace *workspace.Workspace refactors *RefactorStore + retrieval *retrieval.Cache refactorWrite func(string, []byte, os.FileMode) error stateGate chan struct{} provenance []verification.ProvenanceReference @@ -133,7 +135,8 @@ func newCore( return &Core{ workspace: ws, runner: runner, snapshots: snapshots, semantic: semantic, mutator: mutator, artifacts: artifacts, contracts: contracts, refactors: refactors, verifications: verifications, - changes: changes, verifier: verify, refactorWrite: atomicReplace, stateGate: make(chan struct{}, 1), + retrieval: retrieval.NewCache(), changes: changes, verifier: verify, + refactorWrite: atomicReplace, stateGate: make(chan struct{}, 1), }, nil } diff --git a/internal/intelligence/focus.go b/internal/intelligence/focus.go index 0156ae6..b5129ce 100644 --- a/internal/intelligence/focus.go +++ b/internal/intelligence/focus.go @@ -12,6 +12,7 @@ import ( "sort" "strings" + "github.com/agentic-mcps/go/internal/intelligence/retrieval" "github.com/agentic-mcps/go/internal/verification" ) @@ -182,6 +183,15 @@ type storedFocusSymbol struct { File string `json:"file"` } +type focusRetrievalStatus struct { + used bool + complete bool + truncated bool + fallback string + indexedFiles int + skippedFiles int +} + //nolint:govet // Field order keeps persisted evidence readable. type deliveredEvidence struct { Kind string `json:"kind"` @@ -400,13 +410,39 @@ func (c *Core) focusContext(ctx context.Context, observation *snapshotObservatio result.Reasons = append(result.Reasons, "the requested query could not be resolved by the active semantic provider") return nil } - matches, searchErr := reader.Search(ctx, request.Query) + matches, retrievalStatus, searchErr := c.searchFocusCandidates(ctx, observation, reader, request.Query) if searchErr != nil { return searchErr } - normalized, normalizeErr := normalizeSymbolMatches(observation.snapshot, matches.Items) - if normalizeErr != nil { - return normalizeErr + if retrievalStatus.fallback != "" { + result.Uncertainties = append(result.Uncertainties, Uncertainty{ + Code: "retrieval.fallback", + Message: retrievalStatus.fallback, + Locations: []Location{}, + }) + } + if retrievalStatus.used && !retrievalStatus.complete { + result.Uncertainties = append(result.Uncertainties, Uncertainty{ + Code: "retrieval.incomplete", + Message: fmt.Sprintf("hybrid discovery indexed %d Go files; %d files were unavailable to the structural index", retrievalStatus.indexedFiles, retrievalStatus.skippedFiles), + Locations: []Location{}, + }) + } + if retrievalStatus.used && retrievalStatus.truncated { + result.Uncertainties = append(result.Uncertainties, Uncertainty{ + Code: "semantic.query_bounded", + Message: fmt.Sprintf("hybrid discovery returned more than %d candidates", maxFocusCandidates), + Locations: []Location{}, + }) + result.EvidenceStates = append(result.EvidenceStates, EvidenceState{Facet: "declaration_candidates", State: "gathered_but_omitted", Reason: "hybrid discovery candidate bound exceeded"}) + } + normalized := matches.Items + if !retrievalStatus.used { + var normalizeErr error + normalized, normalizeErr = normalizeSymbolMatches(observation.snapshot, matches.Items) + if normalizeErr != nil { + return normalizeErr + } } if len(normalized) > maxFocusCandidates { result.Candidates = append(result.Candidates, normalized[:maxFocusCandidates]...) @@ -551,6 +587,94 @@ func (c *Core) focusContext(ctx context.Context, observation *snapshotObservatio return &bounded, nil } +func (c *Core) searchFocusCandidates(ctx context.Context, observation *snapshotObservation, reader semanticReader, query string) (semanticSymbols, focusRetrievalStatus, error) { + status := focusRetrievalStatus{complete: true} + if c.retrieval == nil { + status.fallback = "hybrid discovery is unavailable; using semantic workspace-symbol search" + matches, err := reader.Search(ctx, query) + return matches, status, err + } + files, err := observedRetrievalFiles(observation) + if err != nil { + return semanticSymbols{}, status, err + } + indexed, err := c.retrieval.Search(ctx, retrieval.Key{ + Workspace: observation.snapshot.RepositoryID, + Scope: observation.snapshot.Scope, + Build: retrievalBuildKey(observation.snapshot), + Provider: observation.snapshot.GoplsVersion, + }, files, query, maxFocusCandidates) + if err != nil { + return semanticSymbols{}, status, err + } + status.complete = indexed.Complete + status.truncated = indexed.Truncated + status.indexedFiles = indexed.IndexedFiles + status.skippedFiles = indexed.SkippedFiles + if len(indexed.Candidates) == 0 { + status.fallback = "hybrid discovery returned no resolvable candidates; using semantic workspace-symbol search" + matches, searchErr := reader.Search(ctx, query) + return matches, status, searchErr + } + + matches := semanticSymbols{Items: []SymbolMatch{}} + seen := make(map[SymbolRef]struct{}, len(indexed.Candidates)) + for _, candidate := range indexed.Candidates { + contents, sourceErr := observation.source(candidate.Path) + if sourceErr != nil { + return semanticSymbols{}, status, sourceErr + } + position, positionErr := sourcePositionFromContents(filepath.Base(candidate.Path), contents, SourcePosition{ + File: candidate.Path, Line: candidate.Line, Column: candidate.Column, + }) + if positionErr != nil { + continue + } + match, symbolErr := reader.SymbolAt(ctx, candidate.Path, position) + if symbolErr != nil { + return semanticSymbols{}, status, symbolErr + } + normalized, normalizeErr := normalizeSymbolMatch(observation.snapshot, match) + if normalizeErr != nil { + continue + } + if _, exists := seen[normalized.Ref]; exists { + continue + } + seen[normalized.Ref] = struct{}{} + matches.Items = append(matches.Items, normalized) + if len(matches.Items) == maxFocusCandidates { + break + } + } + if len(matches.Items) == 0 { + status.fallback = "hybrid discovery candidates did not resolve semantically; using semantic workspace-symbol search" + fallback, searchErr := reader.Search(ctx, query) + return fallback, status, searchErr + } + status.used = true + return matches, status, nil +} + +func observedRetrievalFiles(observation *snapshotObservation) ([]retrieval.File, error) { + files := make([]retrieval.File, 0) + for _, record := range observation.records { + if record.Kind == "deleted" || !strings.HasSuffix(record.Path, ".go") { + continue + } + contents, err := observation.source(record.Path) + if err != nil { + return nil, err + } + files = append(files, retrieval.File{Path: record.Path, Digest: record.Digest, Contents: contents}) + } + return files, nil +} + +func retrievalBuildKey(snapshot SnapshotRef) string { + return fmt.Sprintf("%s\x00%s\x00%t\x00%s\x00%s\x00%s\x00%v", snapshot.Build.GOOS, snapshot.Build.GOARCH, snapshot.Build.CGOEnabled, snapshot.Build.GOFLAGS, snapshot.Build.Workspace, snapshot.GoVersion, snapshot.Build.Tags) +} + func (c *Core) focusSymbol(ctx context.Context, reader semanticReader, observation *snapshotObservation, file string, position Position, maximum int) (*SymbolContext, []SymbolMatch, []SourceExcerpt, []Uncertainty, error) { if !observation.snapshot.Capabilities.DocumentSymbol { return nil, []SymbolMatch{}, []SourceExcerpt{}, []Uncertainty{{Code: "semantic.unsupported_capability", Message: "the active semantic provider does not support document symbols", Locations: []Location{}}}, nil diff --git a/internal/intelligence/focus_test.go b/internal/intelligence/focus_test.go index 33dd43b..b28c1b0 100644 --- a/internal/intelligence/focus_test.go +++ b/internal/intelligence/focus_test.go @@ -298,6 +298,36 @@ func TestFocusContextReturnsAmbiguousCandidatesBeforeExpansion(t *testing.T) { if result.Symbol != nil || len(result.Candidates) != 2 || len(result.Reasons) == 0 || !hasEvidenceState(result.EvidenceStates, "relationships", "unexamined") { t.Fatalf("focusContext(ambiguous query) = %#v", result) } + if !containsUncertainty(result.Uncertainties, "retrieval.fallback") { + t.Fatalf("focusContext(ambiguous query) did not record retrieval fallback: %#v", result.Uncertainties) + } +} + +func TestFocusContextUsesHybridDiscoveryBeforeSemanticSearch(t *testing.T) { + root := snapshotRepository(t) + writeSnapshotFile(t, root, "main.go", "package fixture\n\n// ProcessPayment validates and processes an incoming payment request.\nfunc ProcessPayment() {}\nfunc Unrelated() {}\n") + snapshotter := newTestSnapshotter(t, root) + reader := &fakeSemanticReader{symbol: SymbolMatch{ + Name: "ProcessPayment", Qualified: "fixture.ProcessPayment", Kind: "go.function", Package: "fixture", + Location: Location{File: "main.go", Line: 4, Column: 6, EndLine: 4, EndColumn: 19}, + }} + core := newTestCore(t, snapshotter, reader) + observation, err := core.observe(context.Background(), "HEAD", "./...", "") + if err != nil { + t.Fatal(err) + } + defer observation.release() + + result, err := core.focusContext(context.Background(), &observation, FocusRequest{Query: "process incoming payment", MaxBytes: DefaultBriefBytes}) + if err != nil { + t.Fatal(err) + } + if result.Symbol == nil || result.Symbol.Symbol.Name != "ProcessPayment" { + t.Fatalf("focusContext() symbol = %#v, want ProcessPayment", result.Symbol) + } + if containsUncertainty(result.Uncertainties, "retrieval.fallback") { + t.Fatalf("focusContext() unexpectedly fell back: %#v", result.Uncertainties) + } } func TestFocusContextPreservesUncertaintyForOmittedWorkspaceSymbols(t *testing.T) { diff --git a/internal/intelligence/snapshot.go b/internal/intelligence/snapshot.go index a475466..5b53fc6 100644 --- a/internal/intelligence/snapshot.go +++ b/internal/intelligence/snapshot.go @@ -16,6 +16,7 @@ import ( "sync" "github.com/agentic-mcps/go/internal/execution" + "github.com/agentic-mcps/go/internal/sourceview" "github.com/agentic-mcps/go/internal/workspace" ) @@ -421,6 +422,20 @@ func (s *Snapshotter) readState(ctx context.Context, request SnapshotRequest) (s if err != nil { return snapshotState{}, fmt.Errorf("resolving HEAD: %w", err) } + if filepath.Clean(repositoryRoot) == filepath.Clean(s.workspace.Root()) { + gitEntry, statErr := os.Lstat(filepath.Join(repositoryRoot, ".git")) + if statErr != nil && !errors.Is(statErr, os.ErrNotExist) { + return snapshotState{}, fmt.Errorf("inspecting worktree Git marker: %w", statErr) + } + if statErr == nil && !gitEntry.IsDir() { + if viewErr := sourceview.Validate(ctx, s.runner, head); viewErr != nil { + if errors.Is(viewErr, sourceview.ErrStale) { + return snapshotState{}, fmt.Errorf("%w: %v", ErrSnapshotChanged, viewErr) + } + return snapshotState{}, fmt.Errorf("validating branch source view: %w", viewErr) + } + } + } base, mergeBase := "", "" if request.Base != "" { if strings.HasPrefix(request.Base, "-") || strings.ContainsRune(request.Base, 0) { diff --git a/internal/intelligence/snapshot_test.go b/internal/intelligence/snapshot_test.go index 54f8a64..229b504 100644 --- a/internal/intelligence/snapshot_test.go +++ b/internal/intelligence/snapshot_test.go @@ -12,6 +12,7 @@ import ( "testing" "github.com/agentic-mcps/go/internal/execution" + "github.com/agentic-mcps/go/internal/sourceview" "github.com/agentic-mcps/go/internal/workspace" ) @@ -108,6 +109,57 @@ func TestSnapshotValidationRejectsStaleReference(t *testing.T) { } } +func TestSnapshotRejectsMovedManagedBranchViewRef(t *testing.T) { + root := snapshotRepository(t) + snapshotGit(t, root, "checkout", "-b", "feature") + writeSnapshotFile(t, root, "main.go", "package fixture\n\nvar Value = 2\n") + snapshotGit(t, root, "add", "main.go") + snapshotGit(t, root, "-c", "commit.gpgsign=false", "commit", "-m", "feature value") + featureCommit := snapshotGit(t, root, "rev-parse", "HEAD") + snapshotGit(t, root, "checkout", "main") + + sourceWorkspace, err := workspace.Open(context.Background(), root) + if err != nil { + t.Fatal(err) + } + sourceRunner, err := execution.New(sourceWorkspace, execution.Config{}) + if err != nil { + t.Fatal(err) + } + viewPath := filepath.Join(filepath.Dir(root), filepath.Base(root)+"-feature-view") + view, err := sourceview.Create(context.Background(), sourceRunner, sourceWorkspace, sourceview.Request{ + Branch: "feature", OutputPath: viewPath, + }) + if err != nil { + t.Fatal(err) + } + if view.Commit != featureCommit { + t.Fatalf("source view commit = %s, want %s", view.Commit, featureCommit) + } + + viewWorkspace, err := workspace.Open(context.Background(), viewPath) + if err != nil { + t.Fatal(err) + } + viewRunner, err := sourceRunner.ForWorkspace(viewWorkspace) + if err != nil { + t.Fatal(err) + } + snapshotter, err := NewSnapshotter(viewWorkspace, viewRunner) + if err != nil { + t.Fatal(err) + } + request := SnapshotRequest{Semantic: SemanticIdentity{Version: "test"}} + if _, err := snapshotter.Capture(context.Background(), request); err != nil { + t.Fatalf("Capture() for current branch ref = %v", err) + } + mainCommit := snapshotGit(t, root, "rev-parse", "refs/heads/main") + snapshotGit(t, root, "update-ref", "refs/heads/feature", mainCommit) + if _, err := snapshotter.Capture(context.Background(), request); !errors.Is(err, ErrSnapshotChanged) { + t.Fatalf("Capture() after branch ref movement = %v, want ErrSnapshotChanged", err) + } +} + func TestObservationSourceVerifiesManifestOnCapturedSourceMiss(t *testing.T) { root := snapshotRepository(t) snapshotter := newTestSnapshotter(t, root) diff --git a/internal/tools/intelligence_tools.go b/internal/tools/intelligence_tools.go index 332a6d2..9aa1d8f 100644 --- a/internal/tools/intelligence_tools.go +++ b/internal/tools/intelligence_tools.go @@ -85,7 +85,7 @@ func RegisterSymbolContext(server *mcp.Server, runtime *Runtime) { // RegisterContext adds the post-v1 focus tool after the frozen registry. func RegisterContext(server *mcp.Server, runtime *Runtime) { - mcp.AddTool(server, &mcp.Tool{Name: "go_context", Description: "Before editing unfamiliar or cross-package Go code, call with base and one selector group (query, symbol_ref, file+line+column, or focus_file/focus_package) to map impact and verification applicability. After editing, refresh with base and previous_pack_id only; never reuse a Symbol Ref from before an edit; stale selectors and refs are expected to be rejected, so select current evidence.", Annotations: intelligenceAnnotations()}, runtime.context) + mcp.AddTool(server, &mcp.Tool{Name: "go_context", Description: "Before editing unfamiliar or cross-package Go code, call go_context with base and exactly one selector group: query, symbol_ref, file+line+column, or focus_file/focus_package. Symbol Refs are opaque byte strings: copy them exactly; never decode, edit, shorten, reconstruct, or re-encode them. File coordinates must point to the declaration identifier itself, never whitespace, line starts, comments, or local variables. After ambiguity, copy one current candidate or issue a fresh query/file selector. A selector failure produces no evidence: do not retry the same selector; recover with fresh evidence. After editing, refresh with base and previous_pack_id only; stale selectors and refs require current evidence. Use results for impact and verification applicability, not correctness claims.", Annotations: intelligenceAnnotations()}, runtime.context) } func (r *Runtime) requireIntelligence() (IntelligenceService, error) { diff --git a/internal/tools/mcp_surface_golden_test.go b/internal/tools/mcp_surface_golden_test.go index aa5bdca..2f666db 100644 --- a/internal/tools/mcp_surface_golden_test.go +++ b/internal/tools/mcp_surface_golden_test.go @@ -114,7 +114,7 @@ func TestPostV1FocusToolIsAdditiveAndDiscoverable(t *testing.T) { if tool.Name != "go_context" { continue } - for _, phrase := range []string{"one selector", "previous_pack_id only", "stale selectors", "verification applicability"} { + for _, phrase := range []string{"one selector", "previous_pack_id only", "stale selectors", "verification applicability", "opaque byte strings", "declaration identifier", "no evidence"} { if !strings.Contains(tool.Description, phrase) { t.Fatalf("go_context description %q missing %q", tool.Description, phrase) } @@ -208,7 +208,7 @@ func TestProductionServerInitializeInstructionsAndUniqueFocusRegistration(t *tes if len(ServerInstructions) > 512 { t.Fatalf("server instructions length = %d, want <= 512", len(ServerInstructions)) } - for _, phrase := range []string{"unfamiliar", "cross-package", "go_context", "before editing", "one selector", "previous_pack_id only", "stale selectors", "impact", "verification applicability", "planning guidance", "smallest task owner", "path/package scope", "trivial edits"} { + for _, phrase := range []string{"unfamiliar", "cross-package", "go_context", "before editing", "one selector", "previous_pack_id only", "stale selectors", "opaque", "declaration", "selector failure", "recover fresh"} { if !strings.Contains(ServerInstructions, phrase) { t.Errorf("server instructions %q missing %q", ServerInstructions, phrase) } diff --git a/internal/tools/runtime.go b/internal/tools/runtime.go index da7b976..5ee1a43 100644 --- a/internal/tools/runtime.go +++ b/internal/tools/runtime.go @@ -28,7 +28,7 @@ type Runtime struct { // ServerInstructions is the concise decision rule surfaced during MCP // initialize so clients can discover the focused context workflow. -const ServerInstructions = "For unfamiliar or cross-package Go changes, call go_context before editing with base and one selector (query, symbol_ref, file+line+column, or focus_file/focus_package). After editing, refresh with base and previous_pack_id only; stale selectors and refs mean select current evidence, and never reuse a Symbol Ref after an edit. Use it for impact and verification applicability; treat dependency evidence as planning guidance, edit only the smallest task owner within path/package scope, and skip trivial edits." +const ServerInstructions = "For unfamiliar/cross-package Go, before editing call go_context with base+one selector (query/symbol_ref/file+line/column/focus). opaque Symbol Refs: copy exact; never decode/edit/shorten/reconstruct/re-encode. Coordinates hit declaration identifiers, not whitespace/line-start/comments/locals. Ambiguity: current candidate or fresh query/file. selector failure: no evidence; no retry; recover fresh. After edit refresh with base+previous_pack_id only; stale selectors need current evidence." // NewProductionServer creates the configured MCP server used by the binary. func NewProductionServer(implementation *mcp.Implementation) *mcp.Server { diff --git a/validation/cmd/eval/main.go b/validation/cmd/eval/main.go index 62712f6..15e112c 100644 --- a/validation/cmd/eval/main.go +++ b/validation/cmd/eval/main.go @@ -126,6 +126,8 @@ func run(ctx context.Context, args []string) error { r.FocusToolCalls = fm.Calls r.FocusFailedCalls = fm.FailedCalls r.FocusErrorCategories = fm.ErrorCategories + r.FocusFailureCauses = fm.FailureCauses + r.FocusFailureRecords = fm.FailureRecords r.FirstFocusCallPosition = fm.FirstPosition r.RefreshUse = fm.Refresh r.FocusEvidenceUse = fm.Evidence @@ -503,7 +505,7 @@ func run(ctx context.Context, args []string) error { result.Uncertainties = append(result.Uncertainties, "focus capability delivered but agent made zero go_context calls") } } - r := pilot.Run{SchemaVersion: pilot.Schema, ScenarioID: scenario.ID, TaskID: task.ID, Condition: *condition, Repetition: *repetition, Model: "gpt-5.6-luna", Reasoning: "max", Prompt: scenario.Prompt, Transcript: pilot.SanitizeTranscript(events.Raw), ToolCalls: events.ToolCalls, FocusToolCalls: focusCalls, FocusFailedCalls: fm.FailedCalls, FocusErrorCategories: fm.ErrorCategories, FirstFocusCallPosition: fm.FirstPosition, RefreshUse: fm.Refresh, FocusEvidenceUse: fm.Evidence, FocusResultFollowedByEdit: fm.FocusResultFollowedByEdit, RefreshCompleted: fm.RefreshCompleted, FocusDelivery: delivery, DurationMS: duration, EvidenceBytes: int64(len(data)), Acceptance: result.Status, Qualifying: result.Status == "pass", ScopeViolations: result.UnexpectedPaths, Uncertainty: result.Uncertainties} + r := pilot.Run{SchemaVersion: pilot.Schema, ScenarioID: scenario.ID, TaskID: task.ID, Condition: *condition, Repetition: *repetition, Model: "gpt-5.6-luna", Reasoning: "max", Prompt: scenario.Prompt, Transcript: pilot.SanitizeTranscript(events.Raw), ToolCalls: events.ToolCalls, FocusToolCalls: focusCalls, FocusFailedCalls: fm.FailedCalls, FocusErrorCategories: fm.ErrorCategories, FocusFailureCauses: fm.FailureCauses, FocusFailureRecords: fm.FailureRecords, FirstFocusCallPosition: fm.FirstPosition, RefreshUse: fm.Refresh, FocusEvidenceUse: fm.Evidence, FocusResultFollowedByEdit: fm.FocusResultFollowedByEdit, RefreshCompleted: fm.RefreshCompleted, RedundantRefreshes: fm.RedundantRefreshes, FocusDelivery: delivery, DurationMS: duration, EvidenceBytes: int64(len(data)), Acceptance: result.Status, Qualifying: result.Status == "pass", ScopeViolations: result.UnexpectedPaths, Uncertainty: result.Uncertainties} data, _ = json.Marshal(r.Transcript) r.SourceSHA256 = sourceHash r.BinarySHA256 = binaryHash @@ -667,7 +669,7 @@ func runAdoption(ctx context.Context, s adoption.Scenario, arm string, repetitio if focus { mcpInstructionsDigest = adoption.DigestString(tools.ServerInstructions) } - r := adoption.Run{SchemaVersion: adoption.Schema, ScenarioID: s.ID, TaskID: task.ID, Arm: arm, Repetition: repetition, Model: "gpt-5.6-luna", Reasoning: "max", Prompt: p, PromptSHA256: adoption.DigestString(p), MCPDescriptionSHA256: mcpInstructionsDigest, SkillSHA256: skill.Digest, EffectiveInstructionSurface: adoptionInstructionSurface(arm), SourceSHA256: sh, BinarySHA256: bh, InitialWorkspaceSHA256: initial, PostWorkspaceSHA256: post, WorkspaceSHA256: post, Transcript: tr, TranscriptSHA256: adoption.DigestString(string(raw)), Patch: patch, PatchSHA256: adoption.DigestString(patch), AcceptanceEvidenceSHA256: adoption.DigestString(string(acc)), Acceptance: result.Status, ScopeViolations: result.UnexpectedPaths, EvidenceBytes: int64(len(raw)), ToolCalls: events.ToolCalls, FocusToolCalls: focusCalls, FocusFailedCalls: fm.FailedCalls, FocusErrorCategories: fm.ErrorCategories, FirstFocusCallPosition: fm.FirstPosition, RefreshUse: fm.Refresh, FocusEvidenceUse: fm.Evidence, FocusResultFollowedByEdit: fm.FocusResultFollowedByEdit, RefreshCompleted: fm.RefreshCompleted, SkillDiscovered: skillDiscovered, DurationMS: dur, OperatorIntervention: false, Uncertainty: unc, Qualifying: result.Status == "pass", FocusDelivery: map[bool]string{true: "healthy", false: ""}[focus]} + r := adoption.Run{SchemaVersion: adoption.Schema, ScenarioID: s.ID, TaskID: task.ID, Arm: arm, Repetition: repetition, Model: "gpt-5.6-luna", Reasoning: "max", Prompt: p, PromptSHA256: adoption.DigestString(p), MCPDescriptionSHA256: mcpInstructionsDigest, SkillSHA256: skill.Digest, EffectiveInstructionSurface: adoptionInstructionSurface(arm), SourceSHA256: sh, BinarySHA256: bh, InitialWorkspaceSHA256: initial, PostWorkspaceSHA256: post, WorkspaceSHA256: post, Transcript: tr, TranscriptSHA256: adoption.DigestString(string(raw)), Patch: patch, PatchSHA256: adoption.DigestString(patch), AcceptanceEvidenceSHA256: adoption.DigestString(string(acc)), Acceptance: result.Status, ScopeViolations: result.UnexpectedPaths, EvidenceBytes: int64(len(raw)), ToolCalls: events.ToolCalls, FocusToolCalls: focusCalls, FocusFailedCalls: fm.FailedCalls, FocusErrorCategories: fm.ErrorCategories, FocusFailureCauses: fm.FailureCauses, FocusFailureRecords: fm.FailureRecords, FirstFocusCallPosition: fm.FirstPosition, RefreshUse: fm.Refresh, FocusEvidenceUse: fm.Evidence, FocusResultFollowedByEdit: fm.FocusResultFollowedByEdit, RefreshCompleted: fm.RefreshCompleted, RedundantRefreshes: fm.RedundantRefreshes, SkillDiscovered: skillDiscovered, DurationMS: dur, OperatorIntervention: false, Uncertainty: unc, Qualifying: result.Status == "pass", FocusDelivery: map[bool]string{true: "healthy", false: ""}[focus]} r.Stderr = pilot.SanitizeText(stderr) if q, e := validation.Qualify(ctx, task, bundle, filepath.Join(sourceRoot, task.Repository.Name)); e != nil { r.Qualifying = false diff --git a/validation/internal/adoption/types.go b/validation/internal/adoption/types.go index e4641aa..34cd1ae 100644 --- a/validation/internal/adoption/types.go +++ b/validation/internal/adoption/types.go @@ -32,7 +32,7 @@ const ( ) // Guidance is the generic treatment instruction. -const Guidance = "For an unfamiliar Go change, call go_context with base and one selector: query, symbol_ref, or file with line and column. After editing, refresh with base and previous_pack_id only; never reuse a Symbol Ref from before an edit. Stale selectors and refs are expected to be rejected, so select current evidence." +const Guidance = "For an unfamiliar Go change, call go_context with base and exactly one selector: query, symbol_ref, file with line and column, or focus_file/focus_package. Symbol Refs are opaque byte strings: copy them exactly; never decode, edit, shorten, reconstruct, or re-encode them. File coordinates must point to the declaration identifier itself, never whitespace, line starts, comments, or local variables. After ambiguity, copy one current candidate or issue a fresh query/file selector. A selector failure produces no evidence; do not retry the same selector; recover with fresh evidence. After edits, refresh with base and previous_pack_id only; stale selectors and refs require current evidence." // Scenario describes an adoption task. type Scenario struct { @@ -47,49 +47,52 @@ type Scenario struct { // //nolint:govet // JSON contract groups related evidence fields. type Run struct { - SchemaVersion string `json:"schema_version"` - ScenarioID string `json:"scenario_id"` - TaskID string `json:"task_id"` - Arm string `json:"arm"` - Repetition int `json:"repetition"` - Model string `json:"model"` - Reasoning string `json:"reasoning"` - Prompt string `json:"prompt"` - PromptSHA256 string `json:"prompt_sha256"` - MCPDescriptionSHA256 string `json:"mcp_description_sha256"` - SkillSHA256 string `json:"skill_sha256,omitempty"` - EffectiveInstructionSurface string `json:"effective_instruction_surface,omitempty"` - SourceSHA256 string `json:"source_sha256"` - BinarySHA256 string `json:"binary_sha256"` - InitialWorkspaceSHA256 string `json:"initial_workspace_sha256"` - PostWorkspaceSHA256 string `json:"post_workspace_sha256"` - Transcript []json.RawMessage `json:"transcript"` - TranscriptSHA256 string `json:"transcript_sha256"` - Patch string `json:"patch"` - PatchSHA256 string `json:"patch_sha256"` - Stderr string `json:"stderr,omitempty"` - ProcessError string `json:"process_error,omitempty"` - FocusDelivery string `json:"focus_delivery"` - WorkspaceSHA256 string `json:"workspace_sha256"` - DecisionObligations []string `json:"decision_obligations"` - AcceptanceEvidenceSHA256 string `json:"acceptance_evidence_sha256"` - Acceptance string `json:"acceptance"` - ScopeViolations []string `json:"scope_violations"` - EvidenceBytes int64 `json:"evidence_bytes"` - ToolCalls int `json:"tool_calls"` - FocusToolCalls int `json:"focus_tool_calls"` - FocusFailedCalls int `json:"focus_failed_calls"` - FocusErrorCategories map[string]int `json:"focus_error_categories"` - FirstFocusCallPosition int `json:"first_focus_call_position"` - RefreshUse bool `json:"refresh_use"` - FocusEvidenceUse bool `json:"focus_evidence_use"` - FocusResultFollowedByEdit bool `json:"focus_result_followed_by_edit"` - RefreshCompleted bool `json:"refresh_completed"` - SkillDiscovered bool `json:"skill_discovered"` - DurationMS int64 `json:"duration_ms"` - OperatorIntervention bool `json:"operator_intervention"` - Uncertainty []string `json:"uncertainty"` - Qualifying bool `json:"qualifying"` + SchemaVersion string `json:"schema_version"` + ScenarioID string `json:"scenario_id"` + TaskID string `json:"task_id"` + Arm string `json:"arm"` + Repetition int `json:"repetition"` + Model string `json:"model"` + Reasoning string `json:"reasoning"` + Prompt string `json:"prompt"` + PromptSHA256 string `json:"prompt_sha256"` + MCPDescriptionSHA256 string `json:"mcp_description_sha256"` + SkillSHA256 string `json:"skill_sha256,omitempty"` + EffectiveInstructionSurface string `json:"effective_instruction_surface,omitempty"` + SourceSHA256 string `json:"source_sha256"` + BinarySHA256 string `json:"binary_sha256"` + InitialWorkspaceSHA256 string `json:"initial_workspace_sha256"` + PostWorkspaceSHA256 string `json:"post_workspace_sha256"` + Transcript []json.RawMessage `json:"transcript"` + TranscriptSHA256 string `json:"transcript_sha256"` + Patch string `json:"patch"` + PatchSHA256 string `json:"patch_sha256"` + Stderr string `json:"stderr,omitempty"` + ProcessError string `json:"process_error,omitempty"` + FocusDelivery string `json:"focus_delivery"` + WorkspaceSHA256 string `json:"workspace_sha256"` + DecisionObligations []string `json:"decision_obligations"` + AcceptanceEvidenceSHA256 string `json:"acceptance_evidence_sha256"` + Acceptance string `json:"acceptance"` + ScopeViolations []string `json:"scope_violations"` + EvidenceBytes int64 `json:"evidence_bytes"` + ToolCalls int `json:"tool_calls"` + FocusToolCalls int `json:"focus_tool_calls"` + FocusFailedCalls int `json:"focus_failed_calls"` + FocusErrorCategories map[string]int `json:"focus_error_categories"` + FocusFailureCauses map[string]int `json:"focus_failure_causes,omitempty"` + FocusFailureRecords []pilot.FocusFailureRecord `json:"focus_failure_records,omitempty"` + FirstFocusCallPosition int `json:"first_focus_call_position"` + RefreshUse bool `json:"refresh_use"` + FocusEvidenceUse bool `json:"focus_evidence_use"` + FocusResultFollowedByEdit bool `json:"focus_result_followed_by_edit"` + RefreshCompleted bool `json:"refresh_completed"` + RedundantRefreshes int `json:"redundant_refreshes,omitempty"` + SkillDiscovered bool `json:"skill_discovered"` + DurationMS int64 `json:"duration_ms"` + OperatorIntervention bool `json:"operator_intervention"` + Uncertainty []string `json:"uncertainty"` + Qualifying bool `json:"qualifying"` } // LoadRun loads and canonicalizes one adoption run. @@ -127,11 +130,14 @@ func LoadRun(path string) (Run, error) { r.FocusToolCalls = focus.Calls r.FocusFailedCalls = focus.FailedCalls r.FocusErrorCategories = focus.ErrorCategories + r.FocusFailureCauses = focus.FailureCauses + r.FocusFailureRecords = focus.FailureRecords r.FirstFocusCallPosition = focus.FirstPosition r.RefreshUse = focus.Refresh r.FocusEvidenceUse = focus.Evidence r.FocusResultFollowedByEdit = focus.FocusResultFollowedByEdit r.RefreshCompleted = focus.RefreshCompleted + r.RedundantRefreshes = focus.RedundantRefreshes r.SkillDiscovered = pilot.SkillDiscoveryEvidence(pilot.Events{Raw: r.Transcript}, SkillName) if r.PatchSHA256 != "" && r.PatchSHA256 != DigestString(r.Patch) { return r, fmt.Errorf("patch hash mismatch") @@ -180,11 +186,14 @@ func LoadRuns(path string) ([]Run, error) { r[i].FocusToolCalls = focus.Calls r[i].FocusFailedCalls = focus.FailedCalls r[i].FocusErrorCategories = focus.ErrorCategories + r[i].FocusFailureCauses = focus.FailureCauses + r[i].FocusFailureRecords = focus.FailureRecords r[i].FirstFocusCallPosition = focus.FirstPosition r[i].RefreshUse = focus.Refresh r[i].FocusEvidenceUse = focus.Evidence r[i].FocusResultFollowedByEdit = focus.FocusResultFollowedByEdit r[i].RefreshCompleted = focus.RefreshCompleted + r[i].RedundantRefreshes = focus.RedundantRefreshes r[i].SkillDiscovered = pilot.SkillDiscoveryEvidence(pilot.Events{Raw: r[i].Transcript}, SkillName) } return r, nil @@ -195,9 +204,11 @@ func Aggregate(rs []Run) map[string]any { out := map[string]any{"schema_version": Schema, "runs": len(rs), "arms": map[string]any{}} arms := out["arms"].(map[string]any) for _, a := range []string{ArmBaseline, ArmDiscoverability, ArmGuidance, ArmIntegrated} { - var n, q, c, failed, followedByEdit, refreshCompleted int + var n, q, c, failed, followedByEdit, refreshCompleted, redundantRefreshes int var bytes, calls, durations []int64 categories := map[string]int{} + failureCauses := map[string]int{} + var recovered, repeated int for _, r := range rs { if r.Arm != a { continue @@ -214,14 +225,26 @@ func Aggregate(rs []Run) map[string]any { if r.RefreshCompleted { refreshCompleted++ } + redundantRefreshes += r.RedundantRefreshes for category, count := range r.FocusErrorCategories { categories[category] += count } + for cause, count := range r.FocusFailureCauses { + failureCauses[cause] += count + } + for _, record := range r.FocusFailureRecords { + if record.Recovered { + recovered++ + } + if record.Repeated && record.Cause == pilot.FocusFailureSelectorMisuse { + repeated++ + } + } bytes = append(bytes, r.EvidenceBytes) calls = append(calls, int64(r.ToolCalls)) durations = append(durations, r.DurationMS) } - arms[a] = map[string]any{"runs": n, "qualifying": q, "focus_evidence_use": c, "focus_failed_calls": failed, "focus_result_followed_by_edit": followedByEdit, "refresh_completed": refreshCompleted, "focus_error_categories": categories, "evidence_bytes_median": median(bytes), "tool_calls_median": median(calls), "duration_ms_median": median(durations)} + arms[a] = map[string]any{"runs": n, "qualifying": q, "focus_evidence_use": c, "focus_failed_calls": failed, "focus_result_followed_by_edit": followedByEdit, "refresh_completed": refreshCompleted, "redundant_refreshes": redundantRefreshes, "focus_error_categories": categories, "focus_failure_causes": failureCauses, "focus_recovered_failures": recovered, "focus_repeated_rejected_selectors": repeated, "evidence_bytes_median": median(bytes), "tool_calls_median": median(calls), "duration_ms_median": median(durations)} } return out } diff --git a/validation/internal/adoption/types_test.go b/validation/internal/adoption/types_test.go index 17c0603..c48a887 100644 --- a/validation/internal/adoption/types_test.go +++ b/validation/internal/adoption/types_test.go @@ -52,6 +52,28 @@ func TestGuidanceIsStableAndDigestable(t *testing.T) { if Guidance == "" || DigestString(Guidance) == "" { t.Fatal("guidance digest missing") } + guidance := strings.ToLower(Guidance) + for _, phrase := range []string{ + "opaque byte strings", + "copy them exactly", + "never decode", + "shorten", + "reconstruct", + "re-encode", + "declaration identifier", + "line starts", + "local variables", + "current candidate", + "fresh query/file selector", + "no evidence", + "do not retry the same selector", + "fresh evidence", + "previous_pack_id only", + } { + if !strings.Contains(guidance, phrase) { + t.Errorf("guidance %q missing %q", Guidance, phrase) + } + } } func TestLoadRunRecomputesStaleFocusMetrics(t *testing.T) { diff --git a/validation/internal/pilot/runner.go b/validation/internal/pilot/runner.go index 0e50be8..b3de555 100644 --- a/validation/internal/pilot/runner.go +++ b/validation/internal/pilot/runner.go @@ -181,12 +181,35 @@ func FocusToolCalls(events Events) int { // consumers and are derived without trusting agent prose. type FocusMetric struct { ErrorCategories map[string]int + FailureCauses map[string]int + FailureRecords []FocusFailureRecord Calls, FailedCalls, FirstPosition int Refresh, Evidence bool FocusResultFollowedByEdit bool RefreshCompleted bool + RedundantRefreshes int } +// FocusFailureRecord is a bounded, sanitized attribution for one failed +// go_context call. It deliberately excludes selector values, paths, prompts, +// transcripts, and provider error text. +type FocusFailureRecord struct { + Cause string `json:"cause"` + Selector string `json:"selector"` + Reason string `json:"reason"` + Recovered bool `json:"recovered"` + Repeated bool `json:"repeated"` +} + +const ( + FocusFailureSelectorMisuse = "selector_misuse" + FocusFailureProvider = "provider_failure" + FocusFailureStaleSnapshot = "stale_snapshot" + FocusFailureTimeout = "timeout" + FocusFailureTransport = "transport" + FocusFailureUnknown = "unknown" +) + // SkillDiscoveryEvidence reports whether the transcript contains a concrete // read or load of the named skill's SKILL.md. Generic prose or a tool call is // not sufficient evidence of model discovery. @@ -219,7 +242,11 @@ func SkillDiscoveryEvidence(events Events, skillName string) bool { // counts only completed MCP events and recognizes edits through the explicit // file_change event, not through message wording. func FocusMetrics(events Events) FocusMetric { - m := FocusMetric{ErrorCategories: map[string]int{}} + m := FocusMetric{ + ErrorCategories: map[string]int{}, + FailureCauses: map[string]int{}, + FailureRecords: []FocusFailureRecord{}, + } calls := make([]focusCall, 0) editPositions := make([]int, 0) for i, raw := range events.Raw { @@ -252,6 +279,11 @@ func FocusMetrics(events Events) FocusMetric { m.RefreshCompleted = true } } + m.FailureRecords = focusFailureRecords(calls) + m.RedundantRefreshes = redundantRefreshCount(calls, editPositions) + for _, record := range m.FailureRecords { + m.FailureCauses[record.Cause]++ + } m.Evidence = m.FocusResultFollowedByEdit || m.RefreshCompleted return m } @@ -260,6 +292,13 @@ type focusCall struct { PreviousPackID string PackID string ErrorCategory string + FailureCause string + FailureReason string + SelectorKind string + SelectorDigest string + SnapshotID string + ScopeDigest string + EvidenceDigest string Position int Refresh bool Success bool @@ -276,15 +315,20 @@ func parseCompletedFocusCall(raw json.RawMessage, position int) (focusCall, bool } arguments, _ := item["arguments"].(map[string]any) previousPackID := stringValue(arguments["previous_pack_id"]) + selectorKind, selectorDigest := focusSelector(arguments) result, hasResult := item["result"] status := stringValue(item["status"]) success := status != "failed" && hasResult && result != nil - call := focusCall{Position: position, PreviousPackID: previousPackID, Refresh: previousPackID != "", Success: success} + call := focusCall{Position: position, PreviousPackID: previousPackID, Refresh: previousPackID != "", SelectorKind: selectorKind, SelectorDigest: selectorDigest, Success: success} if !success { - call.ErrorCategory = focusErrorCategory(focusErrorText(item)) + errorText := focusErrorText(item) + call.ErrorCategory = focusErrorCategory(errorText) + call.FailureCause, call.FailureReason = focusFailureClassification(errorText, call.ErrorCategory, selectorKind) return call, true } call.PackID = focusPackID(result) + call.SnapshotID, call.ScopeDigest = focusSnapshot(result) + call.EvidenceDigest = focusEvidenceDigest(arguments) return call, true } @@ -324,6 +368,42 @@ func refreshFollowsEdit(calls []focusCall, refresh focusCall, editPositions []in return false } +// redundantRefreshCount flags a successful refresh repeated against the same +// observation, scope, and evidence requirements without an intervening edit +// or stale rejection. It is diagnostic only and never suppresses a provider +// call or relaxes freshness validation. +func redundantRefreshCount(calls []focusCall, editPositions []int) int { + var previous *focusCall + count := 0 + for i := range calls { + call := &calls[i] + if !call.Success { + if call.FailureCause == FocusFailureStaleSnapshot { + previous = nil + } + continue + } + if !call.Refresh { + previous = nil + continue + } + if previous != nil && previous.SnapshotID != "" && previous.ScopeDigest != "" && previous.EvidenceDigest != "" && call.SnapshotID == previous.SnapshotID && call.ScopeDigest == previous.ScopeDigest && call.EvidenceDigest == previous.EvidenceDigest && !hasEditBetween(editPositions, previous.Position, call.Position) { + count++ + } + previous = call + } + return count +} + +func hasEditBetween(positions []int, start, end int) bool { + for _, position := range positions { + if position > start && position < end { + return true + } + } + return false +} + func stringValue(value any) string { text, _ := value.(string) return text @@ -346,6 +426,71 @@ func focusPackID(result any) string { return "" } +func focusSnapshot(result any) (string, string) { + resultMap, ok := result.(map[string]any) + if !ok { + return "", "" + } + structured, ok := resultMap["structured_content"].(map[string]any) + if !ok { + structured, ok = resultMap["structuredContent"].(map[string]any) + } + if !ok { + return "", "" + } + snapshot, ok := structured["snapshot"].(map[string]any) + if !ok { + return "", "" + } + snapshotID := stringValue(snapshot["id"]) + scopeData := map[string]any{} + for _, key := range []string{"scope", "workspace", "base_commit"} { + if value, exists := snapshot[key]; exists { + scopeData[key] = value + } + } + data, _ := json.Marshal(scopeData) + return snapshotID, DigestString(string(data)) +} + +func focusEvidenceDigest(arguments map[string]any) string { + selected := map[string]any{} + for _, key := range []string{"max_bytes", "max_packages", "fail_on", "race", "min_changed_coverage"} { + if value, exists := arguments[key]; exists { + selected[key] = value + } + } + data, _ := json.Marshal(selected) + return DigestString(string(data)) +} + +func focusSelector(arguments map[string]any) (string, string) { + if arguments == nil { + return "none", focusSelectorDigest("none", nil) + } + if previousPackID := stringValue(arguments["previous_pack_id"]); previousPackID != "" { + return "refresh", focusSelectorDigest("refresh", map[string]any{"previous_pack_id": previousPackID}) + } + for _, key := range []string{"symbol_ref", "query", "file", "focus_file", "focus_package"} { + if _, ok := arguments[key]; !ok { + continue + } + return key, focusSelectorDigest(key, arguments) + } + return "none", focusSelectorDigest("none", nil) +} + +func focusSelectorDigest(kind string, arguments map[string]any) string { + selected := map[string]any{"selector": kind} + for _, key := range []string{"symbol_ref", "query", "file", "line", "column", "focus_file", "focus_package", "previous_pack_id"} { + if value, ok := arguments[key]; ok { + selected[key] = value + } + } + data, _ := json.Marshal(selected) + return DigestString(string(data)) +} + func focusErrorText(item map[string]any) string { parts := make([]string, 0, 2) if errText := focusValueText(item["error"]); errText != "" { @@ -407,3 +552,73 @@ func focusErrorCategory(text string) string { return focusErrorUnknown } } + +func focusFailureClassification(text, category, selectorKind string) (string, string) { + lower := strings.ToLower(text) + // Keep the historical invalid_input category for compatibility, but treat + // an emitted refresh pack rejected by the current observation as stale + // lineage rather than selector misuse. + if strings.Contains(lower, "previous pack id is invalid") || (strings.Contains(lower, "previous pack") && strings.Contains(lower, "invalid")) { + return FocusFailureStaleSnapshot, "stale_snapshot" + } + switch category { + case focusErrorStale: + return FocusFailureStaleSnapshot, "stale_snapshot" + case focusErrorTimeout: + return FocusFailureTimeout, "timeout" + case focusErrorTransport: + return FocusFailureTransport, "transport" + } + if focusSelectorFailure(lower, category) { + return FocusFailureSelectorMisuse, focusSelectorFailureReason(lower, selectorKind) + } + if category == focusErrorProvider { + return FocusFailureProvider, "provider_error" + } + return FocusFailureUnknown, "unknown" +} + +func focusSelectorFailure(text, category string) bool { + if category == focusErrorInvalidInput { + return true + } + for _, marker := range []string{ + "symbol ref", "symbol_ref", "no identifier", "not a function", + "invalid declaration", "invalid position", "line and column", + } { + if strings.Contains(text, marker) { + return true + } + } + return false +} + +func focusSelectorFailureReason(text, selectorKind string) string { + if selectorKind == "symbol_ref" || strings.Contains(text, "symbol ref") || strings.Contains(text, "symbol_ref") { + return "invalid_symbol_ref" + } + if strings.Contains(text, "no identifier") || strings.Contains(text, "not a function") || strings.Contains(text, "declaration") || strings.Contains(text, "position") || strings.Contains(text, "line and column") { + return "invalid_declaration_coordinate" + } + return "invalid_selector" +} + +func focusFailureRecords(calls []focusCall) []FocusFailureRecord { + records := make([]FocusFailureRecord, 0) + for i, call := range calls { + if call.Success || call.FailureCause == "" { + continue + } + record := FocusFailureRecord{Cause: call.FailureCause, Selector: call.SelectorKind, Reason: call.FailureReason} + for _, next := range calls[i+1:] { + if !next.Success && next.SelectorDigest == call.SelectorDigest { + record.Repeated = true + } + if next.Success && (call.FailureCause != FocusFailureSelectorMisuse || next.SelectorDigest != call.SelectorDigest) { + record.Recovered = true + } + } + records = append(records, record) + } + return records +} diff --git a/validation/internal/pilot/runner_test.go b/validation/internal/pilot/runner_test.go index 06e997f..b2e39c5 100644 --- a/validation/internal/pilot/runner_test.go +++ b/validation/internal/pilot/runner_test.go @@ -2,6 +2,7 @@ package pilot import ( "encoding/json" + "fmt" "os" "strings" "testing" @@ -96,6 +97,130 @@ func TestFocusMetricsDoesNotInferRefreshOrEditFromProse(t *testing.T) { } } +func TestFocusMetricsFlagsOnlyRedundantRefreshes(t *testing.T) { + got := FocusMetrics(Events{Raw: []json.RawMessage{ + focusRefreshSuccessEvent(`{"query":"Worker","max_bytes":12000}`, "pack-1", "snapshot-1", "./..."), + json.RawMessage(`{"type":"item.completed","item":{"type":"file_change","status":"completed"}}`), + focusRefreshSuccessEvent(`{"previous_pack_id":"pack-1","max_bytes":12000}`, "pack-2", "snapshot-2", "./..."), + focusRefreshSuccessEvent(`{"previous_pack_id":"pack-2","max_bytes":12000}`, "pack-3", "snapshot-2", "./..."), + }}) + if got.RedundantRefreshes != 1 { + t.Fatalf("redundant refreshes = %d, want 1", got.RedundantRefreshes) + } + + got = FocusMetrics(Events{Raw: []json.RawMessage{ + focusRefreshSuccessEvent(`{"query":"Worker","max_bytes":12000}`, "pack-1", "snapshot-1", "./..."), + json.RawMessage(`{"type":"item.completed","item":{"type":"file_change","status":"completed"}}`), + focusRefreshSuccessEvent(`{"previous_pack_id":"pack-1","max_bytes":12000}`, "pack-2", "snapshot-2", "./..."), + json.RawMessage(`{"type":"item.completed","item":{"type":"file_change","status":"completed"}}`), + focusRefreshSuccessEvent(`{"previous_pack_id":"pack-2","max_bytes":12000}`, "pack-3", "snapshot-3", "./..."), + }}) + if got.RedundantRefreshes != 0 { + t.Fatalf("refresh after edit was flagged: %d", got.RedundantRefreshes) + } + + got = FocusMetrics(Events{Raw: []json.RawMessage{ + focusRefreshSuccessEvent(`{"query":"Worker","max_bytes":12000}`, "pack-1", "snapshot-1", "./..."), + json.RawMessage(`{"type":"item.completed","item":{"type":"file_change","status":"completed"}}`), + focusRefreshSuccessEvent(`{"previous_pack_id":"pack-1","max_bytes":12000}`, "pack-2", "snapshot-2", "./..."), + focusFailureEvent(`{"previous_pack_id":"pack-2"}`, "building change context: previous pack ID is invalid"), + focusRefreshSuccessEvent(`{"previous_pack_id":"pack-2","max_bytes":12000}`, "pack-3", "snapshot-2", "./..."), + }}) + if got.RedundantRefreshes != 0 { + t.Fatalf("refresh after stale rejection was flagged: %d", got.RedundantRefreshes) + } + + got = FocusMetrics(Events{Raw: []json.RawMessage{ + focusRefreshSuccessEvent(`{"query":"Worker","max_bytes":12000}`, "pack-1", "snapshot-1", "./..."), + json.RawMessage(`{"type":"item.completed","item":{"type":"file_change","status":"completed"}}`), + focusRefreshSuccessEvent(`{"previous_pack_id":"pack-1","max_bytes":12000}`, "pack-2", "snapshot-2", "./..."), + focusRefreshSuccessEvent(`{"previous_pack_id":"pack-2","max_bytes":24000}`, "pack-3", "snapshot-2", "./..."), + }}) + if got.RedundantRefreshes != 0 { + t.Fatalf("refresh with changed evidence requirements was flagged: %d", got.RedundantRefreshes) + } +} + +func TestFocusFailureClassificationAndRecovery(t *testing.T) { + e := Events{Raw: []json.RawMessage{ + focusSuccessEvent(`{"query":"Worker"}`, "pack-1"), + focusFailureEvent(`{"symbol_ref":"mutated-ref"}`, "invalid Symbol Ref"), + focusSuccessEvent(`{"file":"internal/worker.go","line":20,"column":6}`, "pack-2"), + focusFailureEvent(`{"file":"internal/worker.go","line":171,"column":1}`, "gopls RPC 0: no identifier found"), + focusSuccessEvent(`{"file":"internal/worker.go","line":171,"column":6}`, "pack-3"), + focusFailureEvent(`{"file":"internal/worker.go","line":172,"column":6}`, "gopls RPC 0: provider exploded"), + }} + got := FocusMetrics(e) + if got.Calls != 3 || got.FailedCalls != 3 || got.FailureCauses[FocusFailureSelectorMisuse] != 2 || got.FailureCauses[FocusFailureProvider] != 1 { + t.Fatalf("failure metrics = %+v", got) + } + if len(got.FailureRecords) != 3 { + t.Fatalf("failure records = %+v", got.FailureRecords) + } + if got.FailureRecords[0].Cause != FocusFailureSelectorMisuse || got.FailureRecords[0].Selector != "symbol_ref" || got.FailureRecords[0].Reason != "invalid_symbol_ref" || !got.FailureRecords[0].Recovered { + t.Fatalf("symbol ref record = %+v", got.FailureRecords[0]) + } + if got.FailureRecords[1].Cause != FocusFailureSelectorMisuse || got.FailureRecords[1].Reason != "invalid_declaration_coordinate" || !got.FailureRecords[1].Recovered { + t.Fatalf("coordinate record = %+v", got.FailureRecords[1]) + } + if got.FailureRecords[2].Cause != FocusFailureProvider || got.FailureRecords[2].Reason != "provider_error" || got.FailureRecords[2].Recovered { + t.Fatalf("provider record = %+v", got.FailureRecords[2]) + } + encoded, err := json.Marshal(got.FailureRecords) + if err != nil { + t.Fatal(err) + } + if strings.Contains(string(encoded), "mutated-ref") || strings.Contains(string(encoded), "internal/worker.go") || strings.Contains(string(encoded), "provider exploded") { + t.Fatalf("raw selector or provider data leaked: %s", encoded) + } +} + +func TestFocusFailureReplaysRejectEvidenceAndDetectRepeatedSelector(t *testing.T) { + e := Events{Raw: []json.RawMessage{ + focusFailureEvent(`{"file":"internal/worker.go","line":171,"column":1}`, "gopls RPC 0: no identifier found"), + focusFailureEvent(`{"file":"internal/worker.go","line":171,"column":1}`, "gopls RPC 0: no identifier found"), + focusSuccessEvent(`{"file":"internal/worker.go","line":171,"column":6}`, "pack-1"), + }} + got := FocusMetrics(e) + if got.Calls != 1 || got.FailedCalls != 2 || got.Evidence || got.FailureCauses[FocusFailureSelectorMisuse] != 2 { + t.Fatalf("replay metrics = %+v", got) + } + if len(got.FailureRecords) != 2 || !got.FailureRecords[0].Repeated || got.FailureRecords[0].Selector != "file" || got.FailureRecords[0].Reason != "invalid_declaration_coordinate" { + t.Fatalf("repeated selector records = %+v", got.FailureRecords) + } + if !got.FailureRecords[0].Recovered || !got.FailureRecords[1].Recovered { + t.Fatalf("fresh-selector recovery not recorded: %+v", got.FailureRecords) + } +} + +func TestFocusFailureClassifiesRejectedPreviousPackAsStale(t *testing.T) { + got := FocusMetrics(Events{Raw: []json.RawMessage{ + focusFailureEvent(`{"previous_pack_id":"emitted-pack"}`, "building change context: previous pack ID is invalid"), + }}) + if got.ErrorCategories[focusErrorInvalidInput] != 1 { + t.Fatalf("historical category changed: %+v", got.ErrorCategories) + } + if got.FailureCauses[FocusFailureStaleSnapshot] != 1 || len(got.FailureRecords) != 1 { + t.Fatalf("stale lineage classification = %+v", got) + } + record := got.FailureRecords[0] + if record.Cause != FocusFailureStaleSnapshot || record.Selector != "refresh" || record.Reason != "stale_snapshot" { + t.Fatalf("stale lineage record = %+v", record) + } +} + +func focusSuccessEvent(arguments, packID string) json.RawMessage { + return json.RawMessage(fmt.Sprintf(`{"type":"item.completed","item":{"type":"mcp_tool_call","tool":"go_context","status":"completed","arguments":%s,"result":{"structured_content":{"pack_id":"%s"}}}}`, arguments, packID)) +} + +func focusRefreshSuccessEvent(arguments, packID, snapshotID, scope string) json.RawMessage { + return json.RawMessage(fmt.Sprintf(`{"type":"item.completed","item":{"type":"mcp_tool_call","tool":"go_context","status":"completed","arguments":%s,"result":{"structured_content":{"pack_id":"%s","snapshot":{"id":"%s","scope":"%s","workspace":".","base_commit":"base"}}}}}`, arguments, packID, snapshotID, scope)) +} + +func focusFailureEvent(arguments, message string) json.RawMessage { + return json.RawMessage(fmt.Sprintf(`{"type":"item.completed","item":{"type":"mcp_tool_call","tool":"go_context","status":"failed","arguments":%s,"result":{"content":[{"type":"text","text":"%s"}]}}}`, arguments, message)) +} + func TestSkillDiscoveryEvidenceRequiresSkillFileMarker(t *testing.T) { if SkillDiscoveryEvidence(Events{Raw: []json.RawMessage{ json.RawMessage(`{"type":"item.completed","item":{"type":"agent_message","text":"agentic-go-context is useful"}}`), diff --git a/validation/internal/pilot/types.go b/validation/internal/pilot/types.go index 9c985f4..e338129 100644 --- a/validation/internal/pilot/types.go +++ b/validation/internal/pilot/types.go @@ -29,42 +29,45 @@ type Scenario struct { // //nolint:govet // JSON record layout is kept grouped by contract field category. type Run struct { - SchemaVersion string `json:"schema_version"` - ScenarioID string `json:"scenario_id"` - TaskID string `json:"task_id"` - Condition string `json:"condition"` - Repetition int `json:"repetition"` - Model string `json:"model"` - Reasoning string `json:"reasoning"` - SourceSHA256 string `json:"source_sha256"` - BinarySHA256 string `json:"binary_sha256"` - WorkspaceSHA256 string `json:"workspace_sha256"` - Prompt string `json:"prompt"` - Transcript []json.RawMessage `json:"transcript"` - Patch string `json:"patch"` - PatchSHA256 string `json:"patch_sha256"` - Acceptance string `json:"acceptance"` - ProcessError string `json:"process_error,omitempty"` - Stderr string `json:"stderr,omitempty"` - AcceptanceEvidenceSHA256 string `json:"acceptance_evidence_sha256"` - ScopeViolations []string `json:"scope_violations"` - EvidenceBytes int64 `json:"evidence_bytes"` - ToolCalls int `json:"tool_calls"` - FocusToolCalls int `json:"focus_tool_calls"` - FocusFailedCalls int `json:"focus_failed_calls"` - FocusErrorCategories map[string]int `json:"focus_error_categories"` - FirstFocusCallPosition int `json:"first_focus_call_position"` - RefreshUse bool `json:"refresh_use"` - FocusEvidenceUse bool `json:"focus_evidence_use"` - FocusResultFollowedByEdit bool `json:"focus_result_followed_by_edit"` - RefreshCompleted bool `json:"refresh_completed"` - FocusDelivery string `json:"focus_delivery"` - DurationMS int64 `json:"duration_ms"` - OperatorIntervention bool `json:"operator_intervention"` - Uncertainty []string `json:"uncertainty"` - Qualifying bool `json:"qualifying"` - DecisionObligations []ObligationResult `json:"decision_obligations"` - TranscriptSHA256 string `json:"transcript_sha256"` + SchemaVersion string `json:"schema_version"` + ScenarioID string `json:"scenario_id"` + TaskID string `json:"task_id"` + Condition string `json:"condition"` + Repetition int `json:"repetition"` + Model string `json:"model"` + Reasoning string `json:"reasoning"` + SourceSHA256 string `json:"source_sha256"` + BinarySHA256 string `json:"binary_sha256"` + WorkspaceSHA256 string `json:"workspace_sha256"` + Prompt string `json:"prompt"` + Transcript []json.RawMessage `json:"transcript"` + Patch string `json:"patch"` + PatchSHA256 string `json:"patch_sha256"` + Acceptance string `json:"acceptance"` + ProcessError string `json:"process_error,omitempty"` + Stderr string `json:"stderr,omitempty"` + AcceptanceEvidenceSHA256 string `json:"acceptance_evidence_sha256"` + ScopeViolations []string `json:"scope_violations"` + EvidenceBytes int64 `json:"evidence_bytes"` + ToolCalls int `json:"tool_calls"` + FocusToolCalls int `json:"focus_tool_calls"` + FocusFailedCalls int `json:"focus_failed_calls"` + FocusErrorCategories map[string]int `json:"focus_error_categories"` + FocusFailureCauses map[string]int `json:"focus_failure_causes,omitempty"` + FocusFailureRecords []FocusFailureRecord `json:"focus_failure_records,omitempty"` + FirstFocusCallPosition int `json:"first_focus_call_position"` + RefreshUse bool `json:"refresh_use"` + FocusEvidenceUse bool `json:"focus_evidence_use"` + FocusResultFollowedByEdit bool `json:"focus_result_followed_by_edit"` + RefreshCompleted bool `json:"refresh_completed"` + RedundantRefreshes int `json:"redundant_refreshes,omitempty"` + FocusDelivery string `json:"focus_delivery"` + DurationMS int64 `json:"duration_ms"` + OperatorIntervention bool `json:"operator_intervention"` + Uncertainty []string `json:"uncertainty"` + Qualifying bool `json:"qualifying"` + DecisionObligations []ObligationResult `json:"decision_obligations"` + TranscriptSHA256 string `json:"transcript_sha256"` } // ObligationResult records evidence for one scenario obligation. @@ -179,11 +182,14 @@ func LoadRun(path string) (Run, error) { r.FocusToolCalls = focus.Calls r.FocusFailedCalls = focus.FailedCalls r.FocusErrorCategories = focus.ErrorCategories + r.FocusFailureCauses = focus.FailureCauses + r.FocusFailureRecords = focus.FailureRecords r.FirstFocusCallPosition = focus.FirstPosition r.RefreshUse = focus.Refresh r.FocusEvidenceUse = focus.Evidence r.FocusResultFollowedByEdit = focus.FocusResultFollowedByEdit r.RefreshCompleted = focus.RefreshCompleted + r.RedundantRefreshes = focus.RedundantRefreshes return r, nil } From 8c737ad44629bca8b1feafbab1d01b61fa8e955c Mon Sep 17 00:00:00 2001 From: Ashwin Gopalsamy <47941624+ashwingopalsamy@users.noreply.github.com> Date: Sun, 27 Sep 2026 12:34:14 +0530 Subject: [PATCH 17/20] feat(retrieval-eval): add bounded branch corpus screen --- internal/intelligence/retrieval/index.go | 773 ++++ internal/intelligence/retrieval/index_test.go | 261 ++ validation/cmd/retrievalbench/README.md | 137 + validation/cmd/retrievalbench/main.go | 56 + .../cmd/retrievalbench/manifest.schema.json | 122 + validation/internal/retrievalstudy/archive.go | 564 +++ .../internal/retrievalstudy/archive_test.go | 68 + validation/internal/retrievalstudy/golist.go | 127 + .../internal/retrievalstudy/golist_test.go | 30 + validation/internal/retrievalstudy/gopls.go | 207 + .../internal/retrievalstudy/identity.go | 182 + .../internal/retrievalstudy/manifest.go | 176 + validation/internal/retrievalstudy/metrics.go | 154 + validation/internal/retrievalstudy/native.go | 278 ++ validation/internal/retrievalstudy/run.go | 657 +++ .../internal/retrievalstudy/run_test.go | 41 + validation/internal/retrievalstudy/types.go | 335 ++ .../heldout-v1/large-kubernetes.json | 54 + .../heldout-v1/medium-agentic-go.json | 57 + .../retrieval/heldout-v1/small-tour.json | 45 + .../heldout-v2/medium-agentic-go-text.json | 48 + .../retrieval/provenance/2026-09-27/README.md | 25 + .../2026-09-27/decision_value_eval.diff.gz | Bin 0 -> 39594 bytes .../2026-09-27/decision_value_eval.status | 34 + .../decision_value_eval.untracked-sha256 | 14 + .../2026-09-27/v1_2_reliability.diff.gz | Bin 0 -> 40777 bytes .../2026-09-27/v1_2_reliability.status | 25 + .../v1_2_reliability.untracked-sha256 | 5 + .../large-kubernetes-capped-v2.json | 3525 ++++++++++++++ .../results/2026-09-27/large-kubernetes.json | 3525 ++++++++++++++ .../medium-agentic-go-anchor-v2.json | 3616 +++++++++++++++ .../medium-agentic-go-text-ablation-v1.json | 3793 +++++++++++++++ .../results/2026-09-27/medium-agentic-go.json | 3618 +++++++++++++++ .../results/2026-09-27/small-tour.json | 4091 +++++++++++++++++ 34 files changed, 26643 insertions(+) create mode 100644 internal/intelligence/retrieval/index.go create mode 100644 internal/intelligence/retrieval/index_test.go create mode 100644 validation/cmd/retrievalbench/README.md create mode 100644 validation/cmd/retrievalbench/main.go create mode 100644 validation/cmd/retrievalbench/manifest.schema.json create mode 100644 validation/internal/retrievalstudy/archive.go create mode 100644 validation/internal/retrievalstudy/archive_test.go create mode 100644 validation/internal/retrievalstudy/golist.go create mode 100644 validation/internal/retrievalstudy/golist_test.go create mode 100644 validation/internal/retrievalstudy/gopls.go create mode 100644 validation/internal/retrievalstudy/identity.go create mode 100644 validation/internal/retrievalstudy/manifest.go create mode 100644 validation/internal/retrievalstudy/metrics.go create mode 100644 validation/internal/retrievalstudy/native.go create mode 100644 validation/internal/retrievalstudy/run.go create mode 100644 validation/internal/retrievalstudy/run_test.go create mode 100644 validation/internal/retrievalstudy/types.go create mode 100644 validation/retrieval/heldout-v1/large-kubernetes.json create mode 100644 validation/retrieval/heldout-v1/medium-agentic-go.json create mode 100644 validation/retrieval/heldout-v1/small-tour.json create mode 100644 validation/retrieval/heldout-v2/medium-agentic-go-text.json create mode 100644 validation/retrieval/provenance/2026-09-27/README.md create mode 100644 validation/retrieval/provenance/2026-09-27/decision_value_eval.diff.gz create mode 100644 validation/retrieval/provenance/2026-09-27/decision_value_eval.status create mode 100644 validation/retrieval/provenance/2026-09-27/decision_value_eval.untracked-sha256 create mode 100644 validation/retrieval/provenance/2026-09-27/v1_2_reliability.diff.gz create mode 100644 validation/retrieval/provenance/2026-09-27/v1_2_reliability.status create mode 100644 validation/retrieval/provenance/2026-09-27/v1_2_reliability.untracked-sha256 create mode 100644 validation/retrieval/results/2026-09-27/large-kubernetes-capped-v2.json create mode 100644 validation/retrieval/results/2026-09-27/large-kubernetes.json create mode 100644 validation/retrieval/results/2026-09-27/medium-agentic-go-anchor-v2.json create mode 100644 validation/retrieval/results/2026-09-27/medium-agentic-go-text-ablation-v1.json create mode 100644 validation/retrieval/results/2026-09-27/medium-agentic-go.json create mode 100644 validation/retrieval/results/2026-09-27/small-tour.json diff --git a/internal/intelligence/retrieval/index.go b/internal/intelligence/retrieval/index.go new file mode 100644 index 0000000..e3afa62 --- /dev/null +++ b/internal/intelligence/retrieval/index.go @@ -0,0 +1,773 @@ +// Package retrieval provides bounded, deterministic discovery over observed Go +// source. It deliberately returns source coordinates only; the intelligence +// layer remains responsible for resolving them through the active semantic +// provider before exposing evidence. +package retrieval + +import ( + "bytes" + "container/heap" + "container/list" + "context" + "crypto/sha256" + "go/ast" + "go/parser" + "go/token" + "math" + "sort" + "strings" + "sync" + "time" + "unicode" + "unicode/utf8" +) + +const ( + defaultMaximumCacheBytes = 64 << 20 + bm25K1 = 1.2 + bm25B = 0.75 + qualifiedBoost = 12.0 + nameBoost = 8.0 + receiverBoost = 3.0 + packageBoost = 2.0 + kindBoost = 1.5 + identifierBoost = 2.0 +) + +// MaximumTextLineFragments bounds transient text candidate expansion in the +// evaluation-only mixed retrieval path. +const MaximumTextLineFragments = 50_000 + +// Key identifies the source interpretation under which fragments were built. +// The caller supplies values from the active snapshot rather than an absolute +// filesystem path so cached fragments cannot become public evidence by +// themselves. +type Key struct { + Workspace string + Scope string + Build string + Provider string +} + +// File is one contained file from an already validated workspace observation. +type File struct { + Path string + Digest string + Contents []byte +} + +// Candidate is a ranked source coordinate for semantic-provider resolution. +// Line and Column are one-based UTF-8 byte coordinates. +type Candidate struct { + Path string + Line int + Column int + Name string + Qualified string + Kind string + Package string + Score float64 +} + +// Result reports ranked candidates and whether every eligible Go file was +// structurally indexed. A partial result is still useful for discovery, but +// callers must preserve the incompleteness as uncertainty. +type Result struct { + Candidates []Candidate + CandidateCount int + IndexedFiles int + SkippedFiles int + TextIndexedFiles int + TextSkippedFiles int + TextIndexedFragments int + Complete bool + Truncated bool +} + +// SearchProfile reports non-overlapping work performed by one SearchProfiled +// call. It is internal diagnostic data and is not part of any MCP contract. +type SearchProfile struct { + SearchDuration time.Duration + ParseDuration time.Duration + AggregateDuration time.Duration + RankDuration time.Duration + FileVisits int + CacheHits int + FilesParsed int +} + +type fileKey struct { + workspace string + scope string + build string + provider string + path string + digest string +} + +type cacheEntry struct { + key fileKey + index indexedFile + complete bool + size int +} + +// Cache reuses parsed per-file fragments across observations while retaining +// only bounded derived metadata in memory. Eviction only causes reparsing. +type Cache struct { + mu sync.Mutex + maximum int + bytes int + entries map[fileKey]*list.Element + order *list.List +} + +// NewCache constructs the process-local retrieval cache. +func NewCache() *Cache { + return &Cache{ + maximum: defaultMaximumCacheBytes, + entries: make(map[fileKey]*list.Element), + order: list.New(), + } +} + +// Search ranks declarations from the supplied observed files. It never reads +// from disk and never returns a cached result without rebuilding the current +// file set's aggregate statistics. +func (c *Cache) Search(ctx context.Context, key Key, files []File, query string, limit int) (Result, error) { + result, _, err := c.SearchProfiled(ctx, key, files, query, limit) + return result, err +} + +// SearchProfiled ranks declarations like Search and returns stage timings for +// benchmark and diagnostic use. It never reads from disk and does not alter +// the search result or cache policy. +func (c *Cache) SearchProfiled(ctx context.Context, key Key, files []File, query string, limit int) (Result, SearchProfile, error) { + return c.searchProfiled(ctx, key, files, nil, query, limit) +} + +// SearchWithTextProfiled runs the declaration scorer over Go declaration +// anchors plus bounded caller-supplied text-line fragments. It does not read +// files from disk. The live intelligence path continues to call +// SearchProfiled, which indexes Go declarations only. +func (c *Cache) SearchWithTextProfiled(ctx context.Context, key Key, goFiles, textFiles []File, query string, limit int) (Result, SearchProfile, error) { + return c.searchProfiled(ctx, key, goFiles, textFiles, query, limit) +} + +func (c *Cache) searchProfiled(ctx context.Context, key Key, files, textFiles []File, query string, limit int) (Result, SearchProfile, error) { + started := time.Now() + profile := SearchProfile{} + if c == nil { + c = NewCache() + } + terms := tokenize(query) + if len(terms) == 0 { + profile.SearchDuration = time.Since(started) + return Result{Candidates: []Candidate{}, Complete: true}, profile, nil + } + if limit < 1 { + limit = 20 + } + + result := Result{Candidates: []Candidate{}, Complete: true} + collectStarted := time.Now() + queryCounts := countTerms(terms) + documentFrequency := make(map[string]int, len(queryCounts)) + totalFragments := 0 + totalLength := 0 + textFragments := make([]fragment, 0) + for _, file := range files { + if err := ctx.Err(); err != nil { + collectDuration := time.Since(collectStarted) + profile.AggregateDuration += collectDuration - profile.ParseDuration + profile.SearchDuration = time.Since(started) + return Result{}, profile, err + } + if !strings.HasSuffix(strings.ToLower(file.Path), ".go") { + continue + } + profile.FileVisits++ + indexed, complete, cacheHit, parseDuration := c.fileIndex(key, file) + if cacheHit { + profile.CacheHits++ + } else { + profile.FilesParsed++ + profile.ParseDuration += parseDuration + } + if !complete { + result.Complete = false + result.SkippedFiles++ + } + if len(indexed.fragments) == 0 { + continue + } + result.IndexedFiles++ + for _, item := range indexed.fragments { + totalFragments++ + totalLength += item.length + for term := range item.terms { + if _, queried := queryCounts[term]; queried { + documentFrequency[term]++ + } + } + } + } + for fileIndex, file := range textFiles { + if err := ctx.Err(); err != nil { + collectDuration := time.Since(collectStarted) + profile.AggregateDuration += collectDuration - profile.ParseDuration + profile.SearchDuration = time.Since(started) + return Result{}, profile, err + } + profile.FileVisits++ + remaining := MaximumTextLineFragments - result.TextIndexedFragments + if remaining <= 0 { + result.Complete = false + result.TextSkippedFiles += len(textFiles) - fileIndex + break + } + fragments, valid, fullyIndexed := indexTextLines(file, remaining) + if !valid { + result.Complete = false + result.TextSkippedFiles++ + continue + } + if !fullyIndexed { + result.Complete = false + result.TextSkippedFiles += len(textFiles) - fileIndex + break + } + result.TextIndexedFiles++ + result.TextIndexedFragments += len(fragments) + for _, item := range fragments { + textFragments = append(textFragments, item) + totalFragments++ + totalLength += item.length + for term := range item.terms { + if _, queried := queryCounts[term]; queried { + documentFrequency[term]++ + } + } + } + } + profile.AggregateDuration += time.Since(collectStarted) - profile.ParseDuration + + if totalFragments == 0 { + profile.SearchDuration = time.Since(started) + return result, profile, nil + } + + averageLength := float64(totalLength) / float64(totalFragments) + if averageLength == 0 { + averageLength = 1 + } + + rankStarted := time.Now() + top := candidateHeap{} + heap.Init(&top) + candidateCount := 0 + secondPassParseDuration := time.Duration(0) + for _, file := range files { + if !strings.HasSuffix(strings.ToLower(file.Path), ".go") { + continue + } + profile.FileVisits++ + if err := ctx.Err(); err != nil { + profile.RankDuration += time.Since(rankStarted) - secondPassParseDuration + profile.SearchDuration = time.Since(started) + return Result{}, profile, err + } + indexed, _, cacheHit, parseDuration := c.fileIndex(key, file) + if cacheHit { + profile.CacheHits++ + } else { + profile.FilesParsed++ + profile.ParseDuration += parseDuration + secondPassParseDuration += parseDuration + } + for _, item := range indexed.fragments { + candidateScore := scoreWithQueryCounts(item, terms, queryCounts, documentFrequency, totalFragments, averageLength) + if candidateScore <= 0 { + continue + } + candidateCount++ + candidate := Candidate{ + Path: item.path, Line: item.line, Column: item.column, + Name: item.name, Qualified: item.qualified, Kind: item.kind, + Package: item.packageName, Score: candidateScore, + } + if top.Len() < limit { + heap.Push(&top, candidate) + } else if candidateRanksBefore(candidate, top[0]) { + top[0] = candidate + heap.Fix(&top, 0) + } + } + } + for _, item := range textFragments { + if err := ctx.Err(); err != nil { + profile.RankDuration += time.Since(rankStarted) - secondPassParseDuration + profile.SearchDuration = time.Since(started) + return Result{}, profile, err + } + candidateScore := scoreWithQueryCounts(item, terms, queryCounts, documentFrequency, totalFragments, averageLength) + if candidateScore <= 0 { + continue + } + candidateCount++ + candidate := Candidate{ + Path: item.path, Line: item.line, Column: item.column, + Name: item.name, Qualified: item.qualified, Kind: item.kind, + Package: item.packageName, Score: candidateScore, + } + if top.Len() < limit { + heap.Push(&top, candidate) + } else if candidateRanksBefore(candidate, top[0]) { + top[0] = candidate + heap.Fix(&top, 0) + } + } + profile.RankDuration += time.Since(rankStarted) - secondPassParseDuration + result.CandidateCount = candidateCount + result.Truncated = candidateCount > limit + result.Candidates = append(result.Candidates, top...) + sort.SliceStable(result.Candidates, func(i, j int) bool { + return candidateRanksBefore(result.Candidates[i], result.Candidates[j]) + }) + profile.SearchDuration = time.Since(started) + return result, profile, nil +} + +func indexTextLines(file File, maximumFragments int) ([]fragment, bool, bool) { + if strings.HasSuffix(strings.ToLower(file.Path), ".go") || !utf8.Valid(file.Contents) || bytes.IndexByte(file.Contents, 0) >= 0 { + return nil, false, false + } + fragments := make([]fragment, 0) + lineNumber := 1 + for start := 0; start < len(file.Contents); { + end := bytes.IndexByte(file.Contents[start:], '\n') + var line []byte + if end < 0 { + line = file.Contents[start:] + start = len(file.Contents) + } else { + line = file.Contents[start : start+end] + start += end + 1 + } + line = bytes.TrimSuffix(line, []byte{'\r'}) + terms := countTerms(tokenize(string(line))) + if len(terms) != 0 { + if len(fragments) >= maximumFragments { + return nil, true, false + } + fragments = append(fragments, fragment{ + path: file.Path, line: lineNumber, kind: "text.line", + terms: terms, length: termCount(terms), + }) + } + lineNumber++ + } + return fragments, true, true +} + +// candidateHeap keeps the worst selected candidate at the root so ranking can +// retain only the requested result count while scanning large workspaces. +type candidateHeap []Candidate + +func (candidates candidateHeap) Len() int { return len(candidates) } + +func (candidates candidateHeap) Less(left, right int) bool { + return candidateRanksBefore(candidates[right], candidates[left]) +} + +func (candidates candidateHeap) Swap(left, right int) { + candidates[left], candidates[right] = candidates[right], candidates[left] +} + +func (candidates *candidateHeap) Push(value any) { + *candidates = append(*candidates, value.(Candidate)) +} + +func (candidates *candidateHeap) Pop() any { + items := *candidates + last := len(items) - 1 + value := items[last] + items[last] = Candidate{} + *candidates = items[:last] + return value +} + +func candidateRanksBefore(left, right Candidate) bool { + if left.Score != right.Score { + return left.Score > right.Score + } + if left.Path != right.Path { + return left.Path < right.Path + } + if left.Line != right.Line { + return left.Line < right.Line + } + if left.Column != right.Column { + return left.Column < right.Column + } + return left.Name < right.Name +} + +type indexedFile struct { + fragments []fragment +} + +type fragment struct { + path string + line int + column int + name string + qualified string + kind string + packageName string + receiver string + terms map[string]int + nameTerms []string + qualifiedTerms []string + receiverTerms []string + packageTerms []string + kindTerms []string + length int +} + +func (c *Cache) fileIndex(key Key, file File) (indexedFile, bool, bool, time.Duration) { + digest := file.Digest + if digest == "" { + digestBytes := sha256.Sum256(file.Contents) + digest = string(digestBytes[:]) + } + cacheKey := fileKey{ + workspace: key.Workspace, scope: key.Scope, build: key.Build, + provider: key.Provider, path: file.Path, digest: digest, + } + + c.mu.Lock() + defer c.mu.Unlock() + if element, ok := c.entries[cacheKey]; ok { + c.order.MoveToFront(element) + entry := element.Value.(cacheEntry) + return entry.index, entry.complete, true, 0 + } + + started := time.Now() + indexed, complete := parseFile(file.Path, file.Contents) + parseDuration := time.Since(started) + entry := cacheEntry{key: cacheKey, index: indexed, complete: complete, size: estimateSize(indexed)} + if entry.size <= c.maximum { + for c.bytes+entry.size > c.maximum && c.order.Len() > 0 { + oldest := c.order.Back() + if oldest == nil { + break + } + removed := oldest.Value.(cacheEntry) + delete(c.entries, removed.key) + c.bytes -= removed.size + c.order.Remove(oldest) + } + element := c.order.PushFront(entry) + c.entries[cacheKey] = element + c.bytes += entry.size + } + return indexed, complete, false, parseDuration +} + +func parseFile(path string, contents []byte) (indexedFile, bool) { + fileSet := token.NewFileSet() + parsed, parseErr := parser.ParseFile(fileSet, path, contents, parser.ParseComments|parser.AllErrors) + if parsed == nil { + return indexedFile{}, false + } + complete := parseErr == nil + fragments := make([]fragment, 0) + for _, declaration := range parsed.Decls { + switch declaration := declaration.(type) { + case *ast.FuncDecl: + name := declaration.Name.Name + receiver := receiverName(declaration.Recv) + kind := "go.function" + qualified := parsed.Name.Name + "." + name + if receiver != "" { + kind = "go.method" + qualified = parsed.Name.Name + "." + receiver + "." + name + } + fragments = append(fragments, makeFragment(fileSet, contents, path, declaration.Pos(), declaration.End(), declaration.Name.Pos(), name, qualified, kind, parsed.Name.Name, receiver, declaration.Doc)) + case *ast.GenDecl: + fragments = append(fragments, fragmentsFromGenDecl(fileSet, contents, path, parsed.Name.Name, declaration)...) + } + } + return indexedFile{fragments: fragments}, complete +} + +func fragmentsFromGenDecl(fileSet *token.FileSet, contents []byte, path, packageName string, declaration *ast.GenDecl) []fragment { + fragments := make([]fragment, 0, len(declaration.Specs)) + for _, specification := range declaration.Specs { + switch specification := specification.(type) { + case *ast.TypeSpec: + kind := "go.type" + if specification.Assign.IsValid() { + kind = "go.type_alias" + } + fragments = append(fragments, makeFragment(fileSet, contents, path, specification.Pos(), specification.End(), specification.Name.Pos(), specification.Name.Name, packageName+"."+specification.Name.Name, kind, packageName, "", firstComment(specification.Doc, declaration.Doc))) + case *ast.ValueSpec: + kind := "go.var" + if declaration.Tok == token.CONST { + kind = "go.const" + } + for _, name := range specification.Names { + fragments = append(fragments, makeFragment(fileSet, contents, path, specification.Pos(), specification.End(), name.Pos(), name.Name, packageName+"."+name.Name, kind, packageName, "", firstComment(specification.Doc, declaration.Doc))) + } + } + } + return fragments +} + +func firstComment(primary, fallback *ast.CommentGroup) *ast.CommentGroup { + if primary != nil { + return primary + } + return fallback +} + +func makeFragment(fileSet *token.FileSet, contents []byte, path string, start, end, namePosition token.Pos, name, qualified, kind, packageName, receiver string, doc *ast.CommentGroup) fragment { + startOffset := fileSet.PositionFor(start, false).Offset + endOffset := fileSet.PositionFor(end, false).Offset + if startOffset < 0 { + startOffset = 0 + } + if endOffset < startOffset || endOffset > len(contents) { + endOffset = len(contents) + } + position := fileSet.PositionFor(namePosition, false) + var text strings.Builder + text.Grow(endOffset - startOffset + len(name) + len(qualified) + len(kind) + len(packageName) + len(receiver) + 8) + text.Write(contents[startOffset:endOffset]) + if doc != nil { + text.WriteByte('\n') + text.WriteString(doc.Text()) + } + text.WriteByte('\n') + text.WriteString(name) + text.WriteByte('\n') + text.WriteString(qualified) + text.WriteByte('\n') + text.WriteString(kind) + text.WriteByte('\n') + text.WriteString(packageName) + text.WriteByte('\n') + text.WriteString(receiver) + terms := countTerms(tokenize(text.String())) + return fragment{ + path: path, line: position.Line, column: position.Column, + name: name, qualified: qualified, kind: kind, packageName: packageName, + receiver: receiver, terms: terms, nameTerms: tokenize(name), + qualifiedTerms: tokenize(qualified), receiverTerms: tokenize(receiver), + packageTerms: tokenize(packageName), kindTerms: tokenize(kind), length: termCount(terms), + } +} + +func receiverName(fields *ast.FieldList) string { + if fields == nil || len(fields.List) == 0 { + return "" + } + var expression ast.Expr = fields.List[0].Type + for { + switch typed := expression.(type) { + case *ast.StarExpr: + expression = typed.X + case *ast.ParenExpr: + expression = typed.X + case *ast.IndexExpr: + expression = typed.X + case *ast.IndexListExpr: + expression = typed.X + case *ast.Ident: + return typed.Name + default: + return "" + } + } +} + +func score(item fragment, query []string, documentFrequency map[string]int, documentCount int, averageLength float64) float64 { + return scoreWithQueryCounts(item, query, countTerms(query), documentFrequency, documentCount, averageLength) +} + +func scoreWithQueryCounts(item fragment, query []string, queryCounts map[string]int, documentFrequency map[string]int, documentCount int, averageLength float64) float64 { + score := 0.0 + for term, queryFrequency := range queryCounts { + frequency := item.terms[term] + if frequency == 0 { + continue + } + documentFrequencyForTerm := documentFrequency[term] + inverseFrequency := math.Log(1 + (float64(documentCount-documentFrequencyForTerm)+0.5)/(float64(documentFrequencyForTerm)+0.5)) + normalizedLength := float64(item.length) + termFrequency := float64(frequency) + bm25 := inverseFrequency * (termFrequency * (bm25K1 + 1)) / + (termFrequency + bm25K1*(1-bm25B+bm25B*normalizedLength/averageLength)) + score += bm25 * float64(queryFrequency) + } + if containsSequence(query, item.qualifiedTerms) { + score += qualifiedBoost + } else if containsSequence(query, item.nameTerms) { + score += nameBoost + } + if containsSequence(query, item.receiverTerms) { + score += receiverBoost + } + if overlaps(query, item.packageTerms) { + score += packageBoost + } + if overlaps(query, item.kindTerms) { + score += kindBoost + } + if overlaps(query, item.nameTerms) || overlaps(query, item.qualifiedTerms) { + score += identifierBoost + } + return score +} + +func estimateSize(index indexedFile) int { + size := 0 + for _, item := range index.fragments { + // Account conservatively for the fragment, term map buckets, and token + // slice headers. The cache limit is intended to bound retained metadata, + // so counting only string payloads materially understates its footprint. + size += 256 + size += len(item.path) + len(item.name) + len(item.qualified) + len(item.kind) + len(item.packageName) + len(item.receiver) + size += len(item.terms) * 64 + for term := range item.terms { + size += len(term) + } + size = estimateTerms(size, item.nameTerms) + size = estimateTerms(size, item.qualifiedTerms) + size = estimateTerms(size, item.receiverTerms) + size = estimateTerms(size, item.packageTerms) + size = estimateTerms(size, item.kindTerms) + } + return size +} + +func estimateTerms(size int, terms []string) int { + size += len(terms) * 16 + for _, term := range terms { + size += len(term) + } + return size +} + +func tokenize(value string) []string { + value = strings.TrimSpace(value) + if value == "" { + return nil + } + result := make([]string, 0, len(value)/8) + var normalized strings.Builder + ends := make([]int, 0, len(value)/8) + start := -1 + flush := func(end int) { + if start < 0 || end <= start { + start = -1 + return + } + normalized.WriteString(strings.ToLower(value[start:end])) + ends = append(ends, normalized.Len()) + normalized.WriteByte(0) + start = -1 + } + var previous rune + for index := 0; index < len(value); { + current, size := utf8.DecodeRuneInString(value[index:]) + if !unicode.IsLetter(current) && !unicode.IsDigit(current) && current != '_' { + flush(index) + previous = 0 + index += size + continue + } + if start < 0 { + start = index + previous = current + index += size + continue + } + if current == '_' { + flush(index) + previous = 0 + index += size + continue + } + if unicode.IsUpper(current) && unicode.IsLower(previous) { + flush(index) + start = index + } + previous = current + index += size + } + flush(len(value)) + if len(ends) == 0 { + return result[:0] + } + packed := normalized.String() + start = 0 + for _, end := range ends { + result = append(result, packed[start:end]) + start = end + 1 + } + return result +} + +func countTerms(terms []string) map[string]int { + counts := make(map[string]int, len(terms)) + for _, term := range terms { + counts[term]++ + } + return counts +} + +func termCount(terms map[string]int) int { + count := 0 + for _, frequency := range terms { + count += frequency + } + return count +} + +func overlaps(left, right []string) bool { + if len(left) == 0 || len(right) == 0 { + return false + } + set := make(map[string]struct{}, len(right)) + for _, value := range right { + set[value] = struct{}{} + } + for _, value := range left { + if _, ok := set[value]; ok { + return true + } + } + return false +} + +func containsSequence(haystack, needle []string) bool { + if len(needle) == 0 || len(needle) > len(haystack) { + return false + } + for start := 0; start <= len(haystack)-len(needle); start++ { + match := true + for offset, value := range needle { + if haystack[start+offset] != value { + match = false + break + } + } + if match { + return true + } + } + return false +} diff --git a/internal/intelligence/retrieval/index_test.go b/internal/intelligence/retrieval/index_test.go new file mode 100644 index 0000000..45c34ca --- /dev/null +++ b/internal/intelligence/retrieval/index_test.go @@ -0,0 +1,261 @@ +package retrieval + +import ( + "bytes" + "context" + "math" + "reflect" + "sort" + "testing" +) + +func TestSearchRanksExactAndStructuralMatches(t *testing.T) { + cache := NewCache() + files := []File{{ + Path: "payments.go", + Digest: "payments-v1", + Contents: []byte(`package payments + +// ProcessPayment validates and processes an incoming payment request. +func ProcessPayment() {} + +// ReconcilePayment compares settled payment records. +func ReconcilePayment() {} +`), + }} + + result, err := cache.Search(context.Background(), Key{Workspace: "repo", Scope: "./...", Build: "go1.25", Provider: "gopls"}, files, "process incoming payment", 10) + if err != nil { + t.Fatal(err) + } + if len(result.Candidates) == 0 || result.Candidates[0].Name != "ProcessPayment" { + t.Fatalf("Search() candidates = %#v, want ProcessPayment first", result.Candidates) + } + if result.Candidates[0].Line != 4 || result.Candidates[0].Column != 6 { + t.Fatalf("Search() location = %d:%d, want 4:6", result.Candidates[0].Line, result.Candidates[0].Column) + } +} + +func TestSearchKeepsGroupedSpecsAsSeparateFragments(t *testing.T) { + file := File{Path: "values.go", Contents: []byte(`package fixture + +const ( + Alpha = "needle" + Beta = "ordinary" +) +`)} + result, err := NewCache().Search(context.Background(), Key{Workspace: "repo"}, []File{file}, "needle", 10) + if err != nil { + t.Fatal(err) + } + if len(result.Candidates) != 1 || result.Candidates[0].Name != "Alpha" { + t.Fatalf("Search() candidates = %#v, want only Alpha", result.Candidates) + } +} + +func TestTokenizePreservesUnicodeAndIdentifierBoundaries(t *testing.T) { + got := tokenize("ProcessPayment café déjàVu HTTPServer _Leading trailing_ x") + want := []string{"process", "payment", "café", "déjà", "vu", "httpserver", "_leading", "trailing", "x"} + if !reflect.DeepEqual(got, want) { + t.Fatalf("tokenize() = %#v, want %#v", got, want) + } +} + +func TestSearchIsDeterministicAndReusesCachedFragments(t *testing.T) { + cache := NewCache() + key := Key{Workspace: "repo", Scope: "./...", Build: "go1.25", Provider: "gopls"} + files := []File{{Path: "a.go", Digest: "a-v1", Contents: []byte("package fixture\n\nfunc Alpha() {}\n")}, {Path: "b.go", Digest: "b-v1", Contents: []byte("package fixture\n\nfunc Beta() {}\n")}} + + first, err := cache.Search(context.Background(), key, files, "function", 10) + if err != nil { + t.Fatal(err) + } + second, err := cache.Search(context.Background(), key, files, "function", 10) + if err != nil { + t.Fatal(err) + } + if !reflect.DeepEqual(first.Candidates, second.Candidates) { + t.Fatalf("repeated Search() differs:\nfirst=%#v\nsecond=%#v", first.Candidates, second.Candidates) + } + if len(cache.entries) != 2 { + t.Fatalf("cached entries = %d, want 2", len(cache.entries)) + } +} + +func TestSearchProfiledReportsColdAndWarmWorkWithoutChangingResults(t *testing.T) { + cache := NewCache() + key := Key{Workspace: "repo", Scope: "./...", Build: "go1.25", Provider: "gopls"} + files := []File{ + {Path: "a.go", Digest: "a-v1", Contents: []byte("package fixture\n\nfunc AlphaWorker() {}\n")}, + {Path: "b.go", Digest: "b-v1", Contents: []byte("package fixture\n\nfunc BetaWorker() {}\n")}, + } + + cold, coldProfile, err := cache.SearchProfiled(context.Background(), key, files, "worker", 10) + if err != nil { + t.Fatal(err) + } + warm, warmProfile, err := cache.SearchProfiled(context.Background(), key, files, "worker", 10) + if err != nil { + t.Fatal(err) + } + plain, err := cache.Search(context.Background(), key, files, "worker", 10) + if err != nil { + t.Fatal(err) + } + if !reflect.DeepEqual(cold.Candidates, warm.Candidates) || !reflect.DeepEqual(cold.Candidates, plain.Candidates) { + t.Fatalf("profiled Search changed candidates:\ncold=%#v\nwarm=%#v\nplain=%#v", cold.Candidates, warm.Candidates, plain.Candidates) + } + if coldProfile.FileVisits != 4 || coldProfile.FilesParsed != 2 || coldProfile.CacheHits != 2 { + t.Fatalf("cold profile = %+v, want two file passes, two parses, and two hits", coldProfile) + } + if warmProfile.FileVisits != 4 || warmProfile.FilesParsed != 0 || warmProfile.CacheHits != 4 { + t.Fatalf("warm profile = %+v, want two file passes and four cache hits", warmProfile) + } + if coldProfile.SearchDuration <= 0 || coldProfile.ParseDuration <= 0 || warmProfile.SearchDuration <= 0 || warmProfile.ParseDuration != 0 { + t.Fatalf("invalid cold/warm durations: cold=%+v warm=%+v", coldProfile, warmProfile) + } +} + +func TestSearchWithTextProfiledAddsBoundedTextCandidatesWithSameScorer(t *testing.T) { + goFiles := []File{{ + Path: "position.go", + Contents: []byte("package fixture\n\n// Position converts source offsets.\nfunc Position() {}\n"), + }} + textFiles := []File{{ + Path: "contracts.md", + Contents: []byte("Public locations use one-based UTF-8 byte columns. UTF-16 positions exist only inside the pinned LSP adapter.\n"), + }} + key := Key{Workspace: "repo", Scope: "commit", Build: "retrievalbench", Provider: "lexical-declaration-index"} + query := "UTF-16 LSP adapter positions" + + goOnly, err := NewCache().Search(context.Background(), key, goFiles, query, 10) + if err != nil { + t.Fatal(err) + } + mixed, profile, err := NewCache().SearchWithTextProfiled(context.Background(), key, goFiles, textFiles, query, 10) + if err != nil { + t.Fatal(err) + } + if len(goOnly.Candidates) != 0 { + t.Fatalf("Go-only candidates = %#v, want no match for text-only terms", goOnly.Candidates) + } + if len(mixed.Candidates) != 1 || mixed.Candidates[0].Path != "contracts.md" || mixed.Candidates[0].Line != 1 || mixed.Candidates[0].Kind != "text.line" { + t.Fatalf("mixed candidates = %#v, want the source-linked Markdown line", mixed.Candidates) + } + if mixed.CandidateCount != 1 || mixed.TextIndexedFiles != 1 || mixed.TextSkippedFiles != 0 || !mixed.Complete { + t.Fatalf("mixed result = %+v, want one complete text candidate", mixed) + } + if profile.FileVisits != 3 { + t.Fatalf("mixed profile file visits = %d, want two Go passes and one text pass", profile.FileVisits) + } +} + +func TestSearchWithTextProfiledMarksInvalidTextIncomplete(t *testing.T) { + files := []File{ + {Path: "bad.md", Contents: []byte{0xff}}, + {Path: "misclassified.go", Contents: []byte("package fixture\n\nfunc NotText() {}\n")}, + } + result, _, err := NewCache().SearchWithTextProfiled(context.Background(), Key{Workspace: "repo"}, nil, files, "needle", 10) + if err != nil { + t.Fatal(err) + } + if result.Complete || result.TextIndexedFiles != 0 || result.TextSkippedFiles != 2 { + t.Fatalf("invalid text result = %+v, want incomplete with two skipped files", result) + } +} + +func TestSearchWithTextProfiledCapsLineCandidateExpansion(t *testing.T) { + contents := bytes.Repeat([]byte("needle\n"), MaximumTextLineFragments+1) + result, _, err := NewCache().SearchWithTextProfiled(context.Background(), Key{Workspace: "repo"}, nil, []File{{ + Path: "large.md", Contents: contents, + }}, "needle", 10) + if err != nil { + t.Fatal(err) + } + if result.Complete || result.TextIndexedFiles != 0 || result.TextSkippedFiles != 1 || result.TextIndexedFragments != 0 || len(result.Candidates) != 0 { + t.Fatalf("capped text result = %+v, want the oversized file skipped as incomplete", result) + } +} + +func TestSearchStreamingTopKMatchesFullRanking(t *testing.T) { + files := []File{ + {Path: "z.go", Contents: []byte("package fixture\n\nfunc ProcessRequest() {}\nfunc ProcessPayment() {}\n")}, + {Path: "a.go", Contents: []byte("package fixture\n\nfunc ProcessTransfer() {}\nfunc HandlePayment() {}\n")}, + {Path: "m.go", Contents: []byte("package fixture\n\nfunc ProcessRefund() {}\n")}, + } + queryTerms := tokenize("process payment") + indexedFiles := make([]indexedFile, 0, len(files)) + documentFrequency := make(map[string]int) + totalFragments, totalLength := 0, 0 + for _, file := range files { + indexed, complete := parseFile(file.Path, file.Contents) + if !complete { + t.Fatalf("parseFile(%q) was incomplete", file.Path) + } + indexedFiles = append(indexedFiles, indexed) + for _, item := range indexed.fragments { + totalFragments++ + totalLength += item.length + for term := range item.terms { + documentFrequency[term]++ + } + } + } + averageLength := float64(totalLength) / float64(totalFragments) + all := make([]Candidate, 0) + for _, indexed := range indexedFiles { + for _, item := range indexed.fragments { + candidateScore := score(item, queryTerms, documentFrequency, totalFragments, averageLength) + if candidateScore <= 0 { + continue + } + all = append(all, Candidate{ + Path: item.path, Line: item.line, Column: item.column, + Name: item.name, Qualified: item.qualified, Kind: item.kind, + Package: item.packageName, Score: candidateScore, + }) + } + } + sort.SliceStable(all, func(i, j int) bool { return candidateRanksBefore(all[i], all[j]) }) + + result, err := NewCache().Search(context.Background(), Key{Workspace: "repo"}, files, "process payment", 3) + if err != nil { + t.Fatal(err) + } + want := all + if len(want) > 3 { + want = want[:3] + } + if !reflect.DeepEqual(result.Candidates, want) { + t.Fatalf("streaming top-k differs from full ranking:\ngot=%#v\nwant=%#v", result.Candidates, want) + } + if result.Truncated != (len(all) > 3) { + t.Fatalf("truncated = %t, want %t for %d candidates", result.Truncated, len(all) > 3, len(all)) + } + for _, candidate := range result.Candidates { + if math.IsNaN(candidate.Score) || math.IsInf(candidate.Score, 0) { + t.Fatalf("candidate has non-finite score: %#v", candidate) + } + } +} + +func TestSearchInvalidatesChangedDigest(t *testing.T) { + cache := NewCache() + key := Key{Workspace: "repo", Scope: "./...", Build: "go1.25", Provider: "gopls"} + oldFile := File{Path: "value.go", Digest: "value-v1", Contents: []byte("package fixture\n\nfunc OldValue() {}\n")} + old, err := cache.Search(context.Background(), key, []File{oldFile}, "old", 10) + if err != nil { + t.Fatal(err) + } + if len(old.Candidates) != 1 || old.Candidates[0].Name != "OldValue" { + t.Fatalf("old Search() = %#v", old.Candidates) + } + + current, err := cache.Search(context.Background(), key, []File{{Path: "value.go", Digest: "value-v2", Contents: []byte("package fixture\n\nfunc NewValue() {}\n")}}, "old", 10) + if err != nil { + t.Fatal(err) + } + if len(current.Candidates) != 0 { + t.Fatalf("changed Search() returned stale candidates = %#v", current.Candidates) + } +} diff --git a/validation/cmd/retrievalbench/README.md b/validation/cmd/retrievalbench/README.md new file mode 100644 index 0000000..558c756 --- /dev/null +++ b/validation/cmd/retrievalbench/README.md @@ -0,0 +1,137 @@ +# Retrieval benchmark + +`retrievalbench` screens the existing declaration retrieval kernel on immutable +Go repository snapshots. It runs no model and makes no claim about full +`go_context` output, semantic resolution, or arbitrary-size repository support. + +## Run + +```sh +go run ./validation/cmd/retrievalbench \ + --manifest validation/retrieval/heldout-v1/medium-agentic-go.json \ + --repo /path/to/agentic-go-clone \ + --source-repo . \ + --gopls /path/to/gopls \ + --out /tmp/agentic-go-retrieval-medium.json +``` + +For the private text-candidate ablation, use a fresh reviewed manifest and add +`--text-candidate-ablation`: + +```sh +go run ./validation/cmd/retrievalbench \ + --manifest validation/retrieval/heldout-v2/medium-agentic-go-text.json \ + --repo /path/to/agentic-go-clone \ + --source-repo . \ + --text-candidate-ablation \ + --out /tmp/agentic-go-retrieval-text-v2.json +``` + +The ablation is evaluation-only. It captures at most 100,000 supported text +files, 1 MiB per file, and 64 MiB total; any omitted content makes text +candidate metrics partial. It adds line-anchored text fragments to Go +declaration anchors and uses the existing BM25 scorer over the mixed pool. The +ablation emits at most 50,000 eligible text-line fragments per query. If that +candidate cap is reached, candidate-pool recall is partial. The +live intelligence path continues to call Go-only `SearchProfiled`; this flag +does not change MCP schemas or runtime behavior. The report records the text +capture counts and limits, mixed top-10 metrics, and a separate query-matched +candidate-pool recall audit with a 10,000-candidate cap. Pool recall is exact +only when the result, archive, and bounded text capture are complete. A capped +pool's recall is a lower bound and is labeled partial. + +The `--repo` clone is used only to validate repository identity and read the +pinned objects. The harness confirms the full commit and tree IDs, then exports +that commit with `git archive`; dirty checkout content is never indexed. For a +local repository with no `origin`, set `repository_id` to +`local/`. The manifest SHA-256, source HEAD, source +dirty-diff SHA-256, and evaluated retrieval source SHA-256 are recorded in the +report. Keep report files outside the source checkout so the output does not +change the dirty-diff fingerprint. + +The manifest is one repository per file. Its strict JSON form is specified in +[`manifest.schema.json`](manifest.schema.json). Each gold span is a reviewed, +one-based inclusive line range in the pinned tree. Allowed evidence types are +`declaration`, `caller`, `test`, `documentation`, and `configuration`. Optional +`gopls_query` gives `workspace/symbol` a concept anchor; when absent, the +question text is passed verbatim. + +## Compared workflows + +Agentic Go calls the current `retrieval.Cache.SearchProfiled` with the complete +archived Go source set. It records independent cache instances for cold samples +and repeats warm samples against the last cold cache for each question and +limit (`5`, `10`). A bounded LRU can evict entries between calls; cache hits and +files reparsed are reported, so “warm” means the same cache instance, not a +claim that every file remained cached. Its per-query profile separates file +parsing, candidate aggregation, ranking, and total search time. + +The native baseline runs one fixed `rg` search over these supported paths: +`.go`, `.md`, `.rst`, `.txt`, `.yaml`, `.yml`, `.json`, `.toml`, `.proto`, +`.mod`, `.sum`, `.work`, `Makefile`, `GNUmakefile`, `Dockerfile`, +`Containerfile`, `.gitignore`, `.editorconfig`, and `.gitattributes`. The source +extensions are matched case-insensitively. Query terms use the same Unicode, +underscore, and lower-to-upper camel-case tokenization as the current retrieval +package. Ripgrep receives the unique tokens in sorted order as fixed-string, +case-insensitive alternatives. Matching lines rank by distinct query-token +count descending, total token occurrences descending, repository-relative path +ascending, then line ascending. The report includes the exact command template, +globs, tokenizer description, and per-question tokens. + +If `--gopls` is supplied, it must report the repository-pinned gopls `v0.21.0`. +The optional arm runs `workspace/symbol`; initialization and per-question query +time are separate. The provider's result cap and exhaustive-coverage behavior +are not established by this harness, so its results are always marked +`unknown_completeness`, even when each request succeeds. The `rg` arm is a +fixed token-overlap workflow, not a complete native-agent baseline: the timed +package inventory has no query-level relevance score, and no `go doc` or +interactive native-agent workflow is measured. Do not use these component +scores alone to claim advantage over native tools. Without `--gopls`, +gopls is explicitly `unavailable`; the retrieval and ripgrep arms still run. +The harness also runs a bounded `go list -e -json ./...` package inventory in +the exported snapshot, with `GOWORK=off` and `GOTOOLCHAIN=local`. It counts +package objects and package load errors without retaining their path-bearing +diagnostics. Its status, count, latency, output-cap state, and timeout are +reported separately; it has no query-level gold scoring. A completed status +covers this command's root `./...` pattern, not nested modules or every possible +build-tag configuration. + +## Scoring and limits + +Candidates are source anchors. Agentic Go returns a Go declaration name +coordinate, gopls returns a workspace-symbol coordinate, and ripgrep returns a +matching source line. A candidate is relevant when its path and line fall +inside a reviewed gold span. Gold spans for declarations must include the +declaration-name line because the candidate is anchored there; a span that +starts inside a declaration body would unfairly mark that declaration missed. +Recall counts gold spans found at least once. Precision uses a fixed +denominator of 5 or 10, so missing result slots count as non-relevant. MRR is +reciprocal rank of the first relevant anchor, capped at the same cutoff. The +report includes per-query metrics, aggregate macro and micro values, and missed +spans grouped by evidence type at both cutoffs. + +File and byte coverage distinguishes all regular committed files, supported +source files expected from the Git tree, extracted Go files and supported +text, rejected binary or non-UTF-8 source, symlinks, and ignored archive +entries. Extracted blobs are checked against their pinned Git object IDs. +Files omitted by `export-ignore` and content changed by `export-subst` are +reported and excluded; such source coverage makes relevance metrics partial. +The existing retrieval kernel indexes +Go declarations only; `text_indexed_by_retrieval` is therefore zero. Heap +statistics are sampled Go runtime heap values; they are not process RSS and do +not include the separate gopls process. + +When text ablation is enabled, its candidate index is reported independently +from `text_indexed_by_retrieval`, which continues to describe the live Go-only +path. Candidate-pool recall measures whether each reviewed gold span appears +anywhere among the query-positive candidates before the top-k cutoff; top-5 +and top-10 ranking remain separate metrics. This ablation tests text candidate +coverage on the pinned corpus. It is not a native-agent comparison or a +product-value result. + +The exporter has explicit limits for total eligible-source bytes and bytes per +file. Ripgrep also has output-byte, matching-line, and per-command time limits. +If one of those bounds truncates an arm, the report marks that arm partial and +does not count its incomplete query in complete-query aggregates. These are +screening limits, not support promises. No database, result cache, persistence, +or branch-selection behavior is introduced by this harness. diff --git a/validation/cmd/retrievalbench/main.go b/validation/cmd/retrievalbench/main.go new file mode 100644 index 0000000..926a66a --- /dev/null +++ b/validation/cmd/retrievalbench/main.go @@ -0,0 +1,56 @@ +// retrievalbench screens Agentic Go's declaration retrieval against a fixed +// ripgrep workflow and, optionally, the repository's pinned gopls provider. +package main + +import ( + "context" + "flag" + "fmt" + "os" + "os/signal" + "syscall" + "time" + + "github.com/agentic-mcps/go/validation/internal/retrievalstudy" +) + +func main() { + var options retrievalstudy.Options + flag.StringVar(&options.Manifest, "manifest", "", "versioned JSON manifest with exact repository commit and reviewed gold spans") + flag.StringVar(&options.RepositoryPath, "repo", "", "local Git clone whose origin and pinned commit are validated") + flag.StringVar(&options.SourceRepositoryPath, "source-repo", ".", "checkout of the harness and retrieval implementation to fingerprint") + flag.StringVar(&options.Output, "out", "", "result JSON output path") + flag.StringVar(&options.GoplsBinary, "gopls", "", "optional path to the pinned gopls v0.21.0 binary") + flag.BoolVar(&options.TextAblation, "text-candidate-ablation", false, "run a bounded private text-line candidate augmentation with the existing scorer") + flag.IntVar(&options.Repetitions, "repetitions", 3, "cold and warm timing samples per question and cutoff (1-20)") + flag.DurationVar(&options.Timeout, "timeout", 5*time.Minute, "per-command and per-search timeout") + flag.Int64Var(&options.MaxSourceBytes, "max-source-bytes", 2<<30, "maximum extracted supported source bytes") + flag.Int64Var(&options.MaxFileBytes, "max-file-bytes", 256<<20, "maximum extracted bytes for one supported source file") + flag.Int64Var(&options.RGOutputBytes, "max-rg-output-bytes", 64<<20, "maximum captured ripgrep output bytes per query") + flag.IntVar(&options.RGLineLimit, "max-rg-lines", 100000, "maximum matching lines captured per ripgrep query") + flag.Usage = func() { + fmt.Fprintln(os.Stderr, "Usage: retrievalbench --manifest study.json --repo /path/to/clone --out result.json [--gopls /path/to/gopls]") + fmt.Fprintln(os.Stderr, "Exports only the manifest's exact Git commit with git archive; dirty checkout files are ignored.") + fmt.Fprintln(os.Stderr, "Gold spans are one-based repository-relative line ranges. Metrics use declaration anchors for Agentic Go and matching lines for rg.") + fmt.Fprintln(os.Stderr, "Supported paths: Go, Markdown, reStructuredText, text, YAML, JSON, TOML, proto, Go mod/sum, and common build metadata files.") + fmt.Fprintln(os.Stderr, "The optional gopls baseline requires v0.21.0; it runs workspace/symbol and reports initialization separately.") + fmt.Fprintln(os.Stderr, "A timed go list -e -json ./... package inventory runs on the archive; it has no gold relevance score.") + fmt.Fprintln(os.Stderr, "Archive omissions and transformed blobs are detected; incomplete source coverage makes retrieval scores partial.") + fmt.Fprintln(os.Stderr, "The report includes the source HEAD, source dirty-diff SHA-256, manifest SHA-256, and retrieval source SHA-256.") + fmt.Fprintln(os.Stderr, "This screens the candidate-retrieval kernel; it does not claim complete go_context coverage or semantic resolution.") + fmt.Fprintln(os.Stderr, "--text-candidate-ablation is evaluation-only; it does not change the public MCP contract or live retrieval path.") + flag.PrintDefaults() + } + flag.Parse() + if flag.NArg() != 0 { + fmt.Fprintln(os.Stderr, "retrievalbench accepts no positional arguments") + os.Exit(2) + } + ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM) + defer stop() + if err := retrievalstudy.Execute(ctx, options); err != nil { + fmt.Fprintln(os.Stderr, "retrievalbench:", err) + os.Exit(1) + } + fmt.Fprintln(os.Stdout, "retrieval report written") +} diff --git a/validation/cmd/retrievalbench/manifest.schema.json b/validation/cmd/retrievalbench/manifest.schema.json new file mode 100644 index 0000000..8da107b --- /dev/null +++ b/validation/cmd/retrievalbench/manifest.schema.json @@ -0,0 +1,122 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://agentic-go.dev/schemas/retrieval-study-v1.json", + "title": "Agentic Go retrieval study manifest v1", + "type": "object", + "additionalProperties": false, + "required": ["version", "repository_id", "commit", "tree", "stratum", "queries"], + "properties": { + "version": { + "const": "agentic-go.retrieval-study/v1" + }, + "repository_id": { + "type": "string", + "minLength": 3, + "maxLength": 512, + "pattern": "^[a-z0-9][a-z0-9.:-]*/[a-z0-9_.-]+(/[a-z0-9_.-]+)*$", + "description": "Canonical lowercase host/owner/repository identity; for a repository without an origin use local/." + }, + "commit": { + "type": "string", + "pattern": "^(?:[0-9a-f]{40}|[0-9a-f]{64})$" + }, + "tree": { + "type": "string", + "pattern": "^(?:[0-9a-f]{40}|[0-9a-f]{64})$" + }, + "stratum": { + "enum": ["small", "medium", "large"] + }, + "queries": { + "type": "array", + "minItems": 1, + "items": { + "$ref": "#/$defs/query" + } + } + }, + "$defs": { + "query": { + "type": "object", + "additionalProperties": false, + "required": ["id", "query", "gold"], + "properties": { + "id": { + "type": "string", + "minLength": 1, + "maxLength": 128 + }, + "query": { + "type": "string", + "minLength": 1, + "maxLength": 4096 + }, + "gopls_query": { + "type": "string", + "maxLength": 4096, + "description": "Optional concept anchor supplied to workspace/symbol; when omitted, the question text is used verbatim." + }, + "gold": { + "type": "array", + "minItems": 1, + "items": { + "$ref": "#/$defs/goldSpan" + } + } + } + }, + "goldSpan": { + "type": "object", + "additionalProperties": false, + "required": ["type", "path", "start_line", "end_line"], + "properties": { + "type": { + "enum": ["declaration", "caller", "test", "documentation", "configuration"] + }, + "path": { + "type": "string", + "minLength": 1, + "pattern": "^(?!/)(?!.*(?:^|/)\\.\\.(?:/|$))[^\\\\]+$", + "description": "Slash-separated path relative to the pinned Git tree." + }, + "start_line": { + "type": "integer", + "minimum": 1 + }, + "end_line": { + "type": "integer", + "minimum": 1 + }, + "label": { + "type": "string", + "maxLength": 512 + } + } + } + }, + "examples": [ + { + "version": "agentic-go.retrieval-study/v1", + "repository_id": "github.com/example/project", + "commit": "0000000000000000000000000000000000000000", + "tree": "1111111111111111111111111111111111111111", + "stratum": "small", + "queries": [ + { + "id": "find-retry-policy", + "query": "Where is retry backoff configured?", + "gopls_query": "Retry Backoff", + "gold": [ + { + "type": "declaration", + "path": "retry/policy.go", + "start_line": 10, + "end_line": 24, + "label": "Policy declaration and its setup path" + } + ] + } + ] + } + ] +} diff --git a/validation/internal/retrievalstudy/archive.go b/validation/internal/retrievalstudy/archive.go new file mode 100644 index 0000000..df51409 --- /dev/null +++ b/validation/internal/retrievalstudy/archive.go @@ -0,0 +1,564 @@ +package retrievalstudy + +import ( + "archive/tar" + "bytes" + "context" + "crypto/sha1" + "crypto/sha256" + "encoding/hex" + "errors" + "fmt" + "io" + "net/url" + "os" + "os/exec" + "path" + "path/filepath" + "sort" + "strconv" + "strings" + "sync" + "time" + "unicode/utf8" + + "github.com/agentic-mcps/go/internal/intelligence/retrieval" +) + +const ( + maximumTextCandidateFiles = 100_000 + maximumTextCandidateFileSize = 1 << 20 + maximumTextCandidateBytes = 64 << 20 + maximumTextCandidateFragments = retrieval.MaximumTextLineFragments +) + +type sourceMeta struct { + lines int + goFile bool +} + +type archivedSource struct { + workspace string + goFiles []retrieval.File + textFiles []retrieval.File + textCandidateIndex TextCandidateIndexCoverage + meta map[string]sourceMeta + expected map[string]archiveExpected + coverage Coverage +} + +type archiveExpected struct { + objectID string + size int64 +} + +type archiveLimits struct { + maxSourceBytes int64 + maxFileBytes int64 + timeout time.Duration + indexTextCandidates bool +} + +// ExportCommit uses git archive for the exact manifest commit. It never reads +// from the checkout's working tree, index, or current branch. +func ExportCommit(ctx context.Context, repositoryPath, workspace string, manifest Manifest, limits archiveLimits) (Repository, archivedSource, error) { + repositoryRoot, err := gitText(ctx, limits.timeout, repositoryPath, "rev-parse", "--show-toplevel") + if err != nil { + return Repository{}, archivedSource{}, fmt.Errorf("locating Git repository: %w", err) + } + repositoryRoot = strings.TrimSpace(repositoryRoot) + remotes, err := gitText(ctx, limits.timeout, repositoryRoot, "remote") + if err != nil { + return Repository{}, archivedSource{}, fmt.Errorf("reading Git remotes: %w", err) + } + hasOrigin := false + for _, remoteName := range strings.Fields(remotes) { + if remoteName == "origin" { + hasOrigin = true + break + } + } + repositoryID := "" + if hasOrigin { + remote, remoteErr := gitText(ctx, limits.timeout, repositoryRoot, "remote", "get-url", "origin") + if remoteErr != nil { + return Repository{}, archivedSource{}, fmt.Errorf("reading origin repository identity: %w", remoteErr) + } + repositoryID, err = canonicalRepositoryID(strings.TrimSpace(remote)) + if err != nil { + return Repository{}, archivedSource{}, err + } + if repositoryID != manifest.RepositoryID { + return Repository{}, archivedSource{}, fmt.Errorf("manifest repository_id %q does not match origin %q", manifest.RepositoryID, repositoryID) + } + } else if strings.HasPrefix(manifest.RepositoryID, "local/") && filepath.Base(repositoryRoot) == strings.TrimPrefix(manifest.RepositoryID, "local/") { + repositoryID = manifest.RepositoryID + } else { + return Repository{}, archivedSource{}, fmt.Errorf("repository has no origin; repository_id must use local/") + } + resolvedCommit, err := gitText(ctx, limits.timeout, repositoryRoot, "rev-parse", "--verify", manifest.Commit+"^{commit}") + if err != nil { + return Repository{}, archivedSource{}, fmt.Errorf("resolving pinned commit: %w", err) + } + resolvedCommit = strings.TrimSpace(resolvedCommit) + if resolvedCommit != manifest.Commit { + return Repository{}, archivedSource{}, fmt.Errorf("manifest commit did not resolve to the exact object id") + } + tree, err := gitText(ctx, limits.timeout, repositoryRoot, "rev-parse", "--verify", manifest.Commit+"^{tree}") + if err != nil { + return Repository{}, archivedSource{}, fmt.Errorf("resolving pinned tree: %w", err) + } + resolvedTree := strings.TrimSpace(tree) + if resolvedTree != manifest.Tree { + return Repository{}, archivedSource{}, fmt.Errorf("manifest tree did not match the exact pinned commit tree") + } + repository := Repository{ID: repositoryID, Commit: resolvedCommit, Tree: resolvedTree} + patterns, coverage, expected, err := archiveSelection(ctx, repositoryRoot, manifest.Commit, limits.timeout) + if err != nil { + return Repository{}, archivedSource{}, err + } + source, err := extractArchive(ctx, repositoryRoot, workspace, manifest.Commit, patterns, expected, coverage, limits) + if err != nil { + return Repository{}, archivedSource{}, err + } + return repository, source, nil +} + +func canonicalRepositoryID(remote string) (string, error) { + var host, repositoryPath string + if strings.Contains(remote, "://") { + parsed, err := url.Parse(remote) + if err != nil || parsed.Host == "" { + return "", fmt.Errorf("origin URL cannot be normalized to host/owner/repository") + } + host, repositoryPath = parsed.Host, parsed.Path + } else { + colon := strings.Index(remote, ":") + if colon < 0 { + return "", fmt.Errorf("origin must be an HTTPS, SSH, or SCP-style remote URL") + } + hostPart := remote[:colon] + if at := strings.LastIndex(hostPart, "@"); at >= 0 { + hostPart = hostPart[at+1:] + } + host, repositoryPath = hostPart, remote[colon+1:] + } + host = strings.ToLower(strings.TrimSpace(host)) + repositoryPath = strings.Trim(strings.TrimSpace(repositoryPath), "/") + repositoryPath = strings.TrimSuffix(repositoryPath, ".git") + if host == "" || repositoryPath == "" { + return "", fmt.Errorf("origin URL cannot be normalized to host/owner/repository") + } + return strings.ToLower(host + "/" + repositoryPath), nil +} + +func gitText(parent context.Context, timeout time.Duration, repositoryPath string, arguments ...string) (string, error) { + ctx, cancel := context.WithTimeout(parent, timeout) + defer cancel() + command := exec.CommandContext(ctx, "git", append([]string{"-C", repositoryPath}, arguments...)...) + output, err := command.Output() + if err != nil { + if ctx.Err() != nil { + return "", ctx.Err() + } + return "", fmt.Errorf("git %s failed: %w", strings.Join(arguments, " "), err) + } + return string(output), nil +} + +func archiveSelection(parent context.Context, repositoryRoot, commit string, timeout time.Duration) ([]string, Coverage, map[string]archiveExpected, error) { + output, err := gitText(parent, timeout, repositoryRoot, "ls-tree", "-r", "-l", "-z", "--full-tree", commit) + if err != nil { + return nil, Coverage{}, nil, fmt.Errorf("enumerating pinned Git tree: %w", err) + } + patterns := make(map[string]struct{}) + expected := make(map[string]archiveExpected) + coverage := Coverage{SourceArchiveComplete: true} + for _, record := range bytes.Split([]byte(output), []byte{0}) { + if len(record) == 0 { + continue + } + separator := bytes.IndexByte(record, '\t') + if separator < 0 { + return nil, Coverage{}, nil, fmt.Errorf("Git tree entry has an invalid record") + } + metadata := strings.Fields(string(record[:separator])) + if len(metadata) < 3 { + return nil, Coverage{}, nil, fmt.Errorf("Git tree entry has incomplete metadata") + } + mode, objectType := metadata[0], metadata[1] + name := string(record[separator+1:]) + if mode == "120000" { + if supportedTextPath(name) { + coverage.SymlinksNotIndexed++ + markArchiveIncomplete(&coverage, "one or more supported source paths are symlinks and were not indexed") + } + continue + } + if objectType != "blob" || (mode != "100644" && mode != "100755") { + if supportedTextPath(name) { + coverage.OtherArchiveEntriesIgnored++ + markArchiveIncomplete(&coverage, "one or more supported source paths are non-regular Git entries and were not indexed") + } + continue + } + if len(metadata) < 4 { + return nil, Coverage{}, nil, fmt.Errorf("Git tree entry has no blob size") + } + size, parseErr := strconv.ParseInt(metadata[3], 10, 64) + if parseErr != nil || size < 0 { + return nil, Coverage{}, nil, fmt.Errorf("Git tree entry has an invalid blob size") + } + coverage.TrackedRegularFiles++ + coverage.TrackedRegularBytes += size + if !supportedTextPath(name) { + continue + } + if !utf8.ValidString(name) { + coverage.RejectedTextFiles++ + coverage.RejectedTextBytes += size + markArchiveIncomplete(&coverage, "one or more supported source paths are not valid UTF-8 and were not indexed") + continue + } + coverage.ExpectedSupportedSourceFiles++ + coverage.ExpectedSupportedSourceBytes += size + expected[name] = archiveExpected{objectID: metadata[2], size: size} + patterns[archivePattern(name)] = struct{}{} + } + result := make([]string, 0, len(patterns)) + for pattern := range patterns { + result = append(result, pattern) + } + sort.Strings(result) + if len(result) == 0 { + return nil, Coverage{}, nil, fmt.Errorf("pinned Git tree contains no supported Go or text source files") + } + return result, coverage, expected, nil +} + +func archivePattern(name string) string { + extension := path.Ext(name) + if extension != "" && extension != ".gitignore" && extension != ".editorconfig" && extension != ".gitattributes" { + return "(glob)**/*" + extension + } + return "(glob)**/" + path.Base(name) +} + +func extractArchive(parent context.Context, repositoryRoot, workspace, commit string, patterns []string, expected map[string]archiveExpected, coverage Coverage, limits archiveLimits) (archivedSource, error) { + ctx, cancel := context.WithTimeout(parent, limits.timeout) + defer cancel() + arguments := []string{"-C", repositoryRoot, "archive", "--format=tar", "--prefix=repo/", commit, "--"} + for _, pattern := range patterns { + arguments = append(arguments, ":"+pattern) + } + command := exec.CommandContext(ctx, "git", arguments...) + stdout, err := command.StdoutPipe() + if err != nil { + return archivedSource{}, fmt.Errorf("opening git archive stream: %w", err) + } + var stderr limitedBuffer + stderr.limit = 64 << 10 + command.Stderr = &stderr + if err := command.Start(); err != nil { + return archivedSource{}, fmt.Errorf("starting git archive: %w", err) + } + source, extractErr := readArchive(tar.NewReader(stdout), workspace, expected, coverage, limits) + if extractErr != nil { + _ = command.Process.Kill() + } + waitErr := command.Wait() + if extractErr != nil { + return archivedSource{}, extractErr + } + if waitErr != nil { + if ctx.Err() != nil { + return archivedSource{}, ctx.Err() + } + return archivedSource{}, fmt.Errorf("git archive failed: %w: %s", waitErr, strings.TrimSpace(stderr.String())) + } + return source, nil +} + +func readArchive(reader *tar.Reader, workspace string, expected map[string]archiveExpected, coverage Coverage, limits archiveLimits) (archivedSource, error) { + source := archivedSource{workspace: workspace, goFiles: []retrieval.File{}, meta: make(map[string]sourceMeta), expected: expected, coverage: coverage} + if limits.indexTextCandidates { + source.textFiles = []retrieval.File{} + source.textCandidateIndex = TextCandidateIndexCoverage{ + Status: "complete", MaximumFiles: maximumTextCandidateFiles, + MaximumFileBytes: maximumTextCandidateFileSize, MaximumTotalBytes: maximumTextCandidateBytes, + MaximumFragments: maximumTextCandidateFragments, + } + } + seen := make(map[string]bool, len(expected)) + var extractedSourceBytes int64 + for { + header, nextErr := reader.Next() + if nextErr == io.EOF { + break + } + if nextErr != nil { + return archivedSource{}, fmt.Errorf("reading Git archive: %w", nextErr) + } + if header.Typeflag == tar.TypeXGlobalHeader || header.Typeflag == tar.TypeXHeader { + continue + } + name, isDirectory, pathErr := archivePath(header.Name) + if errors.Is(pathErr, errArchivePathEncoding) { + continue + } + if pathErr != nil { + return archivedSource{}, pathErr + } + if name == "" && isDirectory { + continue + } + if isDirectory || header.Typeflag == tar.TypeDir { + if err := os.MkdirAll(filepath.Join(workspace, filepath.FromSlash(name)), 0o700); err != nil { + return archivedSource{}, fmt.Errorf("creating archive directory: %w", err) + } + continue + } + if header.Typeflag == tar.TypeSymlink || header.Typeflag == tar.TypeLink { + continue + } + if header.Typeflag != tar.TypeReg && header.Typeflag != tar.TypeRegA { + continue + } + if !supportedTextPath(name) { + continue + } + expectedFile, selected := expected[name] + if !selected { + source.coverage.OtherArchiveEntriesIgnored++ + markArchiveIncomplete(&source.coverage, "git archive yielded a supported source path absent from the pinned tree selection") + continue + } + seen[name] = true + if header.Size < 0 || header.Size > limits.maxFileBytes { + return archivedSource{}, fmt.Errorf("source file %q exceeds the per-file byte limit", name) + } + if extractedSourceBytes+header.Size > limits.maxSourceBytes { + return archivedSource{}, fmt.Errorf("supported source exceeds the configured byte limit") + } + extractedSourceBytes += header.Size + contents, readErr := io.ReadAll(io.LimitReader(reader, header.Size+1)) + if readErr != nil { + return archivedSource{}, fmt.Errorf("reading source file %q: %w", name, readErr) + } + if int64(len(contents)) != header.Size { + return archivedSource{}, fmt.Errorf("archive source file %q has an inconsistent size", name) + } + if gitBlobObjectID(contents, len(expectedFile.objectID)) != expectedFile.objectID { + source.coverage.ArchiveTransformedSourceFiles++ + source.coverage.ArchiveTransformedSourceBytes += expectedFile.size + markArchiveIncomplete(&source.coverage, "git archive transformed one or more source blobs; transformed bytes were excluded") + continue + } + if !utf8.Valid(contents) || containsNUL(contents) { + source.coverage.RejectedTextFiles++ + source.coverage.RejectedTextBytes += header.Size + markArchiveIncomplete(&source.coverage, "one or more supported source blobs are binary or not valid UTF-8 and were not indexed") + continue + } + filename := filepath.Join(workspace, filepath.FromSlash(name)) + if err := os.MkdirAll(filepath.Dir(filename), 0o700); err != nil { + return archivedSource{}, fmt.Errorf("creating source directory: %w", err) + } + file, err := os.OpenFile(filename, os.O_CREATE|os.O_EXCL|os.O_WRONLY, 0o600) + if err != nil { + return archivedSource{}, fmt.Errorf("creating archived source file: %w", err) + } + _, copyErr := file.Write(contents) + closeErr := file.Close() + if copyErr != nil { + return archivedSource{}, fmt.Errorf("writing archived source file: %w", copyErr) + } + if closeErr != nil { + return archivedSource{}, fmt.Errorf("closing archived source file: %w", closeErr) + } + lineCount := countLines(contents) + source.coverage.SupportedSourceFiles++ + source.coverage.SupportedSourceBytes += header.Size + if strings.HasSuffix(strings.ToLower(name), ".go") { + digest := sha256.Sum256(contents) + source.goFiles = append(source.goFiles, retrieval.File{Path: name, Digest: hex.EncodeToString(digest[:]), Contents: contents}) + source.meta[name] = sourceMeta{lines: lineCount, goFile: true} + source.coverage.GoFiles++ + source.coverage.GoBytes += header.Size + } else { + source.meta[name] = sourceMeta{lines: lineCount} + source.coverage.SupportedTextFiles++ + source.coverage.SupportedTextBytes += header.Size + if limits.indexTextCandidates { + source.textCandidateIndex.SupportedFiles++ + source.textCandidateIndex.SupportedBytes += header.Size + if source.textCandidateIndex.Status == "complete" && + len(source.textFiles) < maximumTextCandidateFiles && + header.Size <= maximumTextCandidateFileSize && + source.textCandidateIndex.IndexedBytes+header.Size <= maximumTextCandidateBytes { + digest := sha256.Sum256(contents) + source.textFiles = append(source.textFiles, retrieval.File{ + Path: name, Digest: hex.EncodeToString(digest[:]), Contents: contents, + }) + source.textCandidateIndex.IndexedFiles++ + source.textCandidateIndex.IndexedBytes += header.Size + } else { + source.textCandidateIndex.Status = "partial" + source.textCandidateIndex.OmittedFiles++ + source.textCandidateIndex.OmittedBytes += header.Size + if source.textCandidateIndex.Reason == "" { + source.textCandidateIndex.Reason = "bounded text candidate capture reached a file-count, per-file, or total-byte limit" + } + } + } + } + } + for name, expectedFile := range expected { + if seen[name] { + continue + } + source.coverage.ArchiveOmittedSourceFiles++ + source.coverage.ArchiveOmittedSourceBytes += expectedFile.size + } + if source.coverage.ArchiveOmittedSourceFiles > 0 { + markArchiveIncomplete(&source.coverage, "git archive omitted one or more supported source blobs, possibly due to export-ignore attributes") + } + if limits.indexTextCandidates && !source.coverage.SourceArchiveComplete && source.textCandidateIndex.Status == "complete" { + source.textCandidateIndex.Status = "partial" + source.textCandidateIndex.Reason = source.coverage.SourceArchiveIncompleteReason + } + sortGoFiles(source.goFiles) + return source, nil +} + +var errArchivePathEncoding = errors.New("Git archive path is not valid UTF-8") + +func archivePath(headerName string) (string, bool, error) { + if headerName == "repo/" || headerName == "repo" { + return "", true, nil + } + if !strings.HasPrefix(headerName, "repo/") { + return "", false, fmt.Errorf("Git archive entry %q escaped its expected prefix", headerName) + } + name := strings.TrimSuffix(strings.TrimPrefix(headerName, "repo/"), "/") + if name == "" { + return "", true, nil + } + if !filepath.IsLocal(filepath.FromSlash(name)) || path.Clean(name) != name { + return "", false, fmt.Errorf("Git archive contains an unsafe path") + } + if !utf8.ValidString(name) { + return "", false, errArchivePathEncoding + } + if !isRegularPath(name) { + return name, true, nil + } + return name, false, nil +} + +func gitBlobObjectID(contents []byte, objectIDLength int) string { + var digest interface { + Write([]byte) (int, error) + Sum([]byte) []byte + } + if objectIDLength == 64 { + digest = sha256.New() + } else { + digest = sha1.New() + } + _, _ = fmt.Fprintf(digest, "blob %d\x00", len(contents)) + _, _ = digest.Write(contents) + return hex.EncodeToString(digest.Sum(nil)) +} + +func markArchiveIncomplete(coverage *Coverage, reason string) { + coverage.SourceArchiveComplete = false + if coverage.SourceArchiveIncompleteReason == "" { + coverage.SourceArchiveIncompleteReason = reason + } +} + +func isRegularPath(name string) bool { + return name != "" && path.Base(name) != "." && path.Base(name) != ".." +} + +func supportedTextPath(name string) bool { + switch strings.ToLower(path.Ext(name)) { + case ".go", ".md", ".rst", ".txt", ".yaml", ".yml", ".json", ".toml", ".proto", ".mod", ".sum", ".work": + return true + } + switch path.Base(name) { + case "Makefile", "GNUmakefile", "Dockerfile", "Containerfile", ".gitignore", ".editorconfig", ".gitattributes": + return true + default: + return false + } +} + +func countLines(contents []byte) int { + if len(contents) == 0 { + return 0 + } + count := 0 + for _, current := range contents { + if current == '\n' { + count++ + } + } + if contents[len(contents)-1] != '\n' { + count++ + } + return count +} + +func containsNUL(contents []byte) bool { + for _, current := range contents { + if current == 0 { + return true + } + } + return false +} + +func sortGoFiles(files []retrieval.File) { + sort.Slice(files, func(i, j int) bool { return files[i].Path < files[j].Path }) +} + +type limitedBuffer struct { + mu sync.Mutex + buffer bytes.Buffer + limit int + truncated bool +} + +func (buffer *limitedBuffer) Write(data []byte) (int, error) { + buffer.mu.Lock() + defer buffer.mu.Unlock() + if buffer.limit <= 0 || buffer.buffer.Len() >= buffer.limit { + buffer.truncated = true + return len(data), nil + } + remaining := buffer.limit - buffer.buffer.Len() + if len(data) > remaining { + _, _ = buffer.buffer.Write(data[:remaining]) + buffer.truncated = true + return len(data), nil + } + return buffer.buffer.Write(data) +} + +func (buffer *limitedBuffer) Len() int { + buffer.mu.Lock() + defer buffer.mu.Unlock() + return buffer.buffer.Len() +} + +func (buffer *limitedBuffer) Bytes() []byte { + buffer.mu.Lock() + defer buffer.mu.Unlock() + return append([]byte(nil), buffer.buffer.Bytes()...) +} + +func (buffer *limitedBuffer) String() string { + return string(buffer.Bytes()) +} diff --git a/validation/internal/retrievalstudy/archive_test.go b/validation/internal/retrievalstudy/archive_test.go new file mode 100644 index 0000000..12d27ca --- /dev/null +++ b/validation/internal/retrievalstudy/archive_test.go @@ -0,0 +1,68 @@ +package retrievalstudy + +import ( + "archive/tar" + "bytes" + "testing" +) + +func TestLimitedBufferCapsCapturedOutputAndReportsTruncation(t *testing.T) { + buffer := limitedBuffer{limit: 10} + if written, err := buffer.Write([]byte("12345678")); err != nil || written != 8 { + t.Fatalf("first write = (%d, %v), want (8, nil)", written, err) + } + if written, err := buffer.Write([]byte("abc")); err != nil || written != 3 { + t.Fatalf("second write = (%d, %v), want (3, nil)", written, err) + } + if got := buffer.Len(); got != 10 { + t.Fatalf("captured bytes = %d, want cap 10", got) + } + if got := string(buffer.Bytes()); got != "12345678ab" { + t.Fatalf("captured output = %q, want %q", got, "12345678ab") + } + if !buffer.truncated { + t.Fatal("truncated = false, want true") + } +} + +func TestLimitedBufferExactlyAtLimitIsNotTruncated(t *testing.T) { + buffer := limitedBuffer{limit: 4} + if written, err := buffer.Write([]byte("four")); err != nil || written != 4 { + t.Fatalf("write = (%d, %v), want (4, nil)", written, err) + } + if buffer.truncated { + t.Fatal("truncated = true, want false when output exactly fits") + } +} + +func TestReadArchiveBoundsTextCandidateCapture(t *testing.T) { + contents := bytes.Repeat([]byte("x"), maximumTextCandidateFileSize+1) + var archive bytes.Buffer + writer := tar.NewWriter(&archive) + if err := writer.WriteHeader(&tar.Header{ + Name: "repo/docs/large.md", Mode: 0o600, Size: int64(len(contents)), Typeflag: tar.TypeReg, + }); err != nil { + t.Fatal(err) + } + if _, err := writer.Write(contents); err != nil { + t.Fatal(err) + } + if err := writer.Close(); err != nil { + t.Fatal(err) + } + expected := map[string]archiveExpected{ + "docs/large.md": {objectID: gitBlobObjectID(contents, 40), size: int64(len(contents))}, + } + source, err := readArchive(tar.NewReader(&archive), t.TempDir(), expected, Coverage{SourceArchiveComplete: true}, archiveLimits{ + maxSourceBytes: int64(len(contents) + 1), maxFileBytes: int64(len(contents) + 1), indexTextCandidates: true, + }) + if err != nil { + t.Fatal(err) + } + if source.textCandidateIndex.Status != "partial" || source.textCandidateIndex.IndexedFiles != 0 || source.textCandidateIndex.OmittedFiles != 1 { + t.Fatalf("text candidate coverage = %+v, want one oversized file omitted", source.textCandidateIndex) + } + if source.textCandidateIndex.OmittedBytes != int64(len(contents)) || len(source.textFiles) != 0 { + t.Fatalf("text capture bytes/files = (%d, %d), want (%d, 0)", source.textCandidateIndex.OmittedBytes, len(source.textFiles), len(contents)) + } +} diff --git a/validation/internal/retrievalstudy/golist.go b/validation/internal/retrievalstudy/golist.go new file mode 100644 index 0000000..268312d --- /dev/null +++ b/validation/internal/retrievalstudy/golist.go @@ -0,0 +1,127 @@ +package retrievalstudy + +import ( + "bytes" + "context" + "encoding/json" + "os" + "os/exec" + "io" + "strings" + "time" +) + +const goListOutputLimit = 16 << 20 + +func inventoryGoPackages(parent context.Context, workspace string, timeout time.Duration) (GoPackageInventory, error) { + result := GoPackageInventory{ + Status: "unavailable", Command: "go list -e -json ./...", TimeoutMS: timeout.Milliseconds(), + OutputLimitBytes: goListOutputLimit, Coverage: "go_list_pattern_only", + } + if err := parent.Err(); err != nil { + return result, err + } + ctx, cancel := context.WithTimeout(parent, timeout) + defer cancel() + command := exec.CommandContext(ctx, "go", "list", "-e", "-json", "./...") + command.Dir = workspace + command.Env = setEnv(os.Environ(), "GOWORK", "off", "GOTOOLCHAIN", "local") + var stdout, stderr limitedBuffer + stdout.limit = goListOutputLimit + stderr.limit = 64 << 10 + command.Stdout = &stdout + command.Stderr = &stderr + started := time.Now() + err := command.Run() + result.LatencyMS = float64(time.Since(started)) / float64(time.Millisecond) + result.OutputBytes = stdout.Len() + result.OutputTruncated = stdout.truncated + var parseErr error + result.PackageCount, result.PackagesWithErrors, parseErr = parseGoPackageInventory(stdout.Bytes()) + if parent.Err() != nil { + return result, parent.Err() + } + if ctx.Err() != nil && ctx.Err() == context.DeadlineExceeded { + result.Status = "timeout" + result.Coverage = "partial_timeout" + result.Error = "go list package inventory exceeded the configured timeout" + return result, nil + } + if err != nil { + result.Status = "partial" + result.Coverage = "partial_command_failure" + result.Error = "go list package inventory exited with an error" + return result, nil + } + if result.OutputTruncated { + result.Status = "partial" + result.Coverage = "partial_output_limit" + result.Error = "go list package inventory output exceeded the capture limit" + return result, nil + } + if parseErr != nil { + result.Status = "partial" + result.Coverage = "partial_invalid_json_output" + result.Error = "go list package inventory emitted incomplete JSON" + return result, nil + } + if result.PackagesWithErrors > 0 { + result.Status = "partial" + result.Coverage = "partial_package_load_errors" + result.Error = "one or more listed packages had load errors" + return result, nil + } + result.Status = "complete" + result.Coverage = "go_list_e_dotdot_json_pattern_completed" + return result, nil +} + +func parseGoPackageInventory(data []byte) (int, int, error) { + type packageError struct{} + var packages, withErrors int + decoder := json.NewDecoder(bytes.NewReader(data)) + for { + var record struct { + ImportPath string `json:"ImportPath"` + Incomplete bool `json:"Incomplete"` + Error *packageError `json:"Error"` + DepsErrors []packageError `json:"DepsErrors"` + } + if err := decoder.Decode(&record); err != nil { + if err == io.EOF { + return packages, withErrors, nil + } + return packages, withErrors, err + } + if record.ImportPath != "" { + packages++ + } + if record.Incomplete || record.Error != nil || len(record.DepsErrors) > 0 { + withErrors++ + } + } +} + +func setEnv(environment []string, replacements ...string) []string { + result := make([]string, 0, len(environment)+len(replacements)) + for _, entry := range environment { + key, _, found := strings.Cut(entry, "=") + if !found { + continue + } + replaced := false + for index := 0; index < len(replacements); index += 2 { + if key == replacements[index] { + replaced = true + break + } + } + if !replaced { + result = append(result, entry) + } + } + for index := 0; index < len(replacements); index += 2 { + result = append(result, replacements[index]+"="+replacements[index+1]) + } + return result +} diff --git a/validation/internal/retrievalstudy/golist_test.go b/validation/internal/retrievalstudy/golist_test.go new file mode 100644 index 0000000..9d43952 --- /dev/null +++ b/validation/internal/retrievalstudy/golist_test.go @@ -0,0 +1,30 @@ +package retrievalstudy + +import ( + "context" + "os" + "path/filepath" + "testing" + "time" +) + +func TestInventoryGoPackagesReportsOutputLimit(t *testing.T) { + toolDir := t.TempDir() + goShim := filepath.Join(toolDir, "go") + script := "#!/bin/sh\n/bin/dd if=/dev/zero bs=1048576 count=17 2>/dev/null\n" + if err := os.WriteFile(goShim, []byte(script), 0o755); err != nil { + t.Fatal(err) + } + t.Setenv("PATH", toolDir+string(os.PathListSeparator)+os.Getenv("PATH")) + + result, err := inventoryGoPackages(context.Background(), t.TempDir(), 30*time.Second) + if err != nil { + t.Fatal(err) + } + if result.Status != "partial" || result.Coverage != "partial_output_limit" || !result.OutputTruncated { + t.Fatalf("inventory status = %#v, want explicit output-limit coverage", result) + } + if result.OutputBytes != goListOutputLimit { + t.Fatalf("captured output bytes = %d, want cap %d", result.OutputBytes, goListOutputLimit) + } +} diff --git a/validation/internal/retrievalstudy/gopls.go b/validation/internal/retrievalstudy/gopls.go new file mode 100644 index 0000000..d025298 --- /dev/null +++ b/validation/internal/retrievalstudy/gopls.go @@ -0,0 +1,207 @@ +package retrievalstudy + +import ( + "context" + "fmt" + "net/url" + "path/filepath" + "strings" + "time" + + managedgopls "github.com/agentic-mcps/go/internal/gopls" +) + +type goplsLocation struct { + URI string `json:"uri"` + Range struct { + Start struct { + Line int `json:"line"` + } `json:"start"` + } `json:"range"` +} + +type goplsSymbol struct { + Name string `json:"name"` + ContainerName string `json:"containerName"` + Kind int `json:"kind"` + Location goplsLocation `json:"location"` +} + +type goplsSession struct { + client *managedgopls.Client + root string +} + +func startGopls(parent context.Context, binary, workspace string, timeout time.Duration) (*goplsSession, string, float64, error) { + probeCtx, cancelProbe := context.WithTimeout(parent, timeout) + installation, err := managedgopls.Locate(probeCtx, "", binary) + cancelProbe() + if err != nil { + return nil, "", 0, err + } + start := time.Now() + client, err := managedgopls.Start(parent, managedgopls.Config{ + Command: installation.Path, Workspace: workspace, + ClientVersion: "agentic-go-retrievalbench/v1", InitializeTimeout: timeout, + }) + initializeMS := float64(time.Since(start)) / float64(time.Millisecond) + if err != nil { + return nil, installation.Version, initializeMS, err + } + if !client.Capabilities().WorkspaceSymbol { + closeCtx, cancelClose := context.WithTimeout(context.Background(), timeout) + _ = client.Close(closeCtx) + cancelClose() + return nil, installation.Version, initializeMS, fmt.Errorf("gopls did not advertise workspace/symbol") + } + return &goplsSession{client: client, root: workspace}, installation.Version, initializeMS, nil +} + +func (session *goplsSession) close(timeout time.Duration) { + if session == nil || session.client == nil { + return + } + ctx, cancel := context.WithTimeout(context.Background(), timeout) + defer cancel() + _ = session.client.Close(ctx) +} + +func (session *goplsSession) search(parent context.Context, timeout time.Duration, query Query, source archivedSource, gold []GoldSpan, repetitions int) (Ranking, Latency, error) { + queryText := query.GoplsQuery + if strings.TrimSpace(queryText) == "" { + queryText = query.Text + } + samples := make([]float64, 0, repetitions) + var candidates []Candidate + var complete = true + var reason string + for repetition := 0; repetition < repetitions; repetition++ { + if err := parent.Err(); err != nil { + return Ranking{}, Latency{}, err + } + ctx, cancel := context.WithTimeout(parent, timeout) + start := time.Now() + var symbols []goplsSymbol + err := session.client.Request(ctx, "workspace/symbol", map[string]any{"query": queryText}, &symbols) + duration := time.Since(start) + cancel() + samples = append(samples, float64(duration)/float64(time.Millisecond)) + if err != nil { + complete = false + reason = "gopls workspace/symbol failed or timed out" + continue + } + candidates = session.mapSymbols(symbols, source) + } + ranking := scoreRanking(candidates, len(candidates), gold, complete, reason, "workspace_symbol_anchor", nil) + if complete { + ranking.Status = "unknown_completeness" + ranking.Complete = false + ranking.MetricsUsable = true + ranking.CandidateCountComplete = false + ranking.IncompleteReason = "workspace/symbol provider result cap is not established; scores cover only the returned candidate list" + } + if !source.coverage.SourceArchiveComplete { + ranking.Status = "partial" + ranking.Complete = false + ranking.MetricsUsable = false + ranking.CandidateCountComplete = false + ranking.IncompleteReason = source.coverage.SourceArchiveIncompleteReason + } + if !complete { + ranking.CandidateCountComplete = false + } + return ranking, latency(samples), nil +} + +func (session *goplsSession) mapSymbols(symbols []goplsSymbol, source archivedSource) []Candidate { + result := make([]Candidate, 0, len(symbols)) + seen := make(map[string]struct{}, len(symbols)) + for _, symbol := range symbols { + parsed, err := url.Parse(symbol.Location.URI) + if err != nil || parsed.Scheme != "file" || (parsed.Host != "" && parsed.Host != "localhost") { + continue + } + filename := filepath.Clean(filepath.FromSlash(parsed.Path)) + relative, err := filepath.Rel(session.root, filename) + if err != nil || relative == ".." || strings.HasPrefix(relative, ".."+string(filepath.Separator)) || filepath.IsAbs(relative) { + continue + } + relative = filepath.ToSlash(relative) + meta, ok := source.meta[relative] + line := symbol.Location.Range.Start.Line + 1 + if !ok || !meta.goFile || line < 1 || line > meta.lines { + continue + } + key := fmt.Sprintf("%s:%d:%d:%s", relative, line, symbol.Kind, symbol.Name) + if _, duplicate := seen[key]; duplicate { + continue + } + seen[key] = struct{}{} + result = append(result, Candidate{ + Path: relative, Line: line, Name: symbol.Name, + Container: symbol.ContainerName, Kind: goplsKind(symbol.Kind), + ProviderKind: symbol.Kind, + }) + } + return result +} + +func goplsKind(kind int) string { + switch kind { + case 1: + return "file" + case 2: + return "module" + case 3: + return "namespace" + case 4: + return "package" + case 5: + return "class" + case 6: + return "method" + case 7: + return "property" + case 8: + return "field" + case 9: + return "constructor" + case 10: + return "enum" + case 11: + return "interface" + case 12: + return "function" + case 13: + return "variable" + case 14: + return "constant" + case 15: + return "string" + case 16: + return "number" + case 17: + return "boolean" + case 18: + return "array" + case 19: + return "object" + case 20: + return "key" + case 21: + return "null" + case 22: + return "enum_member" + case 23: + return "struct" + case 24: + return "event" + case 25: + return "operator" + case 26: + return "type_parameter" + default: + return "unknown" + } +} diff --git a/validation/internal/retrievalstudy/identity.go b/validation/internal/retrievalstudy/identity.go new file mode 100644 index 0000000..530e8bf --- /dev/null +++ b/validation/internal/retrievalstudy/identity.go @@ -0,0 +1,182 @@ +package retrievalstudy + +import ( + "context" + "crypto/sha256" + "encoding/binary" + "encoding/hex" + "fmt" + "hash" + "io" + "os" + "os/exec" + "path" + "path/filepath" + "sort" + "strings" + "time" +) + +func sourceIdentity(parent context.Context, sourceRepository string, timeout time.Duration, manifestSHA256 string) (Reproducibility, error) { + root, err := gitText(parent, timeout, sourceRepository, "rev-parse", "--show-toplevel") + if err != nil { + return Reproducibility{}, fmt.Errorf("locating study source repository: %w", err) + } + root = strings.TrimSpace(root) + commit, err := gitText(parent, timeout, root, "rev-parse", "--verify", "HEAD^{commit}") + if err != nil { + return Reproducibility{}, fmt.Errorf("reading study source commit: %w", err) + } + dirtyDigest := sha256.New() + if _, err := dirtyDigest.Write([]byte("agentic-go-dirty-diff/v1\x00")); err != nil { + return Reproducibility{}, err + } + if err := hashGitOutput(parent, timeout, root, dirtyDigest, "diff", "--binary", "HEAD", "--"); err != nil { + return Reproducibility{}, fmt.Errorf("hashing tracked study-source changes: %w", err) + } + if _, err := dirtyDigest.Write([]byte("\x00untracked-files\x00")); err != nil { + return Reproducibility{}, err + } + untracked, err := gitTextBytes(parent, timeout, root, "ls-files", "--others", "--exclude-standard", "-z") + if err != nil { + return Reproducibility{}, fmt.Errorf("listing untracked study-source files: %w", err) + } + paths := make([]string, 0) + for _, rawPath := range strings.Split(string(untracked), "\x00") { + if rawPath != "" { + paths = append(paths, rawPath) + } + } + sort.Strings(paths) + for _, relative := range paths { + if err := parent.Err(); err != nil { + return Reproducibility{}, err + } + if !filepath.IsLocal(filepath.FromSlash(relative)) || path.Clean(relative) != relative { + return Reproducibility{}, fmt.Errorf("untracked study-source path is unsafe") + } + if err := writeField(dirtyDigest, []byte(relative)); err != nil { + return Reproducibility{}, err + } + filename := filepath.Join(root, filepath.FromSlash(relative)) + info, err := os.Lstat(filename) + if err != nil { + return Reproducibility{}, fmt.Errorf("inspecting an untracked study-source file: %w", err) + } + if info.Mode()&os.ModeSymlink != 0 { + target, err := os.Readlink(filename) + if err != nil { + return Reproducibility{}, fmt.Errorf("reading an untracked symlink: %w", err) + } + if _, err := dirtyDigest.Write([]byte("symlink\x00")); err != nil { + return Reproducibility{}, err + } + if err := writeField(dirtyDigest, []byte(target)); err != nil { + return Reproducibility{}, err + } + continue + } + if !info.Mode().IsRegular() { + return Reproducibility{}, fmt.Errorf("untracked study-source entry is not a regular file") + } + if _, err := dirtyDigest.Write([]byte("file\x00")); err != nil { + return Reproducibility{}, err + } + binaryLength := make([]byte, 8) + binary.BigEndian.PutUint64(binaryLength, uint64(info.Size())) + if _, err := dirtyDigest.Write(binaryLength); err != nil { + return Reproducibility{}, err + } + file, err := os.Open(filename) + if err != nil { + return Reproducibility{}, fmt.Errorf("opening an untracked study-source file: %w", err) + } + _, copyErr := copyContext(parent, dirtyDigest, file) + closeErr := file.Close() + if copyErr != nil { + return Reproducibility{}, fmt.Errorf("hashing an untracked study-source file: %w", copyErr) + } + if closeErr != nil { + return Reproducibility{}, fmt.Errorf("closing an untracked study-source file: %w", closeErr) + } + } + retrievalSource, err := os.ReadFile(filepath.Join(root, "internal", "intelligence", "retrieval", "index.go")) + if err != nil { + return Reproducibility{}, fmt.Errorf("reading evaluated retrieval source: %w", err) + } + retrievalDigest := sha256.Sum256(retrievalSource) + dirtyHex := hex.EncodeToString(dirtyDigest.Sum(nil)) + return Reproducibility{ + StudySourceCommit: strings.TrimSpace(commit), + DirtyDiffSHA256: dirtyHex, ManifestSHA256: manifestSHA256, + RetrievalSourceSHA256: hex.EncodeToString(retrievalDigest[:]), + }, nil +} + +func copyContext(ctx context.Context, destination io.Writer, source io.Reader) (int64, error) { + buffer := make([]byte, 64<<10) + var written int64 + for { + if err := ctx.Err(); err != nil { + return written, err + } + count, readErr := source.Read(buffer) + if count > 0 { + copied, writeErr := destination.Write(buffer[:count]) + written += int64(copied) + if writeErr != nil { + return written, writeErr + } + if copied != count { + return written, io.ErrShortWrite + } + } + if readErr == io.EOF { + return written, nil + } + if readErr != nil { + return written, readErr + } + } +} + +func hashGitOutput(parent context.Context, timeout time.Duration, repository string, destination hash.Hash, arguments ...string) error { + ctx, cancel := context.WithTimeout(parent, timeout) + defer cancel() + command := execCommand(ctx, repository, arguments...) + command.Stdout = destination + if err := command.Run(); err != nil { + if ctx.Err() != nil { + return ctx.Err() + } + return fmt.Errorf("git %s failed: %w", strings.Join(arguments, " "), err) + } + return nil +} + +func gitTextBytes(parent context.Context, timeout time.Duration, repository string, arguments ...string) ([]byte, error) { + ctx, cancel := context.WithTimeout(parent, timeout) + defer cancel() + output, err := execCommand(ctx, repository, arguments...).Output() + if err != nil { + if ctx.Err() != nil { + return nil, ctx.Err() + } + return nil, fmt.Errorf("git %s failed: %w", strings.Join(arguments, " "), err) + } + return output, nil +} + +func execCommand(ctx context.Context, repository string, arguments ...string) *exec.Cmd { + return exec.CommandContext(ctx, "git", append([]string{"-C", repository}, arguments...)...) +} + +func writeField(destination hash.Hash, value []byte) error { + length := make([]byte, 8) + binary.BigEndian.PutUint64(length, uint64(len(value))) + if _, err := destination.Write(length); err != nil { + return err + } + _, err := destination.Write(value) + return err +} diff --git a/validation/internal/retrievalstudy/manifest.go b/validation/internal/retrievalstudy/manifest.go new file mode 100644 index 0000000..a448c63 --- /dev/null +++ b/validation/internal/retrievalstudy/manifest.go @@ -0,0 +1,176 @@ +package retrievalstudy + +import ( + "bytes" + "encoding/json" + "crypto/sha256" + "encoding/hex" + "fmt" + "io" + "io/fs" + "os" + "path" + "regexp" + "strings" + "unicode" +) + +var fullCommitPattern = regexp.MustCompile(`^(?:[0-9a-f]{40}|[0-9a-f]{64})$`) +var repositoryIDPattern = regexp.MustCompile(`^[a-z0-9][a-z0-9.:-]*/[a-z0-9_.-]+(/[a-z0-9_.-]+)*$`) + +var evidenceTypes = map[string]struct{}{ + "declaration": {}, + "caller": {}, + "test": {}, + "documentation": {}, + "configuration": {}, +} + +// LoadManifest decodes a strict, versioned JSON manifest and validates all +// fields that do not depend on the archived source tree. +func LoadManifest(filename string) (Manifest, string, error) { + file, err := os.Open(filename) + if err != nil { + return Manifest{}, "", fmt.Errorf("opening manifest: %w", err) + } + defer file.Close() + data, err := io.ReadAll(io.LimitReader(file, (4<<20)+1)) + if err != nil { + return Manifest{}, "", fmt.Errorf("reading manifest: %w", err) + } + if len(data) > 4<<20 { + return Manifest{}, "", fmt.Errorf("manifest exceeds 4 MiB") + } + decoder := json.NewDecoder(bytes.NewReader(data)) + decoder.DisallowUnknownFields() + var manifest Manifest + if err := decoder.Decode(&manifest); err != nil { + return Manifest{}, "", fmt.Errorf("decoding manifest: %w", err) + } + if err := decoder.Decode(new(any)); err != io.EOF { + return Manifest{}, "", fmt.Errorf("manifest must contain exactly one JSON value") + } + if err := manifest.Validate(); err != nil { + return Manifest{}, "", err + } + digest := sha256.Sum256(data) + return manifest, hex.EncodeToString(digest[:]), nil +} + +func (manifest Manifest) Validate() error { + if manifest.Version != ManifestVersion { + return fmt.Errorf("manifest version must be %q", ManifestVersion) + } + if !validRepositoryID(manifest.RepositoryID) { + return fmt.Errorf("repository_id must be a canonical host/owner/repository identifier") + } + if !fullCommitPattern.MatchString(manifest.Commit) { + return fmt.Errorf("commit must be a full lowercase 40- or 64-character Git object ID") + } + if !fullCommitPattern.MatchString(manifest.Tree) { + return fmt.Errorf("tree must be a full lowercase 40- or 64-character Git object ID") + } + switch manifest.Stratum { + case "small", "medium", "large": + default: + return fmt.Errorf("stratum must be small, medium, or large") + } + if len(manifest.Queries) == 0 { + return fmt.Errorf("manifest must contain at least one query") + } + seen := make(map[string]struct{}, len(manifest.Queries)) + for queryIndex, query := range manifest.Queries { + if strings.TrimSpace(query.ID) == "" || len(query.ID) > 128 { + return fmt.Errorf("queries[%d].id must be non-empty and at most 128 bytes", queryIndex) + } + if _, exists := seen[query.ID]; exists { + return fmt.Errorf("duplicate query id %q", query.ID) + } + seen[query.ID] = struct{}{} + if strings.TrimSpace(query.Text) == "" || len(query.Text) > 4096 || len(Tokenize(query.Text)) == 0 { + return fmt.Errorf("queries[%d].query must contain searchable text of at most 4096 bytes", queryIndex) + } + if len(query.GoplsQuery) > 4096 { + return fmt.Errorf("queries[%d].gopls_query exceeds the 4096-byte limit", queryIndex) + } + if len(Tokenize(query.Text)) > 64 { + return fmt.Errorf("queries[%d].query exceeds the 64-token limit", queryIndex) + } + if len(query.Gold) == 0 { + return fmt.Errorf("queries[%d].gold must contain at least one reviewed evidence span", queryIndex) + } + for spanIndex, span := range query.Gold { + if _, ok := evidenceTypes[span.Type]; !ok { + return fmt.Errorf("queries[%d].gold[%d].type must be one of declaration, caller, test, documentation, configuration", queryIndex, spanIndex) + } + if !fs.ValidPath(span.Path) || path.IsAbs(span.Path) || strings.Contains(span.Path, "\\") { + return fmt.Errorf("queries[%d].gold[%d].path must be a slash-separated repository-relative path", queryIndex, spanIndex) + } + if span.StartLine < 1 || span.EndLine < span.StartLine { + return fmt.Errorf("queries[%d].gold[%d] has an invalid one-based line range", queryIndex, spanIndex) + } + } + } + return nil +} + +func validRepositoryID(value string) bool { + if !repositoryIDPattern.MatchString(value) || strings.TrimSpace(value) != value || strings.ContainsAny(value, "\\?#@") { + return false + } + parts := strings.Split(value, "/") + if len(parts) < 2 { + return false + } + if parts[0] == "local" && len(parts) != 2 { + return false + } + for _, part := range parts { + if part == "" || part == "." || part == ".." || strings.HasSuffix(part, ".git") { + return false + } + } + return strings.ToLower(value) == value +} + +// Tokenize intentionally mirrors internal/intelligence/retrieval's current +// Unicode and camel-case tokenizer. Keep this copy versioned with the study: +// changing it changes the fixed native rg baseline. +func Tokenize(value string) []string { + value = strings.TrimSpace(value) + if value == "" { + return nil + } + runes := []rune(value) + result := make([]string, 0, len(runes)/2) + start := -1 + flush := func(end int) { + if start < 0 || end <= start { + start = -1 + return + } + result = append(result, strings.ToLower(string(runes[start:end]))) + start = -1 + } + for index, current := range runes { + if !unicode.IsLetter(current) && !unicode.IsDigit(current) && current != '_' { + flush(index) + continue + } + if start < 0 { + start = index + continue + } + previous := runes[index-1] + if current == '_' { + flush(index) + continue + } + if unicode.IsUpper(current) && unicode.IsLower(previous) { + flush(index) + start = index + } + } + flush(len(runes)) + return result +} diff --git a/validation/internal/retrievalstudy/metrics.go b/validation/internal/retrievalstudy/metrics.go new file mode 100644 index 0000000..788684a --- /dev/null +++ b/validation/internal/retrievalstudy/metrics.go @@ -0,0 +1,154 @@ +package retrievalstudy + +import ( + "math" + "sort" +) + +func scoreRanking(candidates []Candidate, candidateCount int, gold []GoldSpan, complete bool, reason, granularity string, tokens []string) Ranking { + if candidates == nil { + candidates = []Candidate{} + } + ranking := Ranking{ + Status: "complete", Complete: complete, MetricsUsable: complete, IncompleteReason: reason, + Candidates: candidates, CandidateCount: candidateCount, + CandidateCountComplete: true, Granularity: granularity, Tokens: tokens, + } + if !complete { + ranking.Status = "partial" + } + ranking.Metrics.At5, ranking.Misses = metricsAt(candidates, gold, 5) + at10, missesAt10 := metricsAt(candidates, gold, 10) + ranking.Metrics.At10 = at10 + ranking.Misses.At10 = missesAt10.At10 + ranking.Misses.ByTypeAt10 = missesAt10.ByTypeAt10 + return ranking +} + +func metricsAt(candidates []Candidate, gold []GoldSpan, cutoff int) (CutoffMetrics, EvidenceMisses) { + limit := cutoff + if len(candidates) < limit { + limit = len(candidates) + } + goldHit := make([]bool, len(gold)) + relevantCandidates := 0 + firstRelevantRank := 0 + for index := 0; index < limit; index++ { + candidate := candidates[index] + relevant := false + for spanIndex, span := range gold { + if candidate.Path == span.Path && candidate.Line >= span.StartLine && candidate.Line <= span.EndLine { + goldHit[spanIndex] = true + relevant = true + } + } + if relevant { + relevantCandidates++ + if firstRelevantRank == 0 { + firstRelevantRank = index + 1 + } + } + } + hitCount := 0 + misses := EvidenceMisses{GoldSpans: len(gold), ByTypeAt5: make(map[string]int), ByTypeAt10: make(map[string]int)} + for index, span := range gold { + if goldHit[index] { + hitCount++ + } else { + if cutoff == 5 { + misses.ByTypeAt5[span.Type]++ + } else { + misses.ByTypeAt10[span.Type]++ + } + } + } + metric := CutoffMetrics{ + RetrievedCandidates: limit, + RelevantCandidates: relevantCandidates, + GoldSpans: len(gold), + HitGoldSpans: hitCount, + Recall: ratio(hitCount, len(gold)), + Precision: ratio(relevantCandidates, cutoff), + } + if firstRelevantRank != 0 { + metric.MeanReciprocalRank = 1 / float64(firstRelevantRank) + } + missed := len(gold) - hitCount + if cutoff == 5 { + misses.At5 = missed + } else { + misses.At10 = missed + } + return metric, misses +} + +func ratio(numerator, denominator int) float64 { + if denominator == 0 { + return 0 + } + return float64(numerator) / float64(denominator) +} + +func latency(samples []float64) Latency { + values := append([]float64(nil), samples...) + if len(values) == 0 { + return Latency{Samples: []float64{}} + } + sort.Float64s(values) + return Latency{ + Samples: append([]float64(nil), samples...), + P50MS: percentile(values, 0.50), + P95MS: percentile(values, 0.95), + MinMS: values[0], + MaxMS: values[len(values)-1], + } +} + +func percentile(sorted []float64, quantile float64) float64 { + if len(sorted) == 0 { + return 0 + } + index := int(math.Ceil(quantile*float64(len(sorted)))) - 1 + if index < 0 { + index = 0 + } + if index >= len(sorted) { + index = len(sorted) - 1 + } + return sorted[index] +} + +func aggregateRankings(rankings []Ranking, cutoff int) AggregateMetrics { + metric := AggregateMetrics{Queries: len(rankings)} + macroRecall, macroPrecision, macroMRR := 0.0, 0.0, 0.0 + totalGold, totalHit, totalRelevant, totalReturned := 0, 0, 0, 0 + for _, ranking := range rankings { + if !ranking.MetricsUsable { + continue + } + metric.MetricsQueries++ + if ranking.Complete { + metric.CompleteQueries++ + } + cutoffMetric := ranking.Metrics.At10 + if cutoff == 5 { + cutoffMetric = ranking.Metrics.At5 + } + macroRecall += cutoffMetric.Recall + macroPrecision += cutoffMetric.Precision + macroMRR += cutoffMetric.MeanReciprocalRank + totalGold += cutoffMetric.GoldSpans + totalHit += cutoffMetric.HitGoldSpans + totalRelevant += cutoffMetric.RelevantCandidates + totalReturned += cutoff + } + if metric.MetricsQueries > 0 { + denominator := float64(metric.MetricsQueries) + metric.MacroRecall = macroRecall / denominator + metric.MacroPrecision = macroPrecision / denominator + metric.MacroMeanReciprocalRank = macroMRR / denominator + } + metric.MicroRecall = ratio(totalHit, totalGold) + metric.MicroPrecision = ratio(totalRelevant, totalReturned) + return metric +} diff --git a/validation/internal/retrievalstudy/native.go b/validation/internal/retrievalstudy/native.go new file mode 100644 index 0000000..30a0e18 --- /dev/null +++ b/validation/internal/retrievalstudy/native.go @@ -0,0 +1,278 @@ +package retrievalstudy + +import ( + "bufio" + "bytes" + "context" + "errors" + "fmt" + "os/exec" + "path" + "sort" + "strconv" + "strings" + "time" +) + +const ( + nativeOutputLimitDefault = int64(64 << 20) + nativeLineLimitDefault = 100000 +) + +var nativeGlobs = []string{ + "*.go", "*.md", "*.rst", "*.txt", "*.yaml", "*.yml", "*.json", + "*.toml", "*.proto", "*.mod", "*.sum", "*.work", "Makefile", "GNUmakefile", + "Dockerfile", "Containerfile", ".gitignore", ".editorconfig", ".gitattributes", +} + +type nativeLimits struct { + timeout time.Duration + outputBytes int64 + lineCount int +} + +type nativeMeasurement struct { + ranking Ranking + totalLatency Latency + commandLatency Latency + rankingLatency Latency +} + +type rgLine struct { + path string + line int + text string +} + +// ProbeRG returns the first version line without recording the executable path. +func ProbeRG(parent context.Context, timeout time.Duration) (string, error) { + ctx, cancel := context.WithTimeout(parent, timeout) + defer cancel() + command := exec.CommandContext(ctx, "rg", "--version") + output, err := command.Output() + if err != nil { + if ctx.Err() != nil { + return "", ctx.Err() + } + return "", fmt.Errorf("probing ripgrep: %w", err) + } + line := strings.TrimSpace(strings.SplitN(string(output), "\n", 2)[0]) + return line, nil +} + +func RunNativeRG(parent context.Context, workspace string, query Query, source archivedSource, gold []GoldSpan, repetitions int, limits nativeLimits) (nativeMeasurement, error) { + tokens := uniqueSortedTokens(query.Text) + commandSamples := make([]float64, 0, repetitions) + rankingSamples := make([]float64, 0, repetitions) + totalSamples := make([]float64, 0, repetitions) + var finalCandidates []Candidate + var finalCount int + complete := true + var incompleteReason string + for sample := 0; sample < repetitions; sample++ { + if err := parent.Err(); err != nil { + return nativeMeasurement{}, err + } + lines, gatherDuration, runComplete, runReason, err := gatherRG(parent, workspace, tokens, source, limits) + if err != nil { + return nativeMeasurement{}, err + } + commandMS := float64(gatherDuration) / float64(time.Millisecond) + commandSamples = append(commandSamples, commandMS) + startRank := time.Now() + candidates := rankRGLines(lines, tokens) + rankingDuration := time.Since(startRank) + rankingMS := float64(rankingDuration) / float64(time.Millisecond) + rankingSamples = append(rankingSamples, rankingMS) + totalSamples = append(totalSamples, commandMS+rankingMS) + finalCandidates = candidates.top + finalCount = candidates.total + if !runComplete { + complete = false + incompleteReason = runReason + } + } + ranking := scoreRanking(finalCandidates, finalCount, gold, complete, incompleteReason, "matching_source_line", tokens) + ranking.CandidateCountComplete = complete && source.coverage.SourceArchiveComplete + if !source.coverage.SourceArchiveComplete { + ranking.Complete = false + ranking.MetricsUsable = false + ranking.Status = "partial" + if ranking.IncompleteReason == "" { + ranking.IncompleteReason = source.coverage.SourceArchiveIncompleteReason + } + } + return nativeMeasurement{ + ranking: ranking, + totalLatency: latency(totalSamples), + commandLatency: latency(commandSamples), + rankingLatency: latency(rankingSamples), + }, nil +} + +func gatherRG(parent context.Context, workspace string, tokens []string, source archivedSource, limits nativeLimits) ([]rgLine, time.Duration, bool, string, error) { + arguments := []string{ + "--no-ignore", "--hidden", "--glob-case-insensitive", "--no-heading", "--with-filename", "--line-number", + "--color", "never", "--fixed-strings", "--ignore-case", "--text", "--null", + } + for _, glob := range nativeGlobs { + arguments = append(arguments, "--glob="+glob) + } + for _, token := range tokens { + arguments = append(arguments, "-e", token) + } + arguments = append(arguments, "--", ".") + ctx, cancel := context.WithTimeout(parent, limits.timeout) + defer cancel() + command := exec.CommandContext(ctx, "rg", arguments...) + command.Dir = workspace + stdout, err := command.StdoutPipe() + if err != nil { + return nil, 0, false, "", fmt.Errorf("opening ripgrep output: %w", err) + } + start := time.Now() + if err := command.Start(); err != nil { + return nil, 0, false, "", fmt.Errorf("starting ripgrep: %w", err) + } + scanner := bufio.NewScanner(stdout) + scanner.Buffer(make([]byte, 64<<10), 4<<20) + lines := make([]rgLine, 0) + var bytesRead int64 + complete := true + reason := "" + for scanner.Scan() { + raw := scanner.Bytes() + bytesRead += int64(len(raw) + 1) + if bytesRead > limits.outputBytes { + complete = false + reason = "ripgrep output exceeded the configured byte limit" + _ = command.Process.Kill() + break + } + if len(lines) >= limits.lineCount { + complete = false + reason = "ripgrep output exceeded the configured matching-line limit" + _ = command.Process.Kill() + break + } + parsed, ok := parseRGLine(raw) + if !ok { + complete = false + reason = "ripgrep emitted a line that did not match the recorded output format" + _ = command.Process.Kill() + break + } + meta, supported := source.meta[parsed.path] + if !supported || parsed.line < 1 || parsed.line > meta.lines { + complete = false + reason = "ripgrep returned a path or line outside the archived supported-source manifest" + _ = command.Process.Kill() + break + } + lines = append(lines, parsed) + } + if scanner.Err() != nil && complete { + complete = false + reason = "ripgrep output line exceeded the 4 MiB line limit" + _ = command.Process.Kill() + } + waitErr := command.Wait() + duration := time.Since(start) + if parent.Err() != nil { + return nil, duration, false, "", parent.Err() + } + if ctx.Err() != nil && errors.Is(ctx.Err(), context.DeadlineExceeded) { + complete = false + reason = "ripgrep exceeded the per-query timeout" + } else if waitErr != nil && complete { + var exitErr *exec.ExitError + if errors.As(waitErr, &exitErr) && exitErr.ExitCode() == 1 { + // ripgrep uses exit code 1 for a successful search with no matches. + } else { + complete = false + reason = "ripgrep exited unsuccessfully" + } + } + return lines, duration, complete, reason, nil +} + +func parseRGLine(raw []byte) (rgLine, bool) { + separator := bytes.IndexByte(raw, 0) + if separator <= 0 { + return rgLine{}, false + } + pathValue := strings.TrimPrefix(string(raw[:separator]), "./") + if path.Clean(pathValue) != pathValue { + return rgLine{}, false + } + rest := raw[separator+1:] + colon := bytes.IndexByte(rest, ':') + if colon <= 0 { + return rgLine{}, false + } + line, err := strconv.Atoi(string(rest[:colon])) + if err != nil || line < 1 { + return rgLine{}, false + } + return rgLine{path: pathValue, line: line, text: string(rest[colon+1:])}, true +} + +type rankedLines struct { + top []Candidate + total int +} + +func rankRGLines(lines []rgLine, queryTokens []string) rankedLines { + query := make(map[string]struct{}, len(queryTokens)) + for _, token := range queryTokens { + query[token] = struct{}{} + } + candidates := make([]Candidate, 0, len(lines)) + for _, line := range lines { + frequencies := make(map[string]int) + for _, token := range Tokenize(line.text) { + if _, included := query[token]; included { + frequencies[token]++ + } + } + occurrences := 0 + for _, count := range frequencies { + occurrences += count + } + candidates = append(candidates, Candidate{ + Path: line.path, Line: line.line, + MatchedUniqueTerms: len(frequencies), MatchedOccurrences: occurrences, + }) + } + sort.SliceStable(candidates, func(i, j int) bool { + left, right := candidates[i], candidates[j] + if left.MatchedUniqueTerms != right.MatchedUniqueTerms { + return left.MatchedUniqueTerms > right.MatchedUniqueTerms + } + if left.MatchedOccurrences != right.MatchedOccurrences { + return left.MatchedOccurrences > right.MatchedOccurrences + } + if left.Path != right.Path { + return left.Path < right.Path + } + return left.Line < right.Line + }) + count := len(candidates) + if len(candidates) > 10 { + candidates = candidates[:10] + } + return rankedLines{top: candidates, total: count} +} + +func uniqueSortedTokens(text string) []string { + seen := make(map[string]struct{}) + for _, token := range Tokenize(text) { + seen[token] = struct{}{} + } + tokens := make([]string, 0, len(seen)) + for token := range seen { + tokens = append(tokens, token) + } + sort.Strings(tokens) + return tokens +} diff --git a/validation/internal/retrievalstudy/run.go b/validation/internal/retrievalstudy/run.go new file mode 100644 index 0000000..5ed1ed8 --- /dev/null +++ b/validation/internal/retrievalstudy/run.go @@ -0,0 +1,657 @@ +package retrievalstudy + +import ( + "context" + "encoding/json" + "fmt" + "math" + "os" + "path/filepath" + "runtime" + "time" + + "github.com/agentic-mcps/go/internal/intelligence/retrieval" +) + +const retrievalGranularity = "go_declaration_name_anchor" +const candidatePoolAuditLimit = 10_000 + +type Options struct { + Manifest string + RepositoryPath string + SourceRepositoryPath string + Output string + GoplsBinary string + TextAblation bool + Repetitions int + Timeout time.Duration + MaxSourceBytes int64 + MaxFileBytes int64 + RGOutputBytes int64 + RGLineLimit int +} + +func Execute(parent context.Context, options Options) error { + if err := validateOptions(options); err != nil { + return err + } + manifest, manifestSHA256, err := LoadManifest(options.Manifest) + if err != nil { + return err + } + if options.SourceRepositoryPath == "" { + options.SourceRepositoryPath = "." + } + reproducibility, err := sourceIdentity(parent, options.SourceRepositoryPath, options.Timeout, manifestSHA256) + if err != nil { + return err + } + temporaryRoot, err := os.MkdirTemp("", "agentic-go-retrievalbench-*") + if err != nil { + return fmt.Errorf("creating temporary benchmark workspace: %w", err) + } + defer os.RemoveAll(temporaryRoot) + workspace := filepath.Join(temporaryRoot, "snapshot") + if err := os.Mkdir(workspace, 0o700); err != nil { + return fmt.Errorf("creating snapshot workspace: %w", err) + } + archiveStarted := time.Now() + repository, source, err := ExportCommit(parent, options.RepositoryPath, workspace, manifest, archiveLimits{ + maxSourceBytes: options.MaxSourceBytes, + maxFileBytes: options.MaxFileBytes, + timeout: options.Timeout, + indexTextCandidates: options.TextAblation, + }) + archiveMS := float64(time.Since(archiveStarted)) / float64(time.Millisecond) + if err != nil { + return err + } + if err := validateGold(source, manifest); err != nil { + return err + } + goPackages, err := inventoryGoPackages(parent, workspace, options.Timeout) + if err != nil { + return err + } + rgVersion, err := ProbeRG(parent, options.Timeout) + if err != nil { + return err + } + var initial runtime.MemStats + runtime.ReadMemStats(&initial) + peakHeap := initial.HeapAlloc + + goplsMeasurement := GoplsMeasurement{ + Status: "unavailable", UnavailableReason: "gopls baseline not run; provide --gopls with the pinned v0.21.0 binary", + QueryLatency: latency(nil), Granularity: "workspace_symbol_anchor", + } + var session *goplsSession + if options.GoplsBinary != "" { + var version string + var initializeMS float64 + session, version, initializeMS, err = startGopls(parent, options.GoplsBinary, workspace, options.Timeout) + if version != "" { + goplsMeasurement.Version = version + goplsMeasurement.InitializeMS = &initializeMS + } + if err != nil { + goplsMeasurement.UnavailableReason = "configured gopls binary could not initialize the pinned workspace/symbol provider" + } else { + goplsMeasurement.Status = "unknown_completeness" + goplsMeasurement.UnavailableReason = "" + goplsMeasurement.CompletenessNote = "workspace/symbol result caps and exhaustive workspace coverage are not established; per-query scores cover returned candidates" + } + } + defer func() { session.close(options.Timeout) }() + + queryResults := make([]QueryResult, 0, len(manifest.Queries)) + var firstSearchResult *retrieval.Result + var allGoplsLatencies []float64 + var allNativeLatencies []float64 + var allNativeGatherLatencies []float64 + var allNativeRankingLatencies []float64 + var goplsCompletedQueries int + for _, query := range manifest.Queries { + if err := parent.Err(); err != nil { + return err + } + retrievalRanking, retrievalTimings, retrievalProfiles, searchResult, err := measureRetrieval(parent, query, source, repository, options) + if err != nil { + return err + } + if searchResult != nil && firstSearchResult == nil { + copyResult := *searchResult + firstSearchResult = ©Result + } + heapSample(&peakHeap) + var retrievalCandidatePool *CandidatePoolAudit + var textCandidateAblation *TextCandidateResult + if options.TextAblation { + retrievalCandidatePool, textCandidateAblation, err = measureTextAblation(parent, query, source, repository, options) + if err != nil { + return fmt.Errorf("text candidate ablation for query %q: %w", query.ID, err) + } + heapSample(&peakHeap) + } + + native, err := RunNativeRG(parent, workspace, query, source, query.Gold, options.Repetitions, nativeLimits{ + timeout: options.Timeout, outputBytes: options.RGOutputBytes, lineCount: options.RGLineLimit, + }) + if err != nil { + return fmt.Errorf("native rg query %q: %w", query.ID, err) + } + allNativeLatencies = append(allNativeLatencies, native.totalLatency.Samples...) + allNativeGatherLatencies = append(allNativeGatherLatencies, native.commandLatency.Samples...) + allNativeRankingLatencies = append(allNativeRankingLatencies, native.rankingLatency.Samples...) + heapSample(&peakHeap) + + queryResult := QueryResult{ + ID: query.ID, Text: query.Text, Gold: append([]GoldSpan(nil), query.Gold...), + Retrieval: retrievalRanking, RetrievalTimings: retrievalTimings, + RetrievalProfiles: retrievalProfiles, + RetrievalCandidatePool: retrievalCandidatePool, + TextCandidateAblation: textCandidateAblation, + NativeRG: native.ranking, NativeRGLatency: native.totalLatency, + NativeRGCommandLatency: native.commandLatency, + NativeRGRankingLatency: native.rankingLatency, + } + if session != nil { + goplsRanking, goplsLatency, searchErr := session.search(parent, options.Timeout, query, source, query.Gold, options.Repetitions) + if searchErr != nil { + return fmt.Errorf("gopls query %q: %w", query.ID, searchErr) + } + queryResult.GoplsQuery = query.GoplsQuery + if queryResult.GoplsQuery == "" { + queryResult.GoplsQuery = query.Text + } + queryResult.Gopls = goplsRanking + queryResult.GoplsQueryLatency = goplsLatency + allGoplsLatencies = append(allGoplsLatencies, goplsLatency.Samples...) + if goplsRanking.Status == "unknown_completeness" { + goplsCompletedQueries++ + } else { + goplsMeasurement.Status = "partial" + if goplsMeasurement.CompletenessNote == "" { + goplsMeasurement.CompletenessNote = "one or more workspace/symbol queries failed or timed out" + } + } + heapSample(&peakHeap) + } else { + queryResult.Gopls = unavailableRanking("gopls provider did not initialize") + } + queryResults = append(queryResults, queryResult) + } + if firstSearchResult != nil { + source.coverage.GoFilesWithDeclarations = firstSearchResult.IndexedFiles + source.coverage.GoParseIncompleteFiles = firstSearchResult.SkippedFiles + } + if session != nil { + goplsMeasurement.QueryLatency = latency(allGoplsLatencies) + if goplsCompletedQueries == 0 && len(queryResults) > 0 { + goplsMeasurement.Status = "partial" + } + } + var final runtime.MemStats + runtime.ReadMemStats(&final) + heapSampleValue := final.HeapAlloc + if heapSampleValue > peakHeap { + peakHeap = heapSampleValue + } + var summary Summary + summary, err = summarize(queryResults) + if err != nil { + return err + } + summary.GoplsStatus = goplsMeasurement.Status + created := time.Now().UTC() + reportVersion := "agentic-go.retrieval-report/v1" + var textCandidateIndex *TextCandidateIndexCoverage + if options.TextAblation { + reportVersion = "agentic-go.retrieval-report/v2" + coverage := source.textCandidateIndex + textCandidateIndex = &coverage + } + report := Report{ + SchemaVersion: reportVersion, CreatedUTC: created, + Repository: repository, Stratum: manifest.Stratum, + Reproducibility: reproducibility, + GoVersion: runtime.Version(), GOOS: runtime.GOOS, GOARCH: runtime.GOARCH, + RGVersion: rgVersion, SnapshotExportMS: archiveMS, + RetrievalTimingNote: "Cache.SearchProfiled reports parse, aggregation and rank stages; its total covers the full Go declaration candidate retrieval call. Top-5 is scored from the prefix of the same top-10 result, so top-5 and top-10 timing samples are identical and the large corpus is not parsed twice for two cutoffs. The optional text candidate ablation uses the same scorer over a bounded mixed candidate pool and is reported separately. This screen measures the candidate retrieval kernel, not full go_context output or semantic resolution.", + NativeRGWorkflow: NativeWorkflow{ + CommandTemplate: "rg --no-ignore --hidden --glob-case-insensitive --no-heading --with-filename --line-number --color never --fixed-strings --ignore-case --text --null [supported-source globs] -e ... -- .", + Tokenizer: "retrieval Unicode letter/digit/underscore tokenizer with lower-to-upper camel-case splits; duplicate tokens removed and sorted", + Ranking: []string{"distinct query tokens present on line descending", "query-token occurrences on line descending", "repository-relative path ascending", "line ascending"}, + Globs: append([]string(nil), nativeGlobs...), + PathScope: "Git-archive files with supported source suffixes only; archive ignores checkout working-tree changes", + }, + GoPackages: goPackages, Coverage: source.coverage, TextCandidateIndex: textCandidateIndex, + Configuration: Configuration{ + Repetitions: options.Repetitions, CommandTimeoutMS: options.Timeout.Milliseconds(), + SearchTimeoutMS: options.Timeout.Milliseconds(), MaxSourceBytes: options.MaxSourceBytes, + MaxFileBytes: options.MaxFileBytes, NativeOutputLimit: options.RGOutputBytes, + NativeLineLimit: options.RGLineLimit, + }, + Gopls: goplsMeasurement, Queries: queryResults, Summary: summary, + Heap: HeapStats{ + StartHeapAllocBytes: initial.HeapAlloc, EndHeapAllocBytes: final.HeapAlloc, + PeakSampledHeapAllocBytes: peakHeap, EndHeapSysBytes: final.HeapSys, + TotalAllocBytes: final.TotalAlloc - initial.TotalAlloc, NumGC: final.NumGC - initial.NumGC, + }, + } + return writeReport(options.Output, report) +} + +func validateOptions(options Options) error { + if options.Manifest == "" || options.RepositoryPath == "" || options.Output == "" { + return fmt.Errorf("--manifest, --repo, and --out are required") + } + if options.Repetitions < 1 || options.Repetitions > 20 { + return fmt.Errorf("--repetitions must be between 1 and 20") + } + if options.Timeout <= 0 { + return fmt.Errorf("--timeout must be positive") + } + if options.MaxSourceBytes <= 0 || options.MaxFileBytes <= 0 || options.MaxFileBytes > options.MaxSourceBytes { + return fmt.Errorf("source byte limits must be positive and per-file limit cannot exceed total source limit") + } + if options.RGOutputBytes <= 0 || options.RGLineLimit <= 0 { + return fmt.Errorf("native rg output limits must be positive") + } + return nil +} + +func validateGold(source archivedSource, manifest Manifest) error { + for _, query := range manifest.Queries { + for _, span := range query.Gold { + meta, ok := source.meta[span.Path] + if !ok { + if _, expected := source.expected[span.Path]; expected { + continue + } + return fmt.Errorf("query %q gold span path %q is not a supported source file in the pinned tree", query.ID, span.Path) + } + if span.EndLine > meta.lines { + return fmt.Errorf("query %q gold span %q ends past the archived file's %d lines", query.ID, span.Path, meta.lines) + } + } + } + return nil +} + +type searchObservation struct { + result retrieval.Result + profile retrieval.SearchProfile + wallTime float64 + err error +} + +func measureRetrieval(parent context.Context, query Query, source archivedSource, repository Repository, options Options) (Ranking, RetrievalTimings, RetrievalProfiles, *retrieval.Result, error) { + timings := RetrievalTimings{Cold: make(map[string]Latency), Warm: make(map[string]Latency)} + profiles := RetrievalProfiles{Cold: make(map[string]RetrievalProfile), Warm: make(map[string]RetrievalProfile)} + complete := true + reason := "" + var latestCold searchObservation + var cache *retrieval.Cache + coldSamples := make([]float64, 0, options.Repetitions) + coldProfiles := make([]retrieval.SearchProfile, 0, options.Repetitions) + for repetition := 0; repetition < options.Repetitions; repetition++ { + runtime.GC() + cache = retrieval.NewCache() + observation, err := profiledSearch(parent, cache, source, repository, query.Text, 10, options.Timeout) + if err != nil && parent.Err() != nil { + return Ranking{}, timings, profiles, nil, parent.Err() + } + latestCold = observation + coldSamples = append(coldSamples, observation.wallTime) + coldProfiles = append(coldProfiles, observation.profile) + if err != nil { + complete = false + if reason == "" { + reason = "Cache.SearchProfiled cold sample failed or timed out" + } + } + } + coldLatency := latency(coldSamples) + coldProfile := profileSeries(coldProfiles) + warmSamples := make([]float64, 0, options.Repetitions) + warmProfiles := make([]retrieval.SearchProfile, 0, options.Repetitions) + var latestWarm searchObservation + for repetition := 0; repetition < options.Repetitions; repetition++ { + observation, err := profiledSearch(parent, cache, source, repository, query.Text, 10, options.Timeout) + if err != nil && parent.Err() != nil { + return Ranking{}, timings, profiles, nil, parent.Err() + } + latestWarm = observation + warmSamples = append(warmSamples, observation.wallTime) + warmProfiles = append(warmProfiles, observation.profile) + if err != nil { + complete = false + if reason == "" { + reason = "Cache.SearchProfiled warm sample failed or timed out" + } + } + } + warmLatency := latency(warmSamples) + warmProfile := profileSeries(warmProfiles) + for _, label := range []string{"5", "10"} { + timings.Cold[label] = coldLatency + timings.Warm[label] = warmLatency + profiles.Cold[label] = coldProfile + profiles.Warm[label] = warmProfile + } + selected := latestWarm + if selected.err != nil { + selected = latestCold + } + var topTenResult *retrieval.Result + if selected.err == nil { + copyResult := selected.result + topTenResult = ©Result + if !copyResult.Complete { + complete = false + if reason == "" { + reason = "one or more Go files could not be parsed completely" + } + } + } + if topTenResult == nil { + return unavailableRanking("retrieval timed out before producing top-10 results"), timings, profiles, nil, nil + } + resultCandidates := make([]Candidate, 0, len(topTenResult.Candidates)) + for _, candidate := range topTenResult.Candidates { + resultCandidates = append(resultCandidates, Candidate{ + Path: candidate.Path, Line: candidate.Line, Name: candidate.Name, + Kind: candidate.Kind, Container: candidate.Package, + }) + } + ranking := scoreRanking(resultCandidates, len(resultCandidates), query.Gold, complete, reason, retrievalGranularity, nil) + if !source.coverage.SourceArchiveComplete { + ranking.Complete = false + ranking.MetricsUsable = false + ranking.Status = "partial" + if ranking.IncompleteReason == "" { + ranking.IncompleteReason = source.coverage.SourceArchiveIncompleteReason + } + } + ranking.CandidateCountComplete = !topTenResult.Truncated && topTenResult.Complete && source.coverage.SourceArchiveComplete + return ranking, timings, profiles, topTenResult, nil +} + +func profiledSearch(parent context.Context, cache *retrieval.Cache, source archivedSource, repository Repository, query string, limit int, timeout time.Duration) (searchObservation, error) { + ctx, cancel := context.WithTimeout(parent, timeout) + defer cancel() + key := retrieval.Key{ + Workspace: repository.ID, Scope: repository.Commit, + Build: "retrievalbench", Provider: "lexical-declaration-index", + } + start := time.Now() + result, profile, err := cache.SearchProfiled(ctx, key, source.goFiles, query, limit) + wall := float64(time.Since(start)) / float64(time.Millisecond) + observation := searchObservation{result: result, profile: profile, wallTime: wall, err: err} + if err != nil { + return observation, fmt.Errorf("search timed out or failed: %w", err) + } + return observation, nil +} + +func measureTextAblation(parent context.Context, query Query, source archivedSource, repository Repository, options Options) (*CandidatePoolAudit, *TextCandidateResult, error) { + key := retrieval.Key{ + Workspace: repository.ID, Scope: repository.Commit, + Build: "retrievalbench", Provider: "lexical-declaration-index", + } + search := func(includeText bool) (retrieval.Result, RetrievalProfile, float64, error) { + ctx, cancel := context.WithTimeout(parent, options.Timeout) + defer cancel() + cache := retrieval.NewCache() + started := time.Now() + var result retrieval.Result + var profile retrieval.SearchProfile + var err error + if includeText { + result, profile, err = cache.SearchWithTextProfiled(ctx, key, source.goFiles, source.textFiles, query.Text, candidatePoolAuditLimit) + } else { + result, profile, err = cache.SearchProfiled(ctx, key, source.goFiles, query.Text, candidatePoolAuditLimit) + } + wallMS := float64(time.Since(started)) / float64(time.Millisecond) + return result, profileSeries([]retrieval.SearchProfile{profile}), wallMS, err + } + goResult, _, _, err := search(false) + if err != nil { + return nil, nil, fmt.Errorf("auditing declaration candidate pool: %w", err) + } + basePool := candidatePoolAudit(goResult, query.Gold, source, false) + textResult, textProfile, wallMS, err := search(true) + if err != nil { + return nil, nil, fmt.Errorf("searching bounded text candidate pool: %w", err) + } + textPool := candidatePoolAudit(textResult, query.Gold, source, true) + textCandidates := studyCandidates(textResult.Candidates) + usableCount := len(textCandidates) + if usableCount > 10 { + usableCount = 10 + } + rankingComplete := textResult.Complete && source.coverage.SourceArchiveComplete && source.textCandidateIndex.Status == "complete" + rankingReason := "" + if !textResult.Complete { + rankingReason = "one or more Go or text candidate files could not be indexed completely" + } else if !source.coverage.SourceArchiveComplete { + rankingReason = source.coverage.SourceArchiveIncompleteReason + } else if source.textCandidateIndex.Status != "complete" { + rankingReason = source.textCandidateIndex.Reason + } + ranking := scoreRanking(textCandidates[:usableCount], textResult.CandidateCount, query.Gold, rankingComplete, rankingReason, "go_declaration_plus_text_line", nil) + ranking.CandidateCountComplete = textPool.Complete + return &basePool, &TextCandidateResult{ + Ranking: ranking, CandidatePool: textPool, + Latency: latency([]float64{wallMS}), Profile: textProfile, + IndexedTextFragments: textResult.TextIndexedFragments, + SkippedTextFiles: textResult.TextSkippedFiles, + MaximumTextFragments: retrieval.MaximumTextLineFragments, + }, nil +} + +func candidatePoolAudit(result retrieval.Result, gold []GoldSpan, source archivedSource, includesText bool) CandidatePoolAudit { + candidates := studyCandidates(result.Candidates) + hits := make([]bool, len(gold)) + for _, candidate := range candidates { + for index, span := range gold { + if candidate.Path == span.Path && candidate.Line >= span.StartLine && candidate.Line <= span.EndLine { + hits[index] = true + } + } + } + hitCount := 0 + for _, hit := range hits { + if hit { + hitCount++ + } + } + complete := result.Complete && source.coverage.SourceArchiveComplete && !result.Truncated + reason := "" + if !result.Complete { + reason = "one or more candidate source files could not be indexed completely" + } + if result.Truncated { + reason = "candidate audit reached the 10000-candidate cap" + } + if !source.coverage.SourceArchiveComplete { + complete = false + if reason == "" { + reason = source.coverage.SourceArchiveIncompleteReason + } + } + if includesText && source.textCandidateIndex.Status != "complete" { + complete = false + if reason == "" { + reason = source.textCandidateIndex.Reason + } + } + status := "complete" + if !complete { + status = "partial" + } + return CandidatePoolAudit{ + Status: status, Complete: complete, + CandidateCount: result.CandidateCount, CandidatesObserved: len(candidates), + CandidateLimit: candidatePoolAuditLimit, GoldSpans: len(gold), HitGoldSpans: hitCount, + Recall: ratio(hitCount, len(gold)), IncompleteReason: reason, + } +} + +func studyCandidates(candidates []retrieval.Candidate) []Candidate { + result := make([]Candidate, 0, len(candidates)) + for _, candidate := range candidates { + result = append(result, Candidate{ + Path: candidate.Path, Line: candidate.Line, Name: candidate.Name, + Kind: candidate.Kind, Container: candidate.Package, + }) + } + return result +} + +func profileSeries(profiles []retrieval.SearchProfile) RetrievalProfile { + series := RetrievalProfile{ + Samples: make([]RetrievalProfileSample, 0, len(profiles)), + } + searchMS, parseMS, aggregateMS, rankMS := make([]float64, 0, len(profiles)), make([]float64, 0, len(profiles)), make([]float64, 0, len(profiles)), make([]float64, 0, len(profiles)) + for _, profile := range profiles { + search := float64(profile.SearchDuration) / float64(time.Millisecond) + parse := float64(profile.ParseDuration) / float64(time.Millisecond) + aggregate := float64(profile.AggregateDuration) / float64(time.Millisecond) + rank := float64(profile.RankDuration) / float64(time.Millisecond) + series.Samples = append(series.Samples, RetrievalProfileSample{ + SearchMS: search, ParseMS: parse, AggregateMS: aggregate, RankMS: rank, + FileVisits: profile.FileVisits, CacheHits: profile.CacheHits, FilesParsed: profile.FilesParsed, + }) + searchMS, parseMS, aggregateMS, rankMS = append(searchMS, search), append(parseMS, parse), append(aggregateMS, aggregate), append(rankMS, rank) + } + series.Search, series.Parse = latency(searchMS), latency(parseMS) + series.Aggregate, series.Rank = latency(aggregateMS), latency(rankMS) + return series +} + +func heapSample(peak *uint64) { + var stats runtime.MemStats + runtime.ReadMemStats(&stats) + if stats.HeapAlloc > *peak { + *peak = stats.HeapAlloc + } +} + +func unavailableRanking(reason string) Ranking { + return Ranking{ + Status: "unavailable", Complete: false, IncompleteReason: reason, + Candidates: []Candidate{}, CandidateCountComplete: false, + Granularity: retrievalGranularity, + Metrics: Metrics{}, Misses: EvidenceMisses{ByTypeAt5: map[string]int{}, ByTypeAt10: map[string]int{}}, + } +} + +func summarize(queries []QueryResult) (Summary, error) { + retrievalRankings := make([]Ranking, 0, len(queries)) + nativeRankings := make([]Ranking, 0, len(queries)) + goplsRankings := make([]Ranking, 0, len(queries)) + coldSamples := map[string][]float64{"5": {}, "10": {}} + warmSamples := map[string][]float64{"5": {}, "10": {}} + coldStageSamples := map[string][]RetrievalProfileSample{"5": {}, "10": {}} + warmStageSamples := map[string][]RetrievalProfileSample{"5": {}, "10": {}} + var nativeTotal, nativeGather, nativeRanking, goplsQuery []float64 + for _, query := range queries { + retrievalRankings = append(retrievalRankings, query.Retrieval) + nativeRankings = append(nativeRankings, query.NativeRG) + if query.Gopls.Status != "unavailable" { + goplsRankings = append(goplsRankings, query.Gopls) + } + for _, limit := range []string{"5", "10"} { + coldSamples[limit] = append(coldSamples[limit], query.RetrievalTimings.Cold[limit].Samples...) + warmSamples[limit] = append(warmSamples[limit], query.RetrievalTimings.Warm[limit].Samples...) + coldStageSamples[limit] = append(coldStageSamples[limit], query.RetrievalProfiles.Cold[limit].Samples...) + warmStageSamples[limit] = append(warmStageSamples[limit], query.RetrievalProfiles.Warm[limit].Samples...) + } + nativeTotal = append(nativeTotal, query.NativeRGLatency.Samples...) + nativeGather = append(nativeGather, query.NativeRGCommandLatency.Samples...) + nativeRanking = append(nativeRanking, query.NativeRGRankingLatency.Samples...) + goplsQuery = append(goplsQuery, query.GoplsQueryLatency.Samples...) + } + summary := Summary{ + Retrieval: ArmSummary{At5: aggregateRankings(retrievalRankings, 5), At10: aggregateRankings(retrievalRankings, 10)}, + NativeRG: ArmSummary{At5: aggregateRankings(nativeRankings, 5), At10: aggregateRankings(nativeRankings, 10)}, + Gopls: ArmSummary{At5: aggregateRankings(goplsRankings, 5), At10: aggregateRankings(goplsRankings, 10)}, + RetrievalColdP50MS: make(map[string]float64), RetrievalColdP95MS: make(map[string]float64), + RetrievalWarmP50MS: make(map[string]float64), RetrievalWarmP95MS: make(map[string]float64), + RetrievalStages: RetrievalStageSummary{Cold: make(map[string]RetrievalStageLatencySummary), Warm: make(map[string]RetrievalStageLatencySummary)}, + } + for _, limit := range []string{"5", "10"} { + cold, warm := latency(coldSamples[limit]), latency(warmSamples[limit]) + summary.RetrievalColdP50MS[limit], summary.RetrievalColdP95MS[limit] = cold.P50MS, cold.P95MS + summary.RetrievalWarmP50MS[limit], summary.RetrievalWarmP95MS[limit] = warm.P50MS, warm.P95MS + summary.RetrievalStages.Cold[limit] = summarizeRetrievalStages(coldStageSamples[limit]) + summary.RetrievalStages.Warm[limit] = summarizeRetrievalStages(warmStageSamples[limit]) + } + nativeLatency := latency(nativeTotal) + nativeGatherLatency := latency(nativeGather) + nativeRankingLatency := latency(nativeRanking) + summary.NativeRGP50MS, summary.NativeRGP95MS = nativeLatency.P50MS, nativeLatency.P95MS + summary.NativeRGCommandP50MS, summary.NativeRGCommandP95MS = nativeGatherLatency.P50MS, nativeGatherLatency.P95MS + summary.NativeRGRankingP50MS, summary.NativeRGRankingP95MS = nativeRankingLatency.P50MS, nativeRankingLatency.P95MS + if len(goplsQuery) > 0 { + goplsLatency := latency(goplsQuery) + summary.GoplsQueryP50MS, summary.GoplsQueryP95MS = &goplsLatency.P50MS, &goplsLatency.P95MS + } + if math.IsNaN(summary.Retrieval.At5.MacroRecall) { + return Summary{}, fmt.Errorf("summary contains a non-finite metric") + } + return summary, nil +} + +func summarizeRetrievalStages(samples []RetrievalProfileSample) RetrievalStageLatencySummary { + search, parse, aggregate, rank := make([]float64, 0, len(samples)), make([]float64, 0, len(samples)), make([]float64, 0, len(samples)), make([]float64, 0, len(samples)) + for _, sample := range samples { + search = append(search, sample.SearchMS) + parse = append(parse, sample.ParseMS) + aggregate = append(aggregate, sample.AggregateMS) + rank = append(rank, sample.RankMS) + } + searchStats, parseStats := latency(search), latency(parse) + aggregateStats, rankStats := latency(aggregate), latency(rank) + return RetrievalStageLatencySummary{ + SearchP50MS: searchStats.P50MS, SearchP95MS: searchStats.P95MS, + ParseP50MS: parseStats.P50MS, ParseP95MS: parseStats.P95MS, + AggregateP50MS: aggregateStats.P50MS, AggregateP95MS: aggregateStats.P95MS, + RankP50MS: rankStats.P50MS, RankP95MS: rankStats.P95MS, + } +} + +func writeReport(filename string, report Report) error { + if err := os.MkdirAll(filepath.Dir(filename), 0o755); err != nil { + return fmt.Errorf("creating result directory: %w", err) + } + file, err := os.CreateTemp(filepath.Dir(filename), ".retrieval-report-*.tmp") + if err != nil { + return fmt.Errorf("creating result file: %w", err) + } + temporaryName := file.Name() + defer os.Remove(temporaryName) + if err := file.Chmod(0o644); err != nil { + _ = file.Close() + return fmt.Errorf("setting result file permissions: %w", err) + } + encoder := json.NewEncoder(file) + encoder.SetIndent("", " ") + if err := encoder.Encode(report); err != nil { + _ = file.Close() + return fmt.Errorf("encoding result report: %w", err) + } + if err := file.Sync(); err != nil { + _ = file.Close() + return fmt.Errorf("syncing result report: %w", err) + } + if err := file.Close(); err != nil { + return fmt.Errorf("closing result report: %w", err) + } + if err := os.Rename(temporaryName, filename); err != nil { + return fmt.Errorf("publishing result report: %w", err) + } + return nil +} diff --git a/validation/internal/retrievalstudy/run_test.go b/validation/internal/retrievalstudy/run_test.go new file mode 100644 index 0000000..e89732a --- /dev/null +++ b/validation/internal/retrievalstudy/run_test.go @@ -0,0 +1,41 @@ +package retrievalstudy + +import ( + "testing" + + "github.com/agentic-mcps/go/internal/intelligence/retrieval" +) + +func TestCandidatePoolAuditSeparatesCoverageFromRanking(t *testing.T) { + gold := []GoldSpan{ + {Type: "documentation", Path: "docs/contracts.md", StartLine: 10, EndLine: 12}, + {Type: "test", Path: "focus_test.go", StartLine: 20, EndLine: 22}, + } + source := archivedSource{ + coverage: Coverage{SourceArchiveComplete: true}, + textCandidateIndex: TextCandidateIndexCoverage{Status: "complete"}, + } + result := retrieval.Result{ + Candidates: []retrieval.Candidate{{Path: "docs/contracts.md", Line: 11, Kind: "text.line"}}, + CandidateCount: 1, Complete: true, + } + audit := candidatePoolAudit(result, gold, source, true) + if !audit.Complete || audit.HitGoldSpans != 1 || audit.GoldSpans != 2 || audit.Recall != 0.5 { + t.Fatalf("candidate pool audit = %+v, want complete 1/2 coverage", audit) + } + if audit.CandidateCount != 1 || audit.CandidatesObserved != 1 { + t.Fatalf("candidate pool counts = (%d, %d), want (1, 1)", audit.CandidateCount, audit.CandidatesObserved) + } +} + +func TestCandidatePoolAuditMarksCappedResultsPartial(t *testing.T) { + source := archivedSource{coverage: Coverage{SourceArchiveComplete: true}} + result := retrieval.Result{ + Candidates: []retrieval.Candidate{{Path: "docs/contracts.md", Line: 11}}, + CandidateCount: candidatePoolAuditLimit + 1, Complete: true, Truncated: true, + } + audit := candidatePoolAudit(result, []GoldSpan{{Type: "documentation", Path: "docs/contracts.md", StartLine: 10, EndLine: 12}}, source, false) + if audit.Complete || audit.Status != "partial" || audit.IncompleteReason == "" { + t.Fatalf("capped audit = %+v, want explicit partial status", audit) + } +} diff --git a/validation/internal/retrievalstudy/types.go b/validation/internal/retrievalstudy/types.go new file mode 100644 index 0000000..fa4926d --- /dev/null +++ b/validation/internal/retrievalstudy/types.go @@ -0,0 +1,335 @@ +// Package retrievalstudy implements a deterministic, model-free retrieval +// screen over immutable Git archives. +package retrievalstudy + +import "time" + +const ManifestVersion = "agentic-go.retrieval-study/v1" + +type Manifest struct { + Version string `json:"version"` + RepositoryID string `json:"repository_id"` + Commit string `json:"commit"` + Tree string `json:"tree"` + Stratum string `json:"stratum"` + Queries []Query `json:"queries"` +} + +type Query struct { + ID string `json:"id"` + Text string `json:"query"` + GoplsQuery string `json:"gopls_query,omitempty"` + Gold []GoldSpan `json:"gold"` +} + +type GoldSpan struct { + Type string `json:"type"` + Path string `json:"path"` + StartLine int `json:"start_line"` + EndLine int `json:"end_line"` + Label string `json:"label,omitempty"` +} + +type Repository struct { + ID string `json:"id"` + Commit string `json:"commit"` + Tree string `json:"tree"` +} + +type Reproducibility struct { + StudySourceCommit string `json:"study_source_commit"` + DirtyDiffSHA256 string `json:"dirty_diff_sha256"` + ManifestSHA256 string `json:"manifest_sha256"` + RetrievalSourceSHA256 string `json:"retrieval_source_sha256"` +} + +type Coverage struct { + TrackedRegularFiles int `json:"tracked_regular_files"` + TrackedRegularBytes int64 `json:"tracked_regular_bytes"` + ExpectedSupportedSourceFiles int `json:"expected_supported_source_files"` + ExpectedSupportedSourceBytes int64 `json:"expected_supported_source_bytes"` + SupportedSourceFiles int `json:"supported_source_files"` + SupportedSourceBytes int64 `json:"supported_source_bytes"` + GoFiles int `json:"go_files"` + GoBytes int64 `json:"go_bytes"` + GoFilesWithDeclarations int `json:"go_files_with_declarations"` + GoParseIncompleteFiles int `json:"go_parse_incomplete_files"` + SupportedTextFiles int `json:"supported_text_files"` + SupportedTextBytes int64 `json:"supported_text_bytes"` + RejectedTextFiles int `json:"rejected_text_files"` + RejectedTextBytes int64 `json:"rejected_text_bytes"` + SymlinksNotIndexed int `json:"symlinks_not_indexed"` + OtherArchiveEntriesIgnored int `json:"other_archive_entries_ignored"` + ArchiveOmittedSourceFiles int `json:"archive_omitted_source_files"` + ArchiveOmittedSourceBytes int64 `json:"archive_omitted_source_bytes"` + ArchiveTransformedSourceFiles int `json:"archive_transformed_source_files"` + ArchiveTransformedSourceBytes int64 `json:"archive_transformed_source_bytes"` + SourceArchiveComplete bool `json:"source_archive_complete"` + SourceArchiveIncompleteReason string `json:"source_archive_incomplete_reason,omitempty"` + TextIndexedByRetrieval int `json:"text_indexed_by_retrieval"` +} + +type TextCandidateIndexCoverage struct { + Status string `json:"status"` + Reason string `json:"reason,omitempty"` + SupportedFiles int `json:"supported_files"` + SupportedBytes int64 `json:"supported_bytes"` + IndexedFiles int `json:"indexed_files"` + IndexedBytes int64 `json:"indexed_bytes"` + OmittedFiles int `json:"omitted_files"` + OmittedBytes int64 `json:"omitted_bytes"` + MaximumFiles int `json:"maximum_files"` + MaximumFileBytes int64 `json:"maximum_file_bytes"` + MaximumTotalBytes int64 `json:"maximum_total_bytes"` + MaximumFragments int `json:"maximum_fragments_per_query"` +} + +type GoPackageInventory struct { + Status string `json:"status"` + Command string `json:"command"` + PackageCount int `json:"package_count"` + PackagesWithErrors int `json:"packages_with_errors"` + LatencyMS float64 `json:"latency_ms"` + TimeoutMS int64 `json:"timeout_ms"` + OutputBytes int `json:"captured_output_bytes"` + OutputLimitBytes int `json:"captured_output_limit_bytes"` + OutputTruncated bool `json:"output_truncated"` + Error string `json:"error,omitempty"` + Coverage string `json:"coverage"` +} + +type Candidate struct { + Path string `json:"path"` + Line int `json:"line"` + Name string `json:"name,omitempty"` + Kind string `json:"kind,omitempty"` + Container string `json:"container,omitempty"` + ProviderKind int `json:"provider_kind,omitempty"` + MatchedUniqueTerms int `json:"matched_unique_terms,omitempty"` + MatchedOccurrences int `json:"matched_occurrences,omitempty"` +} + +type CutoffMetrics struct { + RetrievedCandidates int `json:"retrieved_candidates"` + RelevantCandidates int `json:"relevant_candidates"` + GoldSpans int `json:"gold_spans"` + HitGoldSpans int `json:"hit_gold_spans"` + Recall float64 `json:"recall"` + Precision float64 `json:"precision"` + MeanReciprocalRank float64 `json:"mean_reciprocal_rank"` +} + +type Metrics struct { + At5 CutoffMetrics `json:"at_5"` + At10 CutoffMetrics `json:"at_10"` +} + +type EvidenceMisses struct { + GoldSpans int `json:"gold_spans"` + At5 int `json:"missed_at_5"` + At10 int `json:"missed_at_10"` + ByTypeAt5 map[string]int `json:"missed_at_5_by_type"` + ByTypeAt10 map[string]int `json:"missed_at_10_by_type"` +} + +type Ranking struct { + Status string `json:"status"` + Complete bool `json:"complete"` + MetricsUsable bool `json:"metrics_usable"` + IncompleteReason string `json:"incomplete_reason,omitempty"` + Metrics Metrics `json:"metrics"` + Misses EvidenceMisses `json:"misses"` + Candidates []Candidate `json:"candidates"` + CandidateCount int `json:"candidate_count"` + CandidateCountComplete bool `json:"candidate_count_complete"` + Granularity string `json:"candidate_granularity"` + Tokens []string `json:"tokens,omitempty"` +} + +type CandidatePoolAudit struct { + Status string `json:"status"` + Complete bool `json:"complete"` + CandidateCount int `json:"candidate_count"` + CandidatesObserved int `json:"candidates_observed"` + CandidateLimit int `json:"candidate_limit"` + GoldSpans int `json:"gold_spans"` + HitGoldSpans int `json:"hit_gold_spans"` + Recall float64 `json:"recall"` + IncompleteReason string `json:"incomplete_reason,omitempty"` +} + +type TextCandidateResult struct { + Ranking Ranking `json:"ranking"` + CandidatePool CandidatePoolAudit `json:"candidate_pool"` + Latency Latency `json:"latency"` + Profile RetrievalProfile `json:"profile"` + IndexedTextFragments int `json:"indexed_text_fragments"` + SkippedTextFiles int `json:"skipped_text_files"` + MaximumTextFragments int `json:"maximum_text_fragments"` +} + +type Latency struct { + Samples []float64 `json:"samples_ms"` + P50MS float64 `json:"p50_ms"` + P95MS float64 `json:"p95_ms"` + MinMS float64 `json:"min_ms"` + MaxMS float64 `json:"max_ms"` +} + +type RetrievalTimings struct { + Cold map[string]Latency `json:"cold_by_limit"` + Warm map[string]Latency `json:"warm_by_limit"` +} + +type RetrievalProfileSample struct { + SearchMS float64 `json:"search_ms"` + ParseMS float64 `json:"parse_ms"` + AggregateMS float64 `json:"aggregate_ms"` + RankMS float64 `json:"rank_ms"` + FileVisits int `json:"file_visits"` + CacheHits int `json:"cache_hits"` + FilesParsed int `json:"files_parsed"` +} + +type RetrievalProfile struct { + Samples []RetrievalProfileSample `json:"samples"` + Search Latency `json:"search"` + Parse Latency `json:"parse"` + Aggregate Latency `json:"aggregate"` + Rank Latency `json:"rank"` +} + +type RetrievalProfiles struct { + Cold map[string]RetrievalProfile `json:"cold_by_limit"` + Warm map[string]RetrievalProfile `json:"warm_by_limit"` +} + +type GoplsMeasurement struct { + Status string `json:"status"` + Version string `json:"version,omitempty"` + UnavailableReason string `json:"unavailable_reason,omitempty"` + CompletenessNote string `json:"completeness_note,omitempty"` + InitializeMS *float64 `json:"initialize_ms,omitempty"` + QueryLatency Latency `json:"query_latency"` + Granularity string `json:"candidate_granularity"` +} + +type QueryResult struct { + ID string `json:"id"` + Text string `json:"query"` + Gold []GoldSpan `json:"gold"` + Retrieval Ranking `json:"retrieval"` + RetrievalTimings RetrievalTimings `json:"retrieval_timings"` + RetrievalProfiles RetrievalProfiles `json:"retrieval_profiles"` + RetrievalCandidatePool *CandidatePoolAudit `json:"retrieval_candidate_pool,omitempty"` + TextCandidateAblation *TextCandidateResult `json:"text_candidate_ablation,omitempty"` + NativeRG Ranking `json:"native_rg"` + NativeRGLatency Latency `json:"native_rg_latency"` + NativeRGCommandLatency Latency `json:"native_rg_candidate_gather_latency"` + NativeRGRankingLatency Latency `json:"native_rg_ranking_latency"` + GoplsQuery string `json:"gopls_query,omitempty"` + Gopls Ranking `json:"gopls,omitempty"` + GoplsQueryLatency Latency `json:"gopls_query_latency,omitempty"` +} + +type AggregateMetrics struct { + Queries int `json:"queries"` + MetricsQueries int `json:"metrics_queries"` + CompleteQueries int `json:"complete_queries"` + MacroRecall float64 `json:"macro_recall"` + MacroPrecision float64 `json:"macro_precision"` + MacroMeanReciprocalRank float64 `json:"macro_mean_reciprocal_rank"` + MicroRecall float64 `json:"micro_recall"` + MicroPrecision float64 `json:"micro_precision"` +} + +type ArmSummary struct { + At5 AggregateMetrics `json:"at_5"` + At10 AggregateMetrics `json:"at_10"` +} + +type RetrievalStageLatencySummary struct { + SearchP50MS float64 `json:"search_p50_ms"` + SearchP95MS float64 `json:"search_p95_ms"` + ParseP50MS float64 `json:"parse_p50_ms"` + ParseP95MS float64 `json:"parse_p95_ms"` + AggregateP50MS float64 `json:"aggregate_p50_ms"` + AggregateP95MS float64 `json:"aggregate_p95_ms"` + RankP50MS float64 `json:"rank_p50_ms"` + RankP95MS float64 `json:"rank_p95_ms"` +} + +type RetrievalStageSummary struct { + Cold map[string]RetrievalStageLatencySummary `json:"cold_by_limit"` + Warm map[string]RetrievalStageLatencySummary `json:"warm_by_limit"` +} + +type Summary struct { + Retrieval ArmSummary `json:"retrieval"` + NativeRG ArmSummary `json:"native_rg"` + Gopls ArmSummary `json:"gopls"` + GoplsStatus string `json:"gopls_status"` + RetrievalStages RetrievalStageSummary `json:"retrieval_stages"` + RetrievalColdP50MS map[string]float64 `json:"retrieval_cold_p50_ms_by_limit"` + RetrievalColdP95MS map[string]float64 `json:"retrieval_cold_p95_ms_by_limit"` + RetrievalWarmP50MS map[string]float64 `json:"retrieval_warm_p50_ms_by_limit"` + RetrievalWarmP95MS map[string]float64 `json:"retrieval_warm_p95_ms_by_limit"` + NativeRGP50MS float64 `json:"native_rg_p50_ms"` + NativeRGP95MS float64 `json:"native_rg_p95_ms"` + NativeRGCommandP50MS float64 `json:"native_rg_candidate_gather_p50_ms"` + NativeRGCommandP95MS float64 `json:"native_rg_candidate_gather_p95_ms"` + NativeRGRankingP50MS float64 `json:"native_rg_ranking_p50_ms"` + NativeRGRankingP95MS float64 `json:"native_rg_ranking_p95_ms"` + GoplsQueryP50MS *float64 `json:"gopls_query_p50_ms,omitempty"` + GoplsQueryP95MS *float64 `json:"gopls_query_p95_ms,omitempty"` +} + +type HeapStats struct { + StartHeapAllocBytes uint64 `json:"start_heap_alloc_bytes"` + EndHeapAllocBytes uint64 `json:"end_heap_alloc_bytes"` + PeakSampledHeapAllocBytes uint64 `json:"peak_sampled_heap_alloc_bytes"` + EndHeapSysBytes uint64 `json:"end_heap_sys_bytes"` + TotalAllocBytes uint64 `json:"total_alloc_bytes"` + NumGC uint32 `json:"num_gc"` +} + +type Report struct { + SchemaVersion string `json:"schema_version"` + CreatedUTC time.Time `json:"created_utc"` + Repository Repository `json:"repository"` + Reproducibility Reproducibility `json:"reproducibility"` + Stratum string `json:"stratum"` + GoVersion string `json:"go_version"` + GOOS string `json:"goos"` + GOARCH string `json:"goarch"` + RGVersion string `json:"rg_version"` + SnapshotExportMS float64 `json:"snapshot_export_ms"` + RetrievalTimingNote string `json:"retrieval_timing_note"` + NativeRGWorkflow NativeWorkflow `json:"native_rg_workflow"` + GoPackages GoPackageInventory `json:"go_package_inventory"` + Coverage Coverage `json:"coverage"` + TextCandidateIndex *TextCandidateIndexCoverage `json:"text_candidate_index,omitempty"` + Configuration Configuration `json:"configuration"` + Gopls GoplsMeasurement `json:"gopls"` + Queries []QueryResult `json:"queries"` + Summary Summary `json:"summary"` + Heap HeapStats `json:"heap"` +} + +type NativeWorkflow struct { + CommandTemplate string `json:"command_template"` + Tokenizer string `json:"tokenizer"` + Ranking []string `json:"ranking_order"` + Globs []string `json:"globs"` + PathScope string `json:"path_scope"` +} + +type Configuration struct { + Repetitions int `json:"repetitions"` + CommandTimeoutMS int64 `json:"command_timeout_ms"` + SearchTimeoutMS int64 `json:"search_timeout_ms"` + MaxSourceBytes int64 `json:"max_source_bytes"` + MaxFileBytes int64 `json:"max_file_bytes"` + NativeOutputLimit int64 `json:"native_rg_output_limit_bytes"` + NativeLineLimit int `json:"native_rg_line_limit"` +} diff --git a/validation/retrieval/heldout-v1/large-kubernetes.json b/validation/retrieval/heldout-v1/large-kubernetes.json new file mode 100644 index 0000000..6d85185 --- /dev/null +++ b/validation/retrieval/heldout-v1/large-kubernetes.json @@ -0,0 +1,54 @@ +{ + "version": "agentic-go.retrieval-study/v1", + "repository_id": "github.com/kubernetes/kubernetes", + "commit": "dfd7b93a1783878be367e1fc4a780318330cb3bf", + "tree": "dcd107007aacbebccbccd6b3ec0ffc4a3adce2c1", + "stratum": "large", + "queries": [ + { + "id": "large-1", + "query": "How does a Deployment define rollout progress, report a timed out rollout, and schedule a later progress check?", + "gopls_query": "syncRolloutStatus", + "gold": [ + {"type": "declaration", "path": "staging/src/k8s.io/api/apps/v1/types.go", "start_line": 419, "end_line": 464, "label": "DeploymentSpec and progress deadline field"}, + {"type": "declaration", "path": "pkg/controller/deployment/progress.go", "start_line": 36, "end_line": 95, "label": "rollout status and timeout handling"}, + {"type": "declaration", "path": "pkg/controller/deployment/util/deployment_util.go", "start_line": 770, "end_line": 809, "label": "DeploymentTimedOut"}, + {"type": "declaration", "path": "pkg/controller/deployment/progress.go", "start_line": 160, "end_line": 199, "label": "stuck deployment requeue"}, + {"type": "test", "path": "pkg/controller/deployment/progress_test.go", "start_line": 196, "end_line": 363, "label": "rollout status tests"} + ] + }, + { + "id": "large-2", + "query": "After a Pod fails scheduling, how can a cluster event move it to the active queue, backoff queue, or leave it unschedulable?", + "gopls_query": "QueueingHint", + "gold": [ + {"type": "declaration", "path": "staging/src/k8s.io/kube-scheduler/framework/types.go", "start_line": 197, "end_line": 249, "label": "queueing hint API"}, + {"type": "declaration", "path": "pkg/scheduler/backend/queue/scheduling_queue.go", "start_line": 583, "end_line": 679, "label": "event hint requeue decision"}, + {"type": "declaration", "path": "pkg/scheduler/backend/queue/scheduling_queue.go", "start_line": 1811, "end_line": 1852, "label": "move to active or backoff queue"}, + {"type": "test", "path": "pkg/scheduler/backend/queue/scheduling_queue_test.go", "start_line": 2480, "end_line": 2658, "label": "queueing hint outcomes"} + ] + }, + { + "id": "large-3", + "query": "When context is canceled while a client go controller is watching, which loops stop and is the underlying watch closed?", + "gopls_query": "RunWithContext", + "gold": [ + {"type": "declaration", "path": "staging/src/k8s.io/client-go/tools/cache/controller.go", "start_line": 126, "end_line": 142, "label": "Controller cancellation contract"}, + {"type": "caller", "path": "staging/src/k8s.io/client-go/tools/cache/controller.go", "start_line": 172, "end_line": 209, "label": "controller context propagation"}, + {"type": "declaration", "path": "staging/src/k8s.io/client-go/tools/cache/reflector.go", "start_line": 423, "end_line": 434, "label": "Reflector.RunWithContext"}, + {"type": "declaration", "path": "staging/src/k8s.io/client-go/tools/cache/reflector.go", "start_line": 561, "end_line": 620, "label": "watch cancellation and stop"}, + {"type": "test", "path": "staging/src/k8s.io/client-go/tools/cache/reflector_test.go", "start_line": 88, "end_line": 202, "label": "watch cancellation tests"} + ] + }, + { + "id": "large-4", + "query": "When a concurrent resource update conflicts, what must the retry callback do on each attempt, and which errors cause another attempt?", + "gopls_query": "RetryOnConflict", + "gold": [ + {"type": "declaration", "path": "staging/src/k8s.io/client-go/util/retry/util.go", "start_line": 48, "end_line": 105, "label": "retry policy and RetryOnConflict"}, + {"type": "test", "path": "staging/src/k8s.io/client-go/util/retry/util_test.go", "start_line": 30, "end_line": 73, "label": "retry error classification tests"}, + {"type": "caller", "path": "staging/src/k8s.io/client-go/examples/create-update-delete-deployment/main.go", "start_line": 46, "end_line": 142, "label": "fetch and update callback in example main"} + ] + } + ] +} diff --git a/validation/retrieval/heldout-v1/medium-agentic-go.json b/validation/retrieval/heldout-v1/medium-agentic-go.json new file mode 100644 index 0000000..71331d8 --- /dev/null +++ b/validation/retrieval/heldout-v1/medium-agentic-go.json @@ -0,0 +1,57 @@ +{ + "version": "agentic-go.retrieval-study/v1", + "repository_id": "github.com/agentic-mcps/go", + "commit": "7b5111c6365a2a12806a561d86745d5bc50d7c9e", + "tree": "17a298ebadc7ca635b91248e03cff18d2ea4b4e9", + "stratum": "medium", + "queries": [ + { + "id": "medium-1", + "query": "After I edit and reuse a prior context pack, how does the system resolve its declaration against the new observation, and when must I select again instead of treating unresolved evidence as deletion?", + "gopls_query": "resolveRefreshSymbol", + "gold": [ + {"type": "declaration", "path": "internal/intelligence/focus.go", "start_line": 193, "end_line": 280, "label": "Core.Focus refresh handling"}, + {"type": "declaration", "path": "internal/intelligence/focus.go", "start_line": 281, "end_line": 311, "label": "refresh symbol resolution"}, + {"type": "test", "path": "internal/intelligence/focus_test.go", "start_line": 84, "end_line": 140, "label": "refresh and unresolved evidence tests"}, + {"type": "documentation", "path": "docs/contracts.md", "start_line": 1154, "end_line": 1163, "label": "focus refresh contract"}, + {"type": "caller", "path": "internal/tools/intelligence_tools.go", "start_line": 155, "end_line": 195, "label": "MCP Runtime.context mapping"} + ] + }, + { + "id": "medium-2", + "query": "If a workspace symbol query matches several declarations or the provider omits locations, what does go_context return before expanding relationships?", + "gopls_query": "focusContext", + "gold": [ + {"type": "declaration", "path": "internal/intelligence/focus.go", "start_line": 86, "end_line": 103, "label": "FocusContext declaration"}, + {"type": "declaration", "path": "internal/intelligence/focus.go", "start_line": 377, "end_line": 552, "label": "query selection and candidate bounds"}, + {"type": "test", "path": "internal/intelligence/focus_test.go", "start_line": 280, "end_line": 326, "label": "ambiguous and omitted symbol tests"}, + {"type": "documentation", "path": "docs/contracts.md", "start_line": 1142, "end_line": 1152, "label": "selector and candidate contract"}, + {"type": "caller", "path": "internal/tools/intelligence_tools.go", "start_line": 155, "end_line": 195, "label": "MCP Runtime.context mapping"} + ] + }, + { + "id": "medium-3", + "query": "After selecting a function, how does focused context gather direct callers and referenced test declarations, and what should an agent infer from those links?", + "gopls_query": "focusTestDeclarations", + "gold": [ + {"type": "declaration", "path": "internal/intelligence/focus.go", "start_line": 377, "end_line": 552, "label": "focused context and incoming call evidence"}, + {"type": "declaration", "path": "internal/intelligence/focus.go", "start_line": 554, "end_line": 656, "label": "symbol relationship gathering"}, + {"type": "declaration", "path": "internal/intelligence/focus.go", "start_line": 658, "end_line": 693, "label": "test declaration resolution"}, + {"type": "test", "path": "internal/intelligence/focus_test.go", "start_line": 337, "end_line": 359, "label": "direct call site context test"}, + {"type": "documentation", "path": "docs/go-intelligence-north-star.md", "start_line": 213, "end_line": 240, "label": "relationship evidence limits"} + ] + }, + { + "id": "medium-4", + "query": "How does a context request keep source bytes tied to one observation and reject a worktree change that occurs before it returns?", + "gopls_query": "observe", + "gold": [ + {"type": "declaration", "path": "internal/intelligence/snapshot.go", "start_line": 122, "end_line": 144, "label": "captured source digest check"}, + {"type": "declaration", "path": "internal/intelligence/snapshot.go", "start_line": 202, "end_line": 228, "label": "double capture observation"}, + {"type": "declaration", "path": "internal/intelligence/snapshot.go", "start_line": 231, "end_line": 246, "label": "snapshot validation"}, + {"type": "test", "path": "internal/intelligence/snapshot_test.go", "start_line": 95, "end_line": 137, "label": "stale reference and source digest tests"}, + {"type": "declaration", "path": "internal/intelligence/focus.go", "start_line": 193, "end_line": 280, "label": "final focus validation"} + ] + } + ] +} diff --git a/validation/retrieval/heldout-v1/small-tour.json b/validation/retrieval/heldout-v1/small-tour.json new file mode 100644 index 0000000..06b0802 --- /dev/null +++ b/validation/retrieval/heldout-v1/small-tour.json @@ -0,0 +1,45 @@ +{ + "version": "agentic-go.retrieval-study/v1", + "repository_id": "local/go-1.27-interactive-tour", + "commit": "61ca6068c067756caf8d98b0a933163b39894416", + "tree": "4ad27ca4c31c77e082a0b66063499e62a51dea2b", + "stratum": "small", + "queries": [ + { + "id": "small-1", + "query": "What does the UUID time ordering example actually execute, and where does the project explain why it cares about sorted inserts?", + "gopls_query": "main", + "gold": [ + {"type": "declaration", "path": "examples/02_uuid_v7/main.go", "start_line": 9, "end_line": 20, "label": "UUID example main declaration"}, + {"type": "documentation", "path": "README.md", "start_line": 23, "end_line": 24, "label": "clustered-index motivation"} + ] + }, + { + "id": "small-2", + "query": "In the payment webhook security example, what duplicate key payload and rejection behavior are shown, and which JSON feature does the tour say this covers?", + "gopls_query": "main", + "gold": [ + {"type": "declaration", "path": "examples/03_json_v2/main.go", "start_line": 8, "end_line": 15, "label": "JSON example main declaration"}, + {"type": "documentation", "path": "README.md", "start_line": 25, "end_line": 26, "label": "JSON v2 security topic"} + ] + }, + { + "id": "small-3", + "query": "What happens in the goroutine leak example if the named profiler is unavailable, and what capability does the README say it demonstrates?", + "gopls_query": "main", + "gold": [ + {"type": "declaration", "path": "examples/05_goroutine_leak/main.go", "start_line": 8, "end_line": 15, "label": "goroutine example main declaration"}, + {"type": "documentation", "path": "README.md", "start_line": 19, "end_line": 20, "label": "profiler capability"} + ] + }, + { + "id": "small-4", + "query": "How does the database row and column example convert a string value and handle a conversion failure?", + "gopls_query": "main", + "gold": [ + {"type": "declaration", "path": "examples/17_db_row_column_scanner/main.go", "start_line": 11, "end_line": 21, "label": "database example main declaration"}, + {"type": "documentation", "path": "README.md", "start_line": 37, "end_line": 38, "label": "database scanning topic"} + ] + } + ] +} diff --git a/validation/retrieval/heldout-v2/medium-agentic-go-text.json b/validation/retrieval/heldout-v2/medium-agentic-go-text.json new file mode 100644 index 0000000..505a800 --- /dev/null +++ b/validation/retrieval/heldout-v2/medium-agentic-go-text.json @@ -0,0 +1,48 @@ +{ + "version": "agentic-go.retrieval-study/v1", + "repository_id": "github.com/agentic-mcps/go", + "commit": "7b5111c6365a2a12806a561d86745d5bc50d7c9e", + "tree": "17a298ebadc7ca635b91248e03cff18d2ea4b4e9", + "stratum": "medium", + "queries": [ + { + "id": "medium-text-1", + "query": "Which public column unit is exposed in context results, and where does the adapter translate the language server's coordinate unit?", + "gold": [ + {"type": "documentation", "path": "docs/contracts.md", "start_line": 118, "end_line": 124, "label": "public coordinate and LSP adapter contract"}, + {"type": "documentation", "path": "docs/adr/0002-context-pack-boundary.md", "start_line": 8, "end_line": 12, "label": "snapshot and coordinate boundary"}, + {"type": "declaration", "path": "internal/gopls/position.go", "start_line": 9, "end_line": 17, "label": "UTF-16 position type and byte offset conversion"}, + {"type": "declaration", "path": "internal/gopls/position.go", "start_line": 41, "end_line": 42, "label": "LSP position to byte offset conversion"} + ] + }, + { + "id": "medium-text-2", + "query": "Which positive and negative evidence must accompany each analyzer rule before it can be trusted as precise?", + "gold": [ + {"type": "documentation", "path": "docs/phase-4a-index.md", "start_line": 86, "end_line": 94, "label": "positive and near-miss fixture requirements"}, + {"type": "documentation", "path": "docs/phase-4a-index.md", "start_line": 108, "end_line": 114, "label": "per-rule positive and near-miss verification assertions"}, + {"type": "documentation", "path": "docs/v0.9.0-release-scope.md", "start_line": 135, "end_line": 137, "label": "rule release evidence gate"}, + {"type": "documentation", "path": "docs/plan.md", "start_line": 79, "end_line": 82, "label": "predicate fixture and calibration gate"} + ] + }, + { + "id": "medium-text-3", + "query": "How can a client page through the complete overflow detail after a context response reaches its byte limit, and how does each cursor continue the artifact?", + "gold": [ + {"type": "documentation", "path": "docs/contracts.md", "start_line": 126, "end_line": 132, "label": "bounded content-addressed artifact pagination"}, + {"type": "documentation", "path": "docs/adr/0002-context-pack-boundary.md", "start_line": 14, "end_line": 17, "label": "compact response overflow and opaque cursors"}, + {"type": "declaration", "path": "internal/intelligence/artifact.go", "start_line": 193, "end_line": 193, "label": "opaque cursor encoding"}, + {"type": "declaration", "path": "internal/intelligence/artifact.go", "start_line": 205, "end_line": 205, "label": "opaque cursor validation"}, + {"type": "declaration", "path": "internal/intelligence/artifact_read.go", "start_line": 39, "end_line": 91, "label": "ReadChunk consumes the cursor offset and emits the next cursor"} + ] + }, + { + "id": "medium-text-4", + "query": "What does workspace containment guarantee about isolation when verification executes the repository's tests and analyzers?", + "gold": [ + {"type": "documentation", "path": "docs/contracts.md", "start_line": 86, "end_line": 90, "label": "verification trust boundary and containment limits"}, + {"type": "documentation", "path": "docs/v0.2.0-release-scope.md", "start_line": 475, "end_line": 479, "label": "subprocess containment and target-code trust boundary"} + ] + } + ] +} diff --git a/validation/retrieval/provenance/2026-09-27/README.md b/validation/retrieval/provenance/2026-09-27/README.md new file mode 100644 index 0000000..b2206a7 --- /dev/null +++ b/validation/retrieval/provenance/2026-09-27/README.md @@ -0,0 +1,25 @@ +# Source worktree preservation record + +Before reconciliation, both audited source worktrees were at base commit +`7b5111c6365a2a12806a561d86745d5bc50d7c9e`. The source worktrees remain +available at their original paths. This directory preserves each complete +tracked `git diff --binary` as a gzip archive, captured status, and SHA-256 +list for its untracked files. Decompress `v1_2_reliability.diff.gz` or +`decision_value_eval.diff.gz` to recover the original diff; the table's +SHA-256 is for those uncompressed bytes. + +| Source worktree | Branch | Tracked diff SHA-256 | +| --- | --- | --- | +| `/Users/ashwin/Desktop/projects/golang/agentic-mcps-go` | `codex/v1.2-reliability` | `387a368c2349345fe888e917fc00961583726cc120d9ef1d6948f9c537d5c6da` | +| `/Users/ashwin/Desktop/projects/golang/agentic-mcps-go-value-eval-2026-09-27` | `codex/decision-value-eval-2026-09-27` | `a34f2424ab1879399822d9c4eab05482c2c19c8a6e5072366ee932c36e3bca0a` | + +The second worktree's 72-run decision-value harness remains parked in that +source worktree. It is pinned to a different snapshot and GPT-6 Sol/high, so it +does not contribute retrieval evidence and is not part of this integration. +The repeated `decision-value-eval-2026-09-27-refresh` worktree was also left +untouched; it is not one of the two audit snapshots above. + +The saved SHA lists include untracked files from their respective source +worktrees. Validate an untracked file against the corresponding manifest before +reusing it. These records capture pre-reconciliation state and are not live +status reports. diff --git a/validation/retrieval/provenance/2026-09-27/decision_value_eval.diff.gz b/validation/retrieval/provenance/2026-09-27/decision_value_eval.diff.gz new file mode 100644 index 0000000000000000000000000000000000000000..9a6f1f614ff73ac28648c54640aec0538b30af4b GIT binary patch literal 39594 zcmV(&K;ge1iwFP!000021H`@SavN8cCi>fWiflQmTPkFNcS<6oI);{HTc>@|NOIMQ zP&>pVkRY>!+aME^Y0DAwAm_Z?Jjq$#y6wFa07c2}(;YJ|5lCe2+?RFz*0)CWc$^i* zW!+}w&Y--kW^J?6ysIaZW{3Z(hsEW*7|v&H^-H_+>OY=;_uXJR$}T?P;$A%)Rlj7% z505U656Z#dVE5tCQF)NYi8x5xth1d#e6jzWz}^(s%FC~&t`L4e0uHW zvKqFts%d)t;iRmm1A2jCZ;-ue%VnFruiLBa{Gx2Ca||V$&#J7cCiIi}@+3R|U#n_) zbDn4C&CT>;K6y(+qQ9E?YB{WW{jGPYekq6TA6{7ggEX51VpYWf#jieaCN24?h@WU)Qsm7PKmdSJ_3`(tqiPys$iNv@xb- z%lFV-qFaDI8DuYYJ7BVVeR`EwJd608Upt#GN_tjyansWC+a>*o_REv`;szs{+M*1y zAF6BGPoru$AJIO+#)8tumJFI!E+bZY%JDt)`c^LHFoX%5rT`ueLVuFzz-nO7aygIhv_Ep4f>!zVuD|sj~!@OcU z*lNCe2iem(A1PVpEo72b=g{AD=_gH9jOg^u@G3X{PFdMxi%B_~VXI!QXfA{LTeT$8>R;tKtte%_ zUpXGjIZ+kTobq?PQDy3 z$VmQP*CedMiK1iCRLg7F9>Z$c(%x=wXggN3-Y7QLUOqc}`u#Jp#{A!EY#o$`!$-sM zU_k$Qbi7}s##XpmY^m@!Or^bt`Ek~#|L*ORtpqFPJeWu4EIL)blJ)ZalNT8ecF9A3 z^4)V1v~6}tYP8`^{|jyR8C^?Sfvp(aiGDFH*;-O#heR-+D*n5kt!RC7TB33}qM4^d zQDR%u&vP{zs^JQ2lI#vl`g}!OczRKddp~Q49>ERkKVXl!B*|DVS2qc*dOfK&LO!pZ)n_mN4%ff`Fzrx@advA z`IW|bO_#2gv<)igp1ni4gkEvg@MYx)(?};KF`#jg%(w600o#K+*L&wMkZCI~SfMRR z3X)pNrscaz?#RZD$zxrT8`*wj%0Z0UY+&>xmgq=|#?aG?)px}KQw8Uq^(E>0VmP4} z9z`k|lhRaJ6Z$#ZxlQi>%-U*7%h{5&YU*ELB}^Ca8TfW#hK6Lkk&Ii?e@Xg8nu?C_ zmtSVjXHXN9REY3O#*!rcK*dRQJUV=QM84bZ{?Wywhx@5G>HI`wO6NlemPh%+hgtto ze#kx;ok*JeTBLafs{pe^d+&-&RLn974y<^ilg`E*4$}k+UC(;in{zDnl7vG;k2wGN ze*EVHnNvJvQn9eugRYoVuo7iE({x3$MWvwq(u#{_cH|R$=0A!}TH=RP_6((cwGkKQYoY45}h@h~UhG=idJE({6^c65~DvWTc*cYbcQXi6C3-Xaj1e4lr>q{Pi zY{@MW{G-B~EE(Fg>nDb!c=8IQI>efm+3qZ4qD()cNAX0F<_lI&qh67or%}CBTNVQ- zhr?>o!iLL+OS*`p0~x62qFK(f`*gUo^W%r(qr;1b=e|=RMVr8Xlq?X zt45<$?G-Qdxng*Jx0;nSgmNMq82<8my5DfJ8o>|SC(*o`!BS$USQjs&iJXh4P?Xu; zK3{_kc}YSDp73bS5kazBCS)%aE1JFMA}@JJB;BWZFtl!H385yIH@)ILsokrpoU~Ur z5V3o^JA1ns-JVWAi7j~Zcosr@^qkm6A%Ac(VWl1h(!;$-3_9VVYy=J48N@Tp!;lW;he7KiQX&b|>qB+>#Kgb^* zJ))1;ytCs&Pq=_wdVKH~QZAR|fS=kO*$TsO$)%>Z$@UM!&B&b;@eJ`A+p4}6yxtSc z^^_w`JXSneRp9iZp4qJMTe3qUwqh=7fDH}2)pZq4BlZsbX8g>|VU__m=3SGRC~2>f zH%_-K$_e5_!*bP>So@qsBTK0`a+;gIAeRm(Zh&T5piEak&r zPUu|Mmov8cS)qtxoTxv_KeGGIVIY(p)6<`Q@T5 zjt0lYgig(QMn3bj`~`Pu<};4bK=MWBz_&l&u~(4uQdO*R+FDNAg5;4sga4qDfl!Z# zCbd6JQn0LELKBeuf_e6iOj7qSiZ_uh;;Syffad4>gTW516v7_RQ(|`Q*ky%YX3j>h zm``BrU+)ec@wFZMn&u|*M`Y4Uj#pjMouGx&sCciCXbjCgc^&lsf_XzPUDKcuhIq-Y z8NTLYnjGZ}?@LdC{b!cw3B&}|C;Tud;E#0SqO$E&)C!597<;xdO zSh^g<=ilxjt^)L&mNGJei)tP#fp8(QxSb#UmIGK_9O=h zHSVSnt=@)6+aP-~q3z!yyTYPvN<+yGibEf!hzCR%MO4{K(X{ckhs9B@@}0eE3*Ba% zUb(6!i}7lru`KLRS#MUQG(b~Jer93Nut6}lSiam5LS6j!5D4}#ex8lHfl%6Tca4O~ zdo8yIEM-Y=jaaIK$ze+wC^d}+BIak?fDC;oc9I12A&rPVn-<2>OiqkFLi9XYN4F_@%OT3W)zxzg zjb0x?{A)F1ND}dBLmUF3U=i;_KcVuiAAoas-XwwF)-#d`=MtcmkLV30;Jw7^;aNHc z7PNdDdhg5SVyGuS+l%hYyZo)~((_(%QDNV4s9g3ZR?Py1umjvKDd}sD;mxOMfSxX{ z$Fz@WM?{Y$LxYEPK<*FExxBhsbv#f#7j%>N&AJuBzBb$?tR|m9j#%LgTIPmES6z}u zK!BK}=!QdO;=E@^D!lUB;biCQ5{J)F?K z@ zkz4vK76{p|)p9OkAft7flp4g~k4+5pidYWVx9b~L4w?-jD3-`~Ea)(aWm!&CBW^VE z_xAFmgNLx_ZJi^_orMKGXGQ0YuI?2FdwN!CxG3f2Yz`ZClMN*~65MmcX-IVm10#Z& z?s0Tcr5vLk^!j&+2*Ehk&~+4#$9+B_5;2aBoxAOMug^P55GN+XSxMy~8e%?SHoO^5 zRNQZjR3PUt^WlDWm7+sOx$MhG#x>oZuBTf~0S7?TBwACo)Z)An z>uj1*tPbRJ%K}+n3{%Gf#>G3k*G)GglMXCS%s9|yLyAUcY6mXKRQ-*aTHw5oct2=@ zfn3R6KLrAvy?DOkA#cluNfHwX2GMmA7$mDz_eUN8D80T%aFWQ+E0GgR(M?9$F(pHl3X4yc@K%iEz0r{CEg&A}bXVw{<7kp^CW<4)_ z2d!_J-aA}sIN;-I{P^NS*LoWF)H;b~&ce$K_E&1MV)~Ej-0oPJ|nQcxD z_WECD(V^a84sy-2v426L=&yezNi-r;=p_5$?B&ZJ{{ieV`66t;oMew4?T|zm{PnNdSF3t5LSPvX zyF9jrznS+o7`BWzirs;e?4q6}lf*G!hNTH+hvmYd2!_Ms{4szj7IS*bIw7~JaL+A?fc5A5S^RGYA!=F%xZIQS#t#`eX6&e3 zRgBt}&DqOViV6;~B-+k&Y-cj+IlV5Ko#Pu63D6Up)n7W7`#c;zPYL`wiAsVAsA6M zdTfBQ)Za@n2*Tl(Oa>lpsW?ol9uS0q6)D7vlo<nv{M7B~N*q-~ z8AsYA@UF3HJue!dgbGDs&?c^f7JL$|g9hEr)eMmFLH0dvT*L?2ok&|-AooeSSLP1O zSb|gB>$5e90A3qI^2w*DFK3MO4;u{<`nBnOg9c{Xkb%Lq@Q_9?+l)#SO9jz_JGdT$ z>`YBz7N`gN`97J92M759gPH3nKIK_937KKA|NKM zON&5F=9mI3#&;CrDisZ_@V!mSQ_IkPgAxQ0^SyWdY6N z;1_m7qBMV~HRm8?->nO&9A0oq5hr0tfM{x3F|7)~%#uv;4X-Z^C{ZplVRA-1r>KWz zZF^%Q+vv!Kz%rhx6Ksoy)J+uJ;xHj_2fpnzYNx4&)l7CQOK_`bjZ-utHW>yqLGDev%NfPlpU+JIVZ#fm)o_rUcM4}#;bUx^FaS;W+a<55z%dMff}1gR#L;js zD@kLi_-C`(lJt1Z7SJp`LRK0%yR-se(r~m3-vd&RWH)7o7#cA)-BP*0xw6?QDh&UEgrp8orHNB25 z4UlmkIuwT>x)|CiWHr6RM_hC_LxtI@W!u+d$TgphCIA^-Fs5=!Qj93QMFg~) zFX@JR)-NEl54GnCVVsr7jy@o9+i10&4GL-tB$jHT`jVoPX7K4d?dO0|qd_JJd|;-OSlP$GiyLoZ7p7Ye-K zQ+2c_r|Otb7k$L}*&CT-adQo75#S=6{d>dWR7@i~y&F_hZxdLqq|;h!&_Ju0I)FK` zi=*Um*qpKN=0QKs7$|m)eKyf*te-MlYm43CWIjn^Kn|&NX<;T+@DpTC*$@bppSWSD z=7Kv|{|ZDD1L7ur2V0^OH@O|x4>#(nlI17O-{gqR1+hsF61(lHUch#ENaG5@)8Ow5~6yNM2U4J~iWIa^r^$wIdsV>a1Vs)yWs&=u|p0WbuvE znO%>(IgpHla~d#;58xQ$Rk1HUbNnfiW`jLfAXJKdo##3fikyCS~l!{ZqUGl1h_$%9Lc{ zN_xOd!cNV$z?rI{nHkl#ec#v6Z+65qQusVgzVj-;5}aIDQ4Ip7t+9&DF~@yu7A8StM>*|N?q?HBFV2zY(3lH8$JXd9990&Ul77xWAYW)d%Y$WtG5T7I~+xe7b zD#BI;J-Jxav)qnoF=v(p{FAxs7Hkn0zMwv!*e67d5rL zb^x}+(n}TnW2xR$cUc2dmoq4q90b2iNS0 zyMPT1MJHZy-6)uUAaj4ftVR&LroldB&lqyeH8eX~UPqHN8L^(y_Z9O!8&)I@k!bMh zR`?@J1lyOA>1bL;x=RsJQO&N&s#=5+h5!MV02_&WE-zrHneM5WOG`$5xOT2hRDVEL zRi-?lnc?$xgMC9D_z2RWlEmw@YOm%{ix{W#@z8Kc)bobn0#~t;e}Al6<0dw-nW+&c$PEcjKIjMXzLt@nru0(&&-H% zk2ypHh0ZGye%C3&&x*3Lmg|4!X>`frn*#;8MnWP*paPkkN{L8~<|-kO)6%ImVw%n& z5;RtvBoWNz+Q1S2p%0t%e23%hCwssHgPjCRS!?peGZK6Dls42ze1OX87&ldB+4)uL z4d=Yp#Fhv2gj7@Ex^k+;$M}%_Pi6t6`QBkm173ELZBvCl(Ho zS~(^I^yrg5+IT=BD3d>ij?pzka#-q}KOviJ@HKva9vgHptejv3k;Z`7yFEY&%v#-v zI_ku1l^?G=Qey3-?^M5(`0?fw?#G8krG_*Wq7rhdQ36bjI1_}ZWu=Wmd(jA6Gl@(5iPqQwvn#Ql zQ!DpklWLU~>8{I-rz!SVf~i_HDHi#T7H4`(EwP$4THVBuGzIG^NvIXoy}@_&G-T~b(+&2 zQWbU@^#J=`hY3Dw0=s2`7^78OAcBLUQmH#cSW?LGh6*MK#7u4~Op(GTQCALGop5SQ zHppD~F3s~SAl+*dNF{?~QU{WMW-la|wGfpYzS(FAOx>dp@WVlevrgMlguVqmfD(O16X0GDcx-{@2pJT3tUMx zeJayxr5+he^)SoF0jfFv==?cNK;JZB$oh3=B_=MNXCoyU@R`(5mF>@7--A=?9d4bS z?BkQ8xXzEyPUe%7Im)9PUBmTnDoNHXQzvXqQ9*^LQ_GOkBntO8OxH2fgB{sw_AQ!Kah#uM|%wwa3c1 zm@7YQ9tRQmcyDKx1^nbn@@6wf=Tys1D+VJarYSS%BIOAS>3?tlGhU;&pJV+ z;koDIo>A8qP5o=NWAq-w8#Y8~Ud(%S7BH)scN@_8RGGe}lrWZr(`!(EhnJueDly&4 z%c554Dzhsr+8cOq)|XEJ@Y4NApdW0wFTx*ljr~7{ny`AgrC| z6S_W48e5j&Qcr^C@4LXUZEU8-jG6#JqJ{((M>S0yv#< zg8cPfsHu=N@&*7w2OyxEQZf0aS=c3FHuNVZgn6}DQmtBtM8}&YB>ho&>;bP~g@RnZ zmlaDWGCh=C?ma4SnN?o8B}F!wXGomK2zQcQM2wmpCY1jXcWpR z^zo}%!25O9G48skLvDgZ5_a3=F&%%)e z2Ux8ZHFK&v7g%oeS@W5sm5c#vBWEHcJX3p-o_ZA6!!+d8ArP}1zC=bO9E3QxWBKTB zl%LYkQpbHO6hwoTGD5_1Anl_#W<(g2313haN4mc6um;_qHV?wb5x-}btfKqh6aODv z_?3dF>BD3C|GPWkzx%t#`6IGlK4g0jH7fJug;-l%Qr8VXK75=X?d{`V`|{&&OcTq+ z$2;M_2fL5*M~@%l;$z`P_$DyOX}o76Zxb&Xen*N8LzcY1ES<3=B36??0+4&= z_Sh@40~N*7^o5OyjPwQ_Mrf5ce+nd2dz2(4^@$jQIt>jEr0R%HNkm1tV-*D9RE3x< zXD69X_weywq-bgZ0j5Hy*B5@SMt=euQ7+$mI5_%?NLeR(!^jQ?kK|(|#f#xVeeZ>K zP3FmJMq-uY1*o|$8V<^-q2EN3 zktS-GC9_bn$e~|qUcV<<`%J+SX+uefkHC-Gbtpq zKZ8P&QRiH=w97@n2aoo&2sk34_53#&;f{9m{l^Cq42|~67QQz&4;w^#=4}EGaspc= zGrWcFaE<4%_gRd70wH~-I7cSmb+r7}eUMzHUjAkp-@ssFA)i6^Lqg`uFKn=(#C^Yo z%{RLnnXjxNDK{o7*x$?d4yBKV&+4G3?Uc~igmKs6vl=NZwnBpHay!29zFF`aPPiuO z{G6X!aK+Z0`C!EQiK^eYRzx-$qgrUOT6LEwiav=qn0(*{O}Ppl>q^^hc(r&-praglqk++{^u zUTUK#sT0z=X_9ulRL*ITXCH50l&>wABD%3^z5d6lZl&DzG^(v}PdToOF7HHlXF=V~ zMYNqQ)ECd%1D@NqM*fjLVw$q-_1CC4w$`4oMT83QP8`YbkHf3^+zSVID|M+BJ2jpW zm#9LWxn!F$dyUv?%KjhYZ=yZ!R2w+p{}x40L}i?+pT&&Y{jhVCrl_hcx`U*e)PZ-R zI$;|ISSx$=-HME@$D?~J*4ldY`k!?ywjKruunkNOx1I*$RJipbWpTK8SPpmhj|PMB z?%w0!(a{DLhud!xaX8$5p>S7`lLGFlg99b6>ee*FzO(Q?rqsIYn9%W{kr2p73|0;i zU86j8P_Uc82ZOX&wpaW&_gFEOVIKMCay(>C@t~{<%JJaEtYuP}`tiOi+O7pu4Na(U zN8Rf7velTUVIxU{AbL|K@J)$;d|Rz(irxdT*Rbt|ifwLb7?a)FFQ!+BM%Yj&l}cY^ zMQ^W<3OpIzN~vvyQrjEO&?|m&F>TzP;`3>FU6F)p*CL{KicE}wp2nze)gR&m-rTjx z`X@`!)FaS=512gk!=8gG)qhwZGBdG^k(8(L*;$g%Z*Kh!GwGh}` z+O;b1Ln{+>S~l)0m!4IWG~b0VCA$lIuNQH5CG%(QxnAEpciEO#LWFF~Moby0qupoc zK_b|9-Fm(L-S%GpllR^w0!lNtz7oP-j!Ua`Y&dH+Y&3H9X*AXyhJ(5AgvAUtuofdb0 zGzb<;v4GlxB;YC6)EerHE?O3lGYcJY=2nYTzvq0RbPSlFT6U>(*_hESi8fR#M7WNq z3P(sZo0T@PN<%!n^Ng%XmH@4EY}(n(VYEeHsLr^N4VR>=S3-MKDr*s;B)>%nAzmQ@ ziFuvHY)Lf2`3!FXY{9awS_Dz_duKx$R<#-d=#-8S=a`XM@gk*YVat#M{^6@ndJw@o zz3m0P8qF*@o^!8TwWV~=HA=KHg4A%@6D+m_hwV5kNd=lSWQ+|<4$pj10gDG3<65fKHqr zF;uM=_z4lFZ2-2fn>L}Sq7%E+GptK5FJ)0xvo>AS)AOGuGQ)`uqdilA+cf4z6aRc=Qt~l3?_5(w<8!lwoUlHa+N4DCZ}bKRorl7eUx3iV=y$ z&LG|t2{sFS`Lm#%L>xA$Jj@fdrb1n6y{h*}E*N+VWd{LAi-W1OqOBs}2HA&%DGkR_ z30tkQX2odE4Fq(OU<|r6x5wleeZLcB{@KhIvx=|4Uel#3u51h00bRBd;vq%J^zYMD-c0Gpm`cb3$r$T(vi{Bnq+& zCl!!LI4^8J2IM7fH*bPkh7lY&_fT3s*Xbgj+xEUj} z%i^V=UA1X|j=(?@u~n6p_dHGUqjEMcm^EVtkcH`Z2~|xheW6+`TdlN@=?KEk6woY* zb%jy(ktr2HE4G};F6%7N!CM>Fdd_fc?jZ0cbfKX2;HIa-n3J^qh(-O+_pjP^(VXm% z4>uT9*V6J}XIz!oc+JjqxImllTBB&5pl;{taH_ymDQ+c3uH5K4Uc_v*5Y7mINS|S? z=9WNVNx-)Bj7+b7`;;$ zF*Qqt*SvYp@yyjSNRafKE>hAmmNfky)6JY(= ztStu?}dcWPqK!f2_!rDvg$An$mX0ceP4>m>y+4P6Xu4) zbcVo-8P~_l8wd(ki`;qMs*x4JYr`&0enLU(%S%+k#$d_7U_i)QE@mcMUh0iR;zm&O6tk)BMFwix{Zl0N~`z6)4Po5Qm;O3JUPyWQcl{fB%-=>8e#&DN-`QyZG$PH~YN zEIVvs&urYHB)e_bhGrj#@jEhxHWVls16>yiUt!R+xrJ&-(fpritUEotEUR6|IMc zp^u6`j$rb3QH`rgL2)rw{$2^88;KG(QU+`x`N3B z=CqC2Bj@85b;54cULVB}d}c7SlwM^NmmJdDbhKE3bOqBWyF&wX=A7Hy0sVEn<(Rvke{7$tw8Vmg$*yOqT+uUU6Hy z^yyj=6qoyWmvqBjgm@7|@LI2Q%g%Yd{`y(jb{?=4+|Tu!Iaf&Kn^WB$$vCwlHs!-J zRFHEtlZ0`HH`!^4tE}ieXFk$IxPzTTh_(@9J4GvYf-{UDS>t;(iZCtRYXaXz5#?=} zje`F7$#>5^3D&w?in*?@H*`R?4mX9nqtTjhNT<|rH-3(`g#*b$-5xnX+@uyeio;v# zvTH_Xz5ck7SV{+9upwrmJVFhJ_7JIn0%z0HA}BEtSH?ar6Fp;WoQd)9OcI9ZEJ%k- zw-YwOCmU$Hv|cz$Gz5U&Q`t8ws0G)73l2A0ka7dvRyVNnWKLUm*lvxig!Fp|1$AC! zL%ikgah5)^!56DbC$eZTvEQT5NgxjJ$Z#HF9Hvy=@PR*|M=wbIC`su{PfFfj$_*dw zHKlBnlHG|iwh>WX6cQ4nI}Hd$1s0$zj7ufD8DXi9z6DB85>xe$kF)mf$kI8~h*2We z!@kFegzb>Ybu1$%qTg{rVMZkc5rRd=FB3Ky4t;L(9=MT=6w~#96B(q;XKWHQD^4%J zkOmXMPm*1!m0)_5l;phUQQWC%HT4=|7o^B z(;aoIAfqcOncyBN!mJGq0>Lnuq+Efuv@Ft=9@4 zBwUzsW5FHCZpM#gmf%Gxyd7$cPT2d?z+%NTn`$4~E3+Li);7Ss{!dM({YFQWYLr{R z$MWKF5y#UY!g8?o2jWN-yG1O(ZYy-@s$k7bCkHm+OLht;V^xgcsZ6>mm6qHaK{6zP zeK-cAcvXxLzQA}=AkixX-gF6V9Rkya;GCTT0^TxLar z09Np4phscT(Q9IuWL6h^Crn$+fv8xdi<=I9s|l?N=t5}tQn)Emr=kd!e`}J{Nc#lu z_Rjmz$vNf;DIiJ4Y*J`mMUok?=~UHlNJE)iLV-2A1FM+_-I$n0)%kkxCRQ;~Pkp>Lz$$W8>cs@Li3CZ;z*btH(%msS%;>-!XtT^YJE5Ex2>Zd%uq&t{p zjJ)XFGhWPG5S9D|$lEEmgP1Yj4O4zFA)4bkgjoE#TuM6LS`o|w*Jo@>;ssxdj?@Wo zlgbge-I>gyCbIFJvW}vH8Ma)49L_{JDI#bgmE}`-TZ+qy=mD^V8u^nFe3pK*$aJCe zM8BDK+-~+cK^7zpF+^Hi61FEouuj5OV!Xl^9xvyKlm|xXAo4`f-3cto+#Xk zS@6nE zYw%1+Y93Xm<(sQjfn$h(GdPB}5v5TwrJ!aE>7;Uiij~Ayni=5(Q@G%a$bi{K^1xlS zyqwRNt3L|-YditSbKy{hZ!{|xB3^{d!(|tsBMQ78DZ5bUm%wC9hCLL6A|seSur-aCmz%9dJs?ii$g*ISU_bcu(!I>fWhL#*g`kua zE=vO>p+I%5jZ|l|*l}!ctb6*NG|=8V+SL}?-^Lx@)22Qi@(GCueKr%L`MI)97p!hz zd`2UUT^PIIb`C(E4p()(zS~82y+QQjO?8A>QHbp46aL|!4LF<;WKn{4LEAZ;FAnc? zOVF6rKJYGvZhsqU@SNB;w9+Vi2XRRdkl(pcr-?h|NEnS%+7^W$Eluons9>ckWCof9 zP#N*-#~2c+;@U#G&%D0%MI~$s1CK;~_v!_k?#k*N2Z56y*l3Qy(y8yx)@!cDw&F}j zoYTQypD+!GTUj=a)u%P>vp;X?-gnyJvS?C=$LDTblV+#!UGW1o1j?OfP)rIGDbG>h z3We6!n#)g|JfMqZ_V{ppsQ@8ispLbu|2W@6&GhkMzRy&cf6iicP*i{>iOI|+J^lHW z#qfeb&H~B3I;~Gwrc@X2>Y1`JSUhMTw|}zUvIF$4;#O$;)9IA) z%F00-O$Q>XX{`Kr1a)c5WW%it7H6&1HgPTnQ(p^}j4Z&r0~GP)kqWe}T?SFo>E@%8 z_Ycp_$hbMGhmr=G!CLw%2J5cAO)FTaGl6_HPM)h3W{jK7?CUf@nY(GE!33p?}FiFb5;~Eb$xWoi?_2Y!V?vG}=&b z76>uYID-!OhxvLl3+3MT!cwl!40wGGWg%X7%2gMkl@86uIPJm_bp({id1uB*jJ&Lp zn|S{YiSud3L&DZqx>h7W3n|F>>GjveBj!v(FKgo-%f3fK4J2AY(~w;NUy;KWEP zZDdL}b%&v8kCJ2uK&IPM;KB!YT0K&4o;3vBn@DhduhP|jftj?HNE+^z>@{Qg6voJ0 z>(8HZk|+MGS9s1yoJ;Mv7r`Fk%91_;Ixl=mBfO-EKT9^6QtIQRgAJ=nEkN^NVDWWi$&G8!oI#>!B8{TDHH#b@6J z?vch=^U$%L{JQCdrAQoh(%jH#U(RO^GJL`x=Y_3~g1U2#xml%$uvAeNYt2mE7lK^wAbB692M|&$^9QflF=9qM$(6 z&WNz5Y~x0=>_#?j7h6qSxRvNzb$lfUEl~<7o!gOgcZ8N+w>El3r}mzsqU58WpceB_ z9lAvOtoKeCL#I5AHgHcTbh-0B0;wr+if}@Q%rJTHxL06Oj2|&F*Z?R%*T2hiJ!bOu z+yVTF{K@RpB}e_&%AR<4KRV9$p*#-u^Swvd-diWtT-mtyBe|6kH%U+(mN%}^71UA4 zi4(8Qn&cLGX!&xUktR4v?~?|N8E7~KW-uCc8VZDzrZuVL>TF$>aHV&{>J@1_EYju@ z<{EYCe>P!3Y+of3BM^2OUJZP+ zswf&{-&N(cq#ZrrDMksWS}&XyM=}uxm=W_t*N;sb?gm)eQfjrQu3Kz4h6Y)rA2f8| zqTE5_{95z|TL?J<>1!Le9Mpi_IDmJ+H_F~0t@OAE?(teigmuT}Ef#Gp=zSGlYmLl8 z!l@H}dgKpNSzj1p;TA>bJ+EQg+j_d?y1r(;{vTL0Q!&|dn|@}J`6JU$5)bVo z>mlG8EvTNFzExv?alh;6!ll5O{C` zlFWc1s*xS`+-)s4V-y0{+nh+Ra@i#xY5OO={NlSc8+-m`-IjmJSwibS5;4{WkO&^oF zarl^n5{Yc&_)WJlGf50BV-t@@DMi@IMKFw6@f8x#i%^JF=?LSC_vaOJ1!PD!Tj!$2-TTKG{jRY;q0sh!hjXCASJx2R`v==Qxjw zdN#ZWUv{u_5FLPmL9;z6gB#xj!g4S5joSf4<#1_B@BsN08r-BGm7)|H0jz}ndPOM7 zl@>gqjL45hTxzYAyqVNhjSQqOcq4R(CXFF+7mEjIp`R8~aG7o%Q^M!IyK?xXocz}E zTw1B5Bv%SG4AUYDd>cXisfZtuOQB0@_Bt+3jQZy}$M{F8J8>%wI;iKMjJdL9bbP!4 zT2dVq*DGzSKj{@B-5srMg(eZj&ZpEOlX|ma{+i!5aTXm&5l13su_kO^gSH zEITXNdZX;Xr^lem=d8{vF;jhu8-&`UnUROQBXBR#g?Z>3t%fbira)ffB6yzE4}TPh zj^K;7;}!~UUH=|0nsh zQmmqAXHQEgYU#%4-rj7e9M(7bgi>HyT(Ji>H?avidSiz(uEx{$lG6CcW8B_aC+5{> z2ui1kqyg!XMPZz;#Wi|;Z-$^8*UH6VV(HDD>4dqN6e9B8I?8(ID)(E0J`)}~XXA_M zy;53IXXBO`i0{s21EZiw0wXG`sK`Dsx(g}^bd)|hzKtRW`QF$ zs|xyzEqpMicA14mBIg1IWu8VolRvD(4e*WXMe6|Wu!gzz`}}CPlB9c@+cN z$E6qE`$Go5za^hb>Rys}QU27jo}TpN{2ccziDY2h%ic@3>4nJftfx|8sY+YZp7cJ< z2$TXT)GXIvxQePc@*SrS^<@rNBgVK7PQefJRF3w)wCw;L-wL< zuefDi{3@hQv46gqci$oEQ5;vN22;r5-+@0Z3v_bBz=6J2r5u>VjJaM|!deoO*vxT^ zCaR$h(Mt#r$8;SHq9&V9EhrSC8@9vXo~P-9CW-s=Ulz${P?8CY$XSmIKau-g&A6?7 zqr87gb{Sar0&R{GiHVN|qQ6?JcD6;q^!3m#LQK3g7T$ zXy&<@A(Hdg8(7K$tXl}mv|`PAPsT8ji2WH3g3y4ok5D&SU8#bl+u;CsdvEU~hzS%r zh~37wXpGv%-~bGL5Q&tat)wM%hnl#{zZjU|9TL74@D`L;a1o8wUmYbwlCCxWg*h*@ zv}=#;k%kE2U=9Z7Dkx!VN@Y0{%mgdbV6Ql#(~F9WK=WRa(FE0ao~*J(NVsax$XMb3 z71>jfSlV1Dq@=tifWHYfs~x*O0PX3SwAB`r}Fd5Q{U! z*KojiOs}P*<+4U=cq^<>9TFdWk5q~T*cea4xGbc$Rw`ujMgq0dMSBx!_b`t7@~SoQ zs#O0S@afc$hzj$|$x5raUV@=^o|lm|iO$9-B8N!j%d}j+12sKH@QOA;(V&ou46vmX z)+bPnXmjVD#o!jutBIkDtT8l(5Nkn1?dWYvF3Pr2z%^Kp$_dum7(5J69-Eh+Nm(=L ziE)yO%Ck(!4xIM+rFK>?pb(7;Qb6;T(-W!2%bu-e*Qp*RhB4=>`IV)cz3hXm_aUc0mjPwtXiq$zZn;Il*XH-8<=hIlPw zi}TINvjBkqVQxaif98einU@9m>#TPRM?=%V{EKJtHt~?a9SLUggkT1s_fgGtA9AWa zO%AMaGbO(2F=j(#g5X@K$kG;7LhwN?QN zr^7ZWZ)g~(|7LGqFkX>%qBSC;W^vWIog4%X6on#^cp}fV$yy0}V0O7OEv4!Zj7i^p zpT~6ZBx=MrL=o?7K2;%PoR}IHDd~o*?A6J=FXHVc8$W* zi2b6NP!v5MWZz*M<6ckZGo_EB%b)N$78k{fgCnq`Nd^}OSu723JA|y!H=*H%muOJw zxR;7$TF92`(M_6dYGA$L)mcYOb^BbZ>`0HZ7^`MuM2&G%EgZV1V(K-|o~5}$Hfft= z{?Tsyy6W~HU2mhCd4xP}sH-gU;f!03W0Ng?DyB2<9@>c0azX$5K42={{GXEk;Sy9~ z+GE}iTr{cq7{dVe^u3w)+_l&3XS|bj{Mu;tBKIM!%HDHGxmS=YG#PRE4onX4dJHvC z&>*=m;hkeO46ZfR_+v_?V0eN?RxQsa@yBGwIhv}4Y4vhb$tr`{osF|aZH0OqoJR~L zy~0Qk!J1-HFMwd{nYY%#<2iwc?yyR5QFR>&PPh4zXelCz0@+7FuplG1uAiAt>*cO< zd^2~L_;|Ss2w7Dl`Wr@Wp%WhoNe*5!^y_6-PMEKp@ft8!lac#AuJ$Hn?}{ z3Q!!6yb8TTo+bRNYtNS+wSj|3U~EdRU>>s|grn$R5G0acJ~c@Oiwy}fjVn9fpa!Tb ztG;`=hcFG?tc_K6>(yCNx(V(XqW)5&g$X*_Tx4Y-AcRLI0P-_1o|9_p+mTnkjZyjR7v|jp%8$cHE;&vKFj%*mAE%JLdYa^_$c)CMZKY6{(G? zyu40BB#prRpvb{@0()hSfHP%lvLB+YN)JRA?J!l+j?g3Rbh3y-L$XdM;3Y2H1nfm9 zr%MTV1{x4a;;9zMTF=Vm)>lZ1+d>6_Z0@|vXO(6RLx&^_eR1+|b)lgO2^j&Z(l8?1 z-v$WU3zE~IJ_s!^v?qgS)NpP)=ScRhcC`^6D4Xg7W)73KLkgSVXbPvmccNw}(vr!T zC4?}{q7!1o2+;%YQJC6tB3^e|X<0U5t$7wAr%_{IBPlfrAya3|&7?^N_bsP_aQB8v zX+JIHj-j`xmW8UBo2WP5#@1RML5dmV#Hoskal9Q-=)|(P zYv`QTOwk)nW*1fqQ;0Dx<%FX+FfloBq#SkK_FC)BaynKB*Q;@f(Nfr{XVn;i=!W}n zjBkW#X1>r)u^!?8eHb^{ow3DNPH+ z(qvheIiwt*In(hTKxcbJYeA*6my7mPuA04dDh5GcvTvUDq<1L-Bae`fq#zJY%Qizur zmd*rx!my`F8X^7UzLAYo`)c2wh~To5n^-a#4cC!JiHfv#KjqBZ2_@}+*D9|}%tocz z%Tmb7sliwqYyYWaKHjj8yM*!aKwxH|jAwFgwe9o{=Yv!3(GU){@b6CRY=3#c(f2d8 z2jX~dJq_(lZ($&X)Pl=p^T+&KuehXbYm`i%#!+BxQI_`5Yu0JSS`7%_SWj7_220hu zW}rj|IfR3En1IQlOWV2y)g?(((n-rK%n3J!H;AVu8<@Z+fLA8-%X+B8V?Y?oS2cSz zAGX2rRfaFi!fIPnCFzOjjuMhIk=0yoFDw{Qt*|N~Hf+>%C3J7^8C-V~+ zlmv?-kd$*c3)hr#l!lhUkcx2=81P{ku)bqyopR*GZ4%rtXmF#O&`$!@=~5s8ERj>Z zl^zbo!qMHt_Nk;tb6n_)=&s?L3xVu~h`T5Mge4t@zodd9ba&ouDQ|Rhc&t3GRLU(c zT+jeD=4akmc825hOq|GVI}H@CsT)3kNKL>2pi2R_H8@AH+)-lBLU0BiV=Zs9Nl|+d{ka)79eD zOZy{_MDi)IaTIL#I&08VZleuyV7I@F2Ac~<12{r$Eiu{I-T*1%G=j%AfjXC7JaxV@ z5XNDN%Z9X1zy;OvB!0@>$RT&W()A#1KqJ{A3RHMlHPFqcHIYW>#QDf{(k%PtzYqe7S zDBvNMsNyz#z{*30mzzYliQC#&OzGKh|Hzsxf&IfQNC9b=Bdr*sZe5cbS8tdNboMGa z(X<*tWxs?+KZU~*pyn5(%-%(S=7wXSGPha{k;a%e<_8ZndhXpzon=N^f)8Rhx?JLOh4X@z~Rse@tJm-b}6J-2tHXmtnk4X$4Aq8ew5E7v%;z(1b$*s$FmSo)bf}m z?^|2$JK=H{k-*No&>{)w3%o3Cl@hJ8$D~EN60x|tSpeGFAim*&tgC7=Dqv}bh!!jo z2rp4|sx#k1U$aG`6-Ga5+)GxK+oSGOMQq;BXdfc4>s95YP9$e7o|%GrRE zLM_u9{K_&}LiD~|;|3zLW>_485G$8k z-*Q;5kp(vkD=KO5Sk-KlSXw?ST=%V}jI{F(h!Sgt;B@s?Fh%2gRT+(z{;X*|(Ms;Z z6jNX|Ko)ay^D@tbEKXqNbl`O@|4jh)Bz_ni6in_k5ZS2r9P%qu?bQf&8p~wPO|E$G z-n&9rs+AO~#Qn;pxG_ly5^`;)IPyu96qMpp(Z^n}I0rIm!T1QFAnlz4BZKytfz3(o zI6mNVIe!mpsWFWkLLTF&%~71QS5SWkS!1|@0IBIGp{EqMatEXeVAT=fAN#$xpjCum zdkw1+iti$J)+}(z=9?H)83Pf;A|sm+aoZ{XJKFfJZ?*go>(6V1-Wom zyV+r^OOq#X5+sA@?2}_zq{xySce3W54ong5)PU_NbrVXT3Z8)XC8&I28YYy8^DAzJ z=paQz{^%;W`!QkDI)sPU@uMJ|$UY0O8yv*4Ub3`ObPw(pDo6&|H_C1uwd)zxZ9;YD zBn%ez3H`@W@Nf_qw7&RswJSVE)%>z*QgVP;FhJ9 z3zDi@-Dy#MX352K8iSuPz##JeV^=>y-*}iVzNOz*G-+eFrWIeS;Nc0=5Y ztQZ~$T1-h_0%-@Q6hfw>Q$U0iFT4Kh1W~NrDF^kMY8mM4kl*qQP9?dYW~Cj7+YY>l z+gygyVh(w6--7G~&|KmJd5oPlJMWuZ0gmwF){sq-Dbg5C)>#2W94qO4m<%Mr0c-&` zai*pyE)gf(p}A^y7)>`N10eacC$&YxYE+I6%faB{@xjA`-MwsYclY@4u!n-C^!L4f zzu&zFvqtB8#~_W)_YOfG*~`AnB4%_`N%)1*8obbt)Zjz2)^I(RLGm_UF!Rma5ETi# zG*dRsM}awmOA3311}qV2RW5*#&u(0T6*KLWq=dK;e(MB7`uFS`u$rb6z-F{a^_{Ww z~0?bfJ!O#ZhQemVg-`0c=8^ z6lEfYt$wMZ+bRyhU&J`-H|N*8gZ;s7p|V)8+H}0nf4;xvn+Ji=-f4DmDL{t;Ki9o7 z@CWQ|eZbyr4@h$1T<)QEv)(yy((>)%KOabsOog`4CU56M>*HX-0D<#r6;@-T<%04; z(UUsP?q_;;A&MxG#HLp{@CYOsu3R%?1)0IKq?0W5kQ4Pgtu+oNFmB<1p$rB)Pk#L2 z^|Sx`wG1Qq`vKHCa+hMuvc>2EPH#|-mOHzW4-ttDl}(%cB>XYI!~|iZxBq6IlTK`{{3s9I86xyX?ievG8PgOt8Qrxi&qS5D+s4m#9^D9C zhvzIC6pAthfO;8llbSlTTjWi&UR+>0Mc!^W9mNtlmyJ88&DB*)t#sGJuS2^QOOMk9 zxVO%7+-{EvFm;6TWGVD>CuC-bcx!Rz>}b3g7u6;5z-{PRqPWNFy>ddwW=}e> zIrEcClI80^A|*tQyb0vYi78e%Lp(A}_J|v4Fi7?ap}zCW`9SCOoQ&l8qL@Iv`U^Uf zjP18(j$#D`fTQW;E6%vAA+AfOgAbD$5tTlRfZg2M7nEH~nh~{rbr^|$$>0PxPKcw? z!EWuP0++Pk;&tV!WeQ5JSVldcVRj$#_o(CYZpmhxkg}M+i^flWALF%k1*MhjYZy;U zXCDpipqx{uM%t~@_ZFcBXEL*5-(JyaaBLiq(lR;*Jp^v-2%M+jQ*8s_lFex5#pSIf zIiU0mBTHs2Vup2hIFt#E#Rz@a7RoOWan6f$g=!~rWW~V_l%U97Ptk222J=FB>A2&8 zCCeOgayaziq5XyrwA+rnWz><0bXa>X93$F9J|+xj6M!l0p&*2Hq=y7ZeIbFAK=7;K zF7khi$WpUh|Leq(hchGJ_?a`SEQAMWQYGXT!U-w{7{I3SA3tOqW<-LW550)=Lfr!k z62a7*NdsQ*zz}M3BZ#8V8*ta5Us~?*ySYvmp9g$KuLz7sxj73sk8BZJW-Si|aqbDH z2x#cOC=5I%o4G1{NSOF&=j=<{>}2OytT^)LPE>6-FpGnmpxtx}ZZx74-#bFErMmQa zpk_D`-CBxUS2+2)pGAH}fp|=%Ts2pl4;9PyFR61Kamk(n>aCy}N9|ombDiRG27M%F zZLM>V7Wle4++j~9BJrl*i)m>#WN5jF>N$|lFcG}WEaJ*drLey$bSq&iif_h!W3T3u z${`aRqm(YvQS(UUOx9c~*RM(ESZgz~v|{n9bym}Y5BZM6-~uZ-yYSJev*4%g8~$-O z9I!+K!j2oU%{&d- z^h^spHGf-3b!r^$)lg=hy?S-F^Y_g?t7uj#OV`}*|@`i{gU zX@>JWG>>J)MYm(-rO;1VX7L}i#3;&hqD6if30<Dz_B_1T-(BlI{{jsG$PRh~oIfNVLmLpK5VMXC9&QppQrM%kg|3=|~s5d;>#j zsV<0dXBxnjUcv&!FNPD2FDVBjFEb)4;-R?QqEW_#1Ez7@6@buCO9H_kmap2a!!uSM zjBUAk1C$Se$S;ozCPqjl^>g-9%gsPVN;?l()o(oR4K=UBL{a*Omj=R`@Hhjh4yBh{J^tPaycixJmt&KC1n zGLF|Oh(jW=V73adT-LfpeKT5#0b5XVu1LUhnndemEo%a#zPR@}ydj!;*2LJ4fkiaL z=C~hup~4_)p~u2~Qb}K2G1K>@W>sYA1F=23G{f^irJ2 z96e9f#@Sl3Kl zMq;L%%r6Jgx=WpTlO`aMV*y2XZB-OXT;LWPmol6tJxFL*!R0NPMn;um72I%`)T>!* z8;_gA6&B9dkC{bS?-k{d(4v7p;V~f8>Q*6FB*Xm*q4y%(m$u2?@OEP1jXObiw zdpFMMw2?`LmKI%jyto@?W~{}=znNQr+VT*0jE!4Dt%qK%D)~HKe*5BdE8YQA4pq9~ zbDmn|kNMENQXAS|Mu$5X-5zUKFP^z8dG7>-7Y$OD(%Ax$PH??Ht>IcS{zFjRb)eX# z_f59!>k#)%YH6Jn&=Ve?6QviEE9Np#!QLxEWSRzf*O2##_0jlfa!v{}#m>{v$1D0Z`@c! zSSACLtrGlLDv(P?XgyX_cA^y}-s``K0CKn>&{Y0>ze6@#1fm_~on@-tn4dPgGiz;> zLMWM-yC2O}p449lp=B9ZF5Fp>sRW_`M)6k*gE~DSicv*SKGy=bM*j&r&Ihi+7kH-W!g5y*0TRZoolNKX~z7R|uTS|9KrvN6~?2#*3SrNmH7YQR)aGETHnmA3U!!K$0OamSO{~acEpIwJ_ zTKO}tvt!br{1wtOB_m7K6c;T@IoI>n`~P_?Q7eGiyik@QE@f_nI2e^WK=fMQE#v;? zkTGw%CA=QgzRaYI%_8fWDs8SXP8)ycQeLjOs<^`3&uXGaILgN#n5WQk7{Fw+=b%wSflbDj` z2jskwshM@SY^`F_Prk+y6u%|tk1b)Pd_B9OW!&x&Nr0OHlFly*uqk(-M}(6-zYV?U zH}s*8+tP)9Y~TV;dt#@s&9xHqb!y#h{VlBUfHS;U;CPp9X+;_ld?+WD zU?LUjXq|0PhKtSND5A8=HaHq8h!Df@lP%t+!rh}lhylhoQ&=!gu|o}F1voKDhuh)I zejIVij}k~yub;ek3{2XP%{1em5;d5Pk6E7hP)9EX=E97B!q^ejk9Kr#(76$oK8n9MoGk# zVHQbOyaeWz)K0hnN?O}qM5Xy&9OaE;_ZH@>ze);N^bp$mRrOdGnNvWX0}l4=T~43E?) ziNUE=)kS7k9o0ID(r<8hG%B#8xM{{n-&PXkMGJu-Kzs(*T#y@GF=L=Ew#meOh&a2X&#%o6-%v@gl29YKv^2Tc zVbjurX;*z*z4|!}SyArZ4nx+_IGON3Xn|^nL2I25LfLCfJ8!0*3N`Bi+^(V{C4iwf zI~)W;Dn;z2rLgVCDW#N-5#w?04BGWV6_pTKv1s1PWYy~*NjDu%3D?HPD>+Lm7Yr@{ z`xBhS6{;ryctbssOpj5fk5@|xy|VY_1v<~$iZt(jal}!kg>XN)e<0TWPv~GMSV5&o z4KY;L4_u`RS?7#Ei-1rkfuU{6q<0jD-Bg=>@73Q@ zv_>e?1V$Oyue68RXbrrGR&<6#((8x5l1r`2(?YC3li~E=NkfARXgW<+Bi(i`O7+~h zx8K5gcw3Lo6~Fn?@+jKmwNZe$r4EC^p#AL8(H3dp_rj_3&k&;2H}Gmlh+nh}(UTy6 zP{AuelW>Z`x>ovOXndih&S-`TX$>MYn}Z0vOlqFQAOhZrVs97XE{Vk;{WihULJI1d zH1>{&mC~MxV)J+P;x?8R>)RkjHY(`41*9fPwf2hpTnZjG>WIhQ6?6d|zIWAH)2%o_ zmQ(_EDd)6U!gw6E(l~UYq=-vc(#td>MhQEt@2G}7OH0^= zr3K0wRgN2339>qX3L9#NK zq;7N?ev<)dLa=0<4n|cXsYR~&1-$zvbjZQfSsCtZ)M@mb5FqYYWQ!}sS)RNvmmDJ- z$!6^JuY^h=?pCI(3XsI8VsQ?ZSXXUgiKHDBaby|~MW7P6n$tYDNvT>wp~$DlBWGZN z_&5in#4tZbzR1p_R6J+f3!asCX>+-vg)i-Vn8;AAdxZ_-Sv4t9M9arbb%VUom_6c# zoq+hLg(Ql%s#2y+v_eFV!1z)503g@WwxCgc7I#xDNm!Ate4j-}C9+nb>h%K4oUlwh z^NgFntyi6Jyz$Kscg~BPgdWT~QwjYGEJ6v8q^4HhX}YE<19o@=qZ0?O-aPrNl*>ud z6HIDe1A4X7v`c$5TP*X<_3q#?PFVG8#Tm!P8_Ra^%QwD~or0Vb4Nd)8MZp3s+cld% z)Z*wv_eHtiQj4TU{~HDzhl@AX6`gFsAKUZxEDN|X|CxnmbIKFWQIg;sMg72K^pD~u zuVHMD=*HoTczA0>UAdNQy^jNPga$g(dM^YoFm5*p;vLo%cGj)W(t(Tpiul%dU)^1`p8-Rj4P zX$BgVp4Rcbkp!LtiM5+g$G#Mdf&7G(`V4p=W&|4K8szm`1W?Odr1tz>)CV5)D07b0k%c2_fzq zk0<_oUw&KoCs88W{UQr}DT{?NHiXE#bm$P4h8^K-$RnBqY^^R0!}iV~`8wx8!n0s$ z;tmW;nHB<8E8w!6Wue;~rEWLXcdyAQtI}TkLTkRO$9o4y$H#*~_4sgD?Qg32UjJ!S z{k{IB)PWyy75JVOfxny!<-J_p%alOOJMCEdKTar&AQ&RaD&_F0#EwKPxPZH=Htm6_ zMM=^UM%ktR2;@Pe<7XlsEVgV;XmFD}kh7sYw9@)Y$A>!7gdxp+n(I(gAHYRfFRq)W zY5fD0^(X1B+n9OXfrc8;qb_d$-gpNd?H=xr_XdN*^7!a*I7;{6?RSu^c>6WHBaiZ9 z+LA{(cnLHSkln}3d~t8P%nCPl5qs~hzhGDRUax<*tCN8X9~p`LJ3lPeCwKml=ZEPR z4|tZ2b}>swdooMRnEV8xEpd3V-{{U^)8YUAYQlv`FiT|0~D6rmGgsIZrbENA5gD%}MxY z9t6cn_-F121^U$gnZh0Do8-^DSA6k>1~*PZltUZUEcNP}aBE&58cI8#*29;Q_|7w5 zh7BW2!izOU(KGm6WTAu%t?8}pTh<)EX zm!<0q1w3x>(uVp<3s&t;{VzDHXa9Er)Or6452Hq((Jy90j%bI^#^q$fS@AFs`jl6h zMvgFi#uoh-2>ULQkA8>t-r1RfvZOshBdwPH%)Zw45wH$3pq^O^b&fD59D%517o`At)W1vuQ@&G!zn{^8zke(>1k+WjE}{*ojrPojE-_lNB- z7PcNdF(?5vs~t5O@ICl$@YcvphDilcYU^8W+#8+*{zm?SNs`XHn*V+ z{wD_V08%FWcA0Li&dr7c`IEQt6HnU6ZIkYD^4PU!VYKx)y6M0C4R66q8{W&mj^0j4 zkU=_|XZ~*;mj5j$>w9^b`*BBQTf^MwVh5b5e8Bgn5hb_uK^5_2M_2Hi(XTqQHSiNe zcerHE<#a`(Dr~_Q-p=UiLjwF_(Phw&B^O z-_2c9zWLeyAjKhH|1MkO_iZoBqUK-!HT#$D7cOu9|9JlZ2ir6Q<7qp1wIE5+j_=>Q zy17`^qo`^{MO*e4b|5s-NPqCZxRyTY52AT0OWmU*H*Rom?bHlnnYsRLEIuFnxogq= zJc#-Ky0tJVHDJN^R+Cqsz8Z4Qn1qq}-^1Dcr6YPiZT{@_O|x(hnDzHh zzs|Gd@87;g7w?noUWCbcn~ah#qYJF}`rC_D`&QdS)7PRG==-%92!z|xV50hESXG$I zWch<6gzQVinQf1M*|Uc$@)Ei9t;NRXOfEvNAuCc{PE#`k$GXBm!77` zgztBrwpAm4@?%|XyW#OtwYWC+VUz1YcHh31onW~1mtE6=_tgVlU}aXJ{krMW-%Ny` zfMun^Hm8Li4uUKFw`i|49bNmk4JWBi8aDN#5;m1RBbh&{+An0 zx=g|J3a?II<3+}z1N3Hw&+`djB@iE?2lDb&=YY6nl*~Mwy!A(__VI^C6WOr??bH9L zM6&-Qxc8gR_28u!8a{u@lJynRJ)#IU{(!&dE4f%nuJq^fXY}liLPR9{3$=Vtwmje) z#Hs!+?2h=s@6x)h@kVYNRK_2l;ihBg{LM^>!4q$aV;2P#9Ikq8pM}d3H!EEEJo-3X zoH$_N;#bkf@{$miKD00cI8VEGKTWQgXY(xTs~VKv=N+(*`6(YiX2DV5b)o-SGYGn4 zcO_XD!8(W(*l)k{a)R%lb?Lk6{(}{v3Sre&R^X3n+{%(Il;352g-kyXwi@FVgDhnXM4h`L-Nf2rGt@6U+9Nl zFXz)II_mp4VoVVBRoSc!QCo}i+r3KeA%n#BJjv1@<=J=jEcq0F=h>6_WHp_|pUWpK z8-BOhhzFj`hjm}{m*wq2v%nM6*?IigSu1<5J3M=YFM!|SQy0K~=wtd+&-ONx%Z9hz z&{~@ky83Ny9WnXXRI@i>Oaq*PQ~Q3)JvRM@B#9lK-+qtIz1W1o5KBIejWBx`9K6(? zg!tNM^kHxpw4U#5WWNV7xPFUIkbp7fz8-0UOL|A!S_6~&NpjG=@lO|o(N8x^LVGE; z`~SvPm82#%ic#2>9KsXZ^5aIzM(dFb!#6(%o8oKkpc>hL;#HfNUB7fU5~T`+bU7Mt zc}utcru#$w9&lukR`s4zmfg!fe8~PqgTjMv%jOk|Hh;OVm&>ufH@KYNd$4U=2RT-4 zmCw=bY{OPtqyl|lMugmdtomC|!nhMI;CcBCcRtCIQ9rFOtETl2<KR#oMn$d<|eEr?oH|ux$N2nu>tVS~#yH~$1FJT~ZJbRN{CvteR zd$q-R`ncD{C9ReNG)Tw>ADb97{HN&nUxw9k!QcNB)1ml@ z%s5i8(W-tDAvy<773ZL73wEvJ9o;dg)VF>d$)zf zL-~jm;yB?BHNDIy|@-3@-K`4fif~@0=s^ z`+PJ}=2s#@vH$oHA{6+agJX|SyrwI@j$#upc@dkl=|z3Hn#Zd2E3OG~ilqB(_E+2< z2+G&Kbu@ya+d>5%$y_Mgwdl$NT(IByS~Dc7j3KVC$|ec6ZHvg=OKOv8b?>caC=~@z z?4FG9i+1oe$eu_tB>|bzURTvB$=|_2BaJ3#vY)c2yi;nS*E!O9tq z@nxmW1jyT#_Ru%p>IGZ=KWF^H>7DM=*nih=ewy)SsYd$YJ&fAk21P-zv#a-*u~*?! zdy5CG?O3qWb^kgh+vQav;C~YlbCShU@c(P?+L{};k@fTJuV7S_bEJ?%Te2$SeH<8W*o2o`$1oDB|wpKHp%8p9x@gPGy$*ybT_*DgAf00 zK7)W|pZv?tyeHWBy6K z+SeM@Fp0jHPgJq^TR!O@qtzTWV{k$;<_G;k*80#~(-j{~1I&|k11OD8;KsxfWmg`V zpNoNz?-sa1biVrH^cC<(x|60U{Lu;6Xm~AkMY)u0AXQ!I$cN})*6pD|fL4p$9os<#^a2QT=6C>qlB z9&^VdN+N{B{@mweC5Uc8k(mZmp7f_r|6+HjZC9Q@gBquk2}G80E3vYn5K10+gVtX4 zf~j6@kXh5bW$#FvY#|M#aUKB*6cyoQt_E2%CgABcT#faJ&S%q|n+I9D%#CqJ|+{7v2Bmg66a-6d8-!+=81_a;3qE za0)q`r1)-fv+LMfY*L53p-IbX+fG5w+@5`iolm^@G=ji5+13;~gT z6J`zzijV+rM+lxPZpA3+_3t6(hShZp#iYELgh0ob<0a>I#sI5JNF>YLO)@p-wp!Bs zRzl1%Y0YYNhe;7sr1||0DdGGzAa+@O89+P*-+N9VhL~HDIWrlU4#(TQi3UQAU&sUw z!~*m}9i9G^RT*4o5`wR?kT4a5L}13urA_vZ#0ELQO98NB459k_d-c!1! zKKhmR_t9A!|1ov?@5Pz=Lm#Ca*wW5cryMbMI#=2rtP_tK4>BCQT+5EeM#x?i$}AeO zFCN=ymp3JFsIt+XU(oLI&@uNl)ct%63=bR@hN3VZb{0)8qRi4SS$^)~M=;JEIli8BWzj;iajZ;qgXq(rRJ8V|KXnKz-b<0)Y0yrzdju>+ z|5z-guu(|BI-|h$=`Uho>iBWNU;ms>#f-3igkl>`1G5?#iooI{$>_%5xpUATxV9P= z?#wWCAaE@yMQLwAMM(4)Pw?pTrsqVE{e&-mDxjPV@z0NMhMjHoF-j19kgRO+hZ)!3Gl8A+yPAh<$4%O)~x7$8!B&QxtC zAoa_16Od3rUbu3st&^ingP))*>g$g@dafWF`?*n-n3Oz*~U1TBc?_{xw*QfNqnd3wwA~YU(3ka z-O>)E#A(VMJ1%P^E(-(|x!A)FJDMAI6f^7)Gtk!;U!7$LLrY_A3>7h34co!Ac&vo& z7&_RF(!U@Ni`Kesrt5&7#ZVYpzuJM$VL0%5M=Do4s91qMB{iwcuRytqQ(DoMGE+-> zL7rnmL(=*Xk}wxRbpgZ;nSYKz_7$@MZ(H9M?_F_5YzYO<7jIq&R8v?$&bak(I1VIV z)8;4-y{-f1fTpQ4tE^|*t6Y^u=SbKAXiu<`?y&7%?B6v{`Ft1-4*NnQNz?v8(eDq& z_W4lPo=%9m8Wepur0~0;5hp6n5&u1SUS6-1Ai?pge885eQd3^$kd^Etx|*f3a zd=knLG0P`K^h;4L%*_)v(^aRZ{rORy#CsXc{*ns8;5V^*MSc=od z*{e*P1N3)#_9}_~A6#k^*#^iBT0CtAp5kVRas zO8CynsBtAZOg=N#MCPm!uHmeN%(FU@$j`o6g?p1oGj3+^t$dee+05ivb$JH7^qTXb zeSZIi_#pI-&{Oq3=~6v0->M3kzQhE~jC8867<}xTw;876rcua_@dce=>*0PE)JRBg z))QalRAo`i)V&uw%F)&BJ+g!K2p!KLjU1>2AfNm&P2w_cjZi z#6B*M-*NB!@d3LbZ{_?q+k74IUV_B`Z-FM>M;qM_!tiR;qVX+`{D4| zKI~69)2RQ@U#U!_soWDl5|}A3L$$WaL`P>jN(DknHI*iEE(M`L5~Py1NT^Gf$sP+F z8{lyFQcXP@n9)zq9_HV?PRki&2$|XY)swcXrTmk=LlzkBQt+Wr5)8AdtnZz+gp%*d zI+!IpS}VJta)>0+cL;&CDB-5Cgqf!i12(&}rHdW3;gR7S zVGtSyxl*K!9KtFfj=RWj*Ee9y#!?iiB!GzzIv=zNs|!52!gmlTu3{Yy*OU2D_lNua zJWtcT{qdxq7ePC54W)|3xTY>BFnf@*E#BkIZFD!&rEFbull>>DZsK zUS0jER(z$C&&@kOb;Y;I5zoBC@w(o@ea4hQO~@aOlB&av&Tr>anLBpo{3syNWpONMC~!YMShaPz5yI=g%yyAhQ2c z*$&lzoc56NwvT?MR-GS_j?^Ba%6~OP1c_PNkk>Oov(lhkA7W@T`m`Vb<&^wCe<^7Y;LWXdUXSZ5A~&8sX)=L(tNb zni3@8N?yqzoe&4r((y9TEA|5X?(%ke&8UT%fm~WlNxBY>7P=K3BmsLz{XwD{jpX5L z_E6Job2zC+e9T(r5@(qNMv1;_qn}kP+5k;^O7*lG?)y$VH7t;TiH`QFX>n@81u(~+k+r7LXZ(sG@B_&+cV&}?5YS&J4fnbBA^BYJ+9oeh11MkAsM zT^q?v9V{;>ub#?(V7CZ}EZArylt5#ghVxGoB0NsX4;C$-&1f?%vdN4jdYxdg5eW0< zIF)Is)i@P8iWVfaU8&iWksvWqt!T$l1}QJ+`&W>>dLGS)DSSq(jon#rqtVo8h2#vg zm**G_><QDS#gLpQFxkYK5(oTl4 z`GeiCiXHD3j&{S!)quB9xSOJFM!W^%-OSoLa7sK|g|6>DB1a=<@hgubv;1BTsY?&PK>xT0p=V@~89DIC zZdm9%#OYR3Q;sKqF+hBBi!J9re%iCL_Ld1IZkDMIP{5CF^3@HbH^AiJogO0@?h^YS z4L$6?a`^xJ!d(H+%Jb>Prd426+Po5s%FS1U^M=hUqS|2Ns^BK=Rt7byT^;6t>sN?T zbL&-Nb*RxwkuBP-7HakM6$4bli{??|D3z4t8JYU{AUTGdH~WO}O+0;ieO(sUm~j}Q zWd$eqOhGiC-y=sa?$(sS-hA@l8cDNAdUAKfQL+-h(K~hpxlZ$P1_cgvO7ik#_BhFf7j8l6#W2oR%{{IK>*ue0flBZ;_G_YUEsI=^E>d}?N=5K=aRw&ybFmK zJ)&3Kfp4&%zP(E`K%Dgn&#}EhJ+`QO5G4Fk&Vz!cV=RR?^7|XObX*e05|6IZh8YC` z7_kU@3Es?AP4g`~w|z5RK^^@XB4WRVQ15UF=(v1d5oKx_IVh}t%!emezG#v3&J7(Sy955x%QP~+i6D;vVC40ZAU!I71iP)aXntG8?L zlExI5pzs6|T#8*}m z271<*LAaPL?*xltJO@ez-#}XY@rF$EzQu!*1tjyE7iD;fJGFRi?-mEUm&H+hq{s0UK{bcVzEoj>^z{x)%((rm{kwt%d3>vzd6&w!$(I1nd zoc=5%E(j_BolGz|&K+kTT9%7dA>HhaX*S&Wp9QWpqR>SI*u4E077d!XR}hwcv?`d2 z?#X~hEBev2AFx`Y0t}@v=rx&&iggBit74ty-Kbb^PFEIc1}zp}(R1B06<;(`5XUQy z%@dz^E#~m@4u=UlCs=1mVnIqyN~ttDTVr~WOR!HDw~$J2e0w(${ZY{_f>ZB6K3Gx+ zD>+E@z&e^?QO9W@(R%swpCCvI zmPk3wZGju1gx$dw7y>ks2s3W|e@tfuEO`BfHI-BY#epBpsq zs6?<4spckD$-4uBPLu85{+q(T7YUS*C1_ zIL4%}`vcPk7O=Zn+j2rZ#B|}(@{7DBcj`N+i#a`&f`DlinFF9W1Cq)IhsP${G@ouTIm~?KWyB|q$lQVRoBHkUb}6e=Sl?@b zcD2oyz@5@2V$Den^k}6o__3W#0_UdP=K^gwFcux*56XOvab74T)rryC-waMs`~t&W z0`?ONcTYtrs?Tl8H{Y{344`3WuXwRSIsLYsbsTxRSdq& z%T=Miin`u`bQh81CTPHS5?H6P68Yt1>A$|?xZe?Q8aNSv$)8L8cK@aQL6yS!1Bf7BgO8@*`<-J>){dP`k;Qdv%}bQRhO`A~a#QUsR2rK}SSA zo_(x{P2e(<#5=sw)UQEe6U2s%$DZsiOzeZs#-X9iZRXD+d+t=$4EF(Dt0oaM zNEUHl5en}&d9m@4vdI2q_5#YBPw4A3A-u}@#`_(Vm+hs_L$V6Hx(G96z*j`5`)7qL zQ*+_eLg3_vZ&Ac_D|5S{Tc6b4Y3!rmk?~1d&Z6U7#Ap?Lh?XO z`P1+V=~q{xgc3ufdiUZtn%kCKR9wex>+EE5hOAO|Ay&Ez+7+eQ*oTv0Z+MjK4|JkH zVmCT%R&dnN@fTS#f(+3&ifoAiBnAC?N>-xuI`1hPUESu_Aonu&UESTP*QkrN{2GMg zwKDS{ISY8v$n{=mnHy69;G`%603w8vR}v^p(0!s-a!vuV@Inp|&=#kFEf5x2aLMPz`h8J|UC!i~M7PvgsA&JKkh=ar?ACj$YzV=C__DdXs0eUa%$UQdcH&2%AwI(fIBke{PTg`}G)Q%V{pPYA; zJdLY5U*{f&phwO2?}~dLx{|_XQV}5tqm-`-xLBP{*ByOnfea~D)`#h8y)vP=&js;| z7xDqE4Am`s7%@AWi$o?a&YaA}s%1~Fq?IAw__7L8Zf0|vsFh)yc`_D<)K7}ZR?|7l zKX9xDqK`m4Lm$L+ZMyn{lmO`KFwE-hRfpp)uhtCb_Ds zRi@XwgNX6z7AFP-wOkSN#T+KddI^6?`AoDiR_jwV(-&S%$nFpej-fv3$F9Fb<$bQ+ z^(aTNVYR9RW*v0BfM76SETOV?v}=O(OG2e-rW(op%4H_CB~@h}o&izMKX!azs0h|E z*!)8VI7_|+EIE>sp1PLxO-B|1;U`Z%Tu~!J9cGc+^f^38bk%k5|4iyD{g&(9kS%OI zJ3Rf-vnAZu8)jbl?RJxBz~1CSQMQzLT&jxrc?T6~T%NL~T$h+DZTB$Qi{gWWL2|ff zOF*`Qk9Tx(^D%m;BsJB9FJVakcKFe+>#teO_bK<&Yh%^OYU%e%FDGu}|bWrygN^s~LhJ87Zi5;4f*NC@*v2?pjmk0)3%JMZm3B=G3|Hinm6=*Ss^hl zfN4juPbkkVY+~NN2F@ipANCI5-bdC9;EZ_l4A-$a2v#VKLurh=F&Gkl9&k`Huc5T$qoE;)(f3?~-JKwXmuK9UuE2KrzX{1^a@1SFO2Q4j@6;q}( zPfDYum>;jXQmPL=aa!K(dT!IPeHM9cA>OKl#`J#cT87p`=DDhPG|_VGfKY==GaN=mC-}Xh_$%)+uE@nHd8<23KRyx~b#cSNRbgOgt$w|#=0@Lp6*(ctf z^(tM{(O!1nj1OdC5OM~7`oqz1_$lyfD}Tuie$l?F(sOcH93LNErRn9M*gGD#^qg$q z6YZ^zm(zIhPg&F@U@cY#Pt)lPU7fl@eD0ukuh=zqK=**P-np6y(Ly{~tneS{S`2@) z&uIr=ag27&hf@au9h`yQN>JZ{*Jrp42x z_!_Q?zMJ3PSDiR2q~d4Si?!o@v+9~bftA!f53~%?^#aK<^JAm#j}NR;W*tcF0w=X$ zP0IR%-QC^C9}!x8{O)-lk9igy=~udwrfH{#r@8ZHAPx>1f;MPXsScS)u%MHv=Z2?~ zc*xh~Sx2{ceEgB#Q8$vRy+d`oW>5obPy=^R+q`g4`@+527d}3+srDKm537DXVx#HP zuEJjWS6v-#5|xJk&!=L0eh-E4q5RZ%Lt(!9$X34{Tm4Z7)rNKV+OzI%m37G=kZU09 z_d@4~`}1o+U-@+Mk$L}^dm@WR4LAzvUfxY_Cx|Ie4z29Mu1(TpkU4v08D##bTkr-x zcAPFgvh@FlV}7vz8}S%oiSYeh$T6rL!)b+ANvB?`^XYuTD$Kn{HpR%khFrRVJxyk; z*>eorM%J0#twHU34!^Lyv?6Kn9FvyY1H3VTWSp#uyvLph{X=|Zyk*G-yNSGK$%ZCp z)%0*P)&5~$lkFqf>%~%N{cpa18SQ3~Ws-@mAYABa!?0dBG#U1Z;~2PPt#RnEq7Td4 z#iW?Xg%Hqc3S)s#QXH`+_CWuJ5;IwjTdw`m!+QJ$v>I$~C2hGEUcp zyH7$5t)vrrF0>6)h9!h&LlCN;kp5F9r0*SGz`ht7YQJWvx_N`4_BRZ55E!buxbP2h zU=7l_xu|Os7?WD?n@2QIBckB!2`192+#vfG0o~_nd_mobF5(-8joy6Y%nqMhskc$MCl^~w{tl0LaP^UdH=XHkisfy7FQgq1Po=XG zq!JJ^*my0?r^>LZ8NK}WL3c}j%h^z>qz&zWZc!S!LfGzjU zjwPJvRtHjO$F7c%`k@wHtK)d%Ik&|Df%5uut3R02o3cefd)x-<8>M@*Mpwo@hm|6fP%N_1u3q?w9!tj~I2m&dQki0i0L(Pi)^BH2pO$ z)B>+rQI#H&^+Ec|j4ch$w_x?XpqFu}*H&t;PwJu~L~psWwOXa0qVe2Vp^jTh{g51i z!nz!DukX#x*$6ybnsK2$H;|Fv`Y<<GL#6=uV}1k8nK~Y}?Ij9Fnm>ob8?ub8@F{>Ae}i5@{(M=1G!PpOwdg5kQOw#i;z8-LWASwS z{Ce?R=6ex`n%I7??rzyq+VI+c175maiLd&P$HzwpSNmyN?Cl-(2ggmX`WqmnI~DPI z{}|#nWBBJ_OZZ_v6ohaJ5s_l1-W5y6A|@A8g-?t)#Om;FjYsTB=)MgcaUGeaXvAR} zCK$0$Kqr1kuWt=AKVImHiX&8a2b{>PMv|*5_8fNq1 z<%Us=nvBhh{u+iG@?ceL2i)p`=GX?eHN3YphHVX#Xo=doiBJpBRu96@Q!ra=dbTBG zs|RJIHDYU+Mr**2*p4{95v*6@DN9H=$i zBMYgf4mSJzLYHx7oK2|NVLZYbiOGvX*h&oC4mP%)P*b?rIw~eiY%5{VD-Oo(#KHF9 zuyq}20t#E#-UfqhB@}LjCt63RiX+-eFsz4FY$Y1{@NHWOhaq6lR^nkR@Xn3xo8dLq zk#TSuTM3Dc;5l1~iS-zwtpvrUFqf@F#cdHOy*O^$AW_!sJ&h44>xi^Op4?2T1>&R^ z7V;@blbZ={i7@Gnh5A*HC2J>iOGHU8qF!qx$vP^n5hOPg+YULh$xxdjM%JJ6n~)+~ z$t--Nf~+L{?XtgPUT0O$~v4%3l9%@@c#f<<;|7|I068<>pS!S literal 0 HcmV?d00001 diff --git a/validation/retrieval/provenance/2026-09-27/decision_value_eval.status b/validation/retrieval/provenance/2026-09-27/decision_value_eval.status new file mode 100644 index 0000000..0f1bd5e --- /dev/null +++ b/validation/retrieval/provenance/2026-09-27/decision_value_eval.status @@ -0,0 +1,34 @@ +1 .M N... 100644 100644 100644 675b63a3ea026a9530247f527bcfc4bd88f3264d 675b63a3ea026a9530247f527bcfc4bd88f3264d .agents/skills/agentic-go-context/SKILL.md +1 .M N... 100644 100644 100644 3a4c8cfd9ca9fc535807d49a8c8a52f52a641cdf 3a4c8cfd9ca9fc535807d49a8c8a52f52a641cdf README.md +1 .M N... 100644 100644 100644 e6d495a1527e53cf4bb93c5f23dd0d079dbc4fe6 e6d495a1527e53cf4bb93c5f23dd0d079dbc4fe6 docs/README.md +1 .M N... 100644 100644 100644 66ef9b40ddcbe31f11d5a67a86e7b49dc0bbda16 66ef9b40ddcbe31f11d5a67a86e7b49dc0bbda16 docs/continuation/astra-understanding.md +1 .M N... 100644 100644 100644 d12d50cfc3ae1eba142deb03f249e50f1482e5de d12d50cfc3ae1eba142deb03f249e50f1482e5de docs/continuation/go-intelligence.md +1 .M N... 100644 100644 100644 b7ac0256f53df0f8840d2481e9435b22aa8b816a b7ac0256f53df0f8840d2481e9435b22aa8b816a docs/go-intelligence-north-star.md +1 .M N... 100644 100644 100644 cedad4a163d4b5e96dc54ab6efd7dd265baa67fc cedad4a163d4b5e96dc54ab6efd7dd265baa67fc docs/plan.md +1 .M N... 100644 100644 100644 e9135660feae19677d29865ddfc12046b43015a9 e9135660feae19677d29865ddfc12046b43015a9 docs/v1.0.0-roadmap.md +1 .M N... 100644 100644 100644 c8042f1273eb760b6780c5c6a928cdfde4f539c6 c8042f1273eb760b6780c5c6a928cdfde4f539c6 internal/intelligence/core.go +1 .M N... 100644 100644 100644 0156ae670821d1434510cc3f2d8722691915133f 0156ae670821d1434510cc3f2d8722691915133f internal/intelligence/focus.go +1 .M N... 100644 100644 100644 33dd43bfbb303eec2a2e8a7676c2524feb19045e 33dd43bfbb303eec2a2e8a7676c2524feb19045e internal/intelligence/focus_test.go +1 .M N... 100644 100644 100644 332a6d2a4e3e11cd9dc7e7f4df88f3df06ab8f10 332a6d2a4e3e11cd9dc7e7f4df88f3df06ab8f10 internal/tools/intelligence_tools.go +1 .M N... 100644 100644 100644 aa5bdca00581d9b79ce20ab891dc5ad8d93ba69a aa5bdca00581d9b79ce20ab891dc5ad8d93ba69a internal/tools/mcp_surface_golden_test.go +1 .M N... 100644 100644 100644 da7b976e5e8d47910b5882d1758469c3cf759437 da7b976e5e8d47910b5882d1758469c3cf759437 internal/tools/runtime.go +1 .M N... 100644 100644 100644 62712f63eccb9257acaf01bc014085c89294f9fc 62712f63eccb9257acaf01bc014085c89294f9fc validation/cmd/eval/main.go +1 .M N... 100644 100644 100644 e4641aab1a376266fb8b8d663d26faecabd26977 e4641aab1a376266fb8b8d663d26faecabd26977 validation/internal/adoption/types.go +1 .M N... 100644 100644 100644 17c0603261a84f478824405a3fa8fa0cc01a5b9a 17c0603261a84f478824405a3fa8fa0cc01a5b9a validation/internal/adoption/types_test.go +1 .M N... 100644 100644 100644 0e50be83b3765c83b4a7689c17cd13f2c27d4ea1 0e50be83b3765c83b4a7689c17cd13f2c27d4ea1 validation/internal/pilot/runner.go +1 .M N... 100644 100644 100644 06e997f6ba9820dff5e20ff63cbd864ba30ecd41 06e997f6ba9820dff5e20ff63cbd864ba30ecd41 validation/internal/pilot/runner_test.go +1 .M N... 100644 100644 100644 9c985f4f4c852bbed7b3bbb9de5573ef789e0cf4 9c985f4f4c852bbed7b3bbb9de5573ef789e0cf4 validation/internal/pilot/types.go +? internal/intelligence/retrieval/index.go +? internal/intelligence/retrieval/index_test.go +? validation/cmd/decisioneval/main.go +? validation/internal/decisionstudy/decisionstudy_test.go +? validation/internal/decisionstudy/plan.go +? validation/internal/decisionstudy/prepare.go +? validation/internal/decisionstudy/review.go +? validation/internal/decisionstudy/run.go +? validation/internal/decisionstudy/snapshot.go +? validation/internal/decisionstudy/stats.go +? validation/internal/decisionstudy/types.go +? validation/internal/decisionstudy/util.go +? validation/v1.0.0/adoption-remediation-2026-09-24.md +? validation/v1.0.0/command-activity-audit-2026-09-25.md diff --git a/validation/retrieval/provenance/2026-09-27/decision_value_eval.untracked-sha256 b/validation/retrieval/provenance/2026-09-27/decision_value_eval.untracked-sha256 new file mode 100644 index 0000000..151f430 --- /dev/null +++ b/validation/retrieval/provenance/2026-09-27/decision_value_eval.untracked-sha256 @@ -0,0 +1,14 @@ +3d85204f2b537d09d7b537427d852ec0a964e581a21670e858ed196f09f94dc3 internal/intelligence/retrieval/index.go +43da731f8f38ed76152193dc15137986814a9de00cd517ddf5c90c5c00098162 internal/intelligence/retrieval/index_test.go +6d7edbd984ee7bc1d38dc5108e5dbf7882c0f62a4ee562d28bd65ebc5097c580 validation/cmd/decisioneval/main.go +b178dc3f8bffc3e426beace0b689fd1b6da29d04ccfa3dbd315063f68fa88f09 validation/internal/decisionstudy/decisionstudy_test.go +e5fc5a28f61b2fbfaa6505626018d2adfbf4f8e9d10b9cd196713d1bc2170e3b validation/internal/decisionstudy/plan.go +625956a022fbc4287d29679b056f21947cc4d9de9f69dfc268df903955f9c6ec validation/internal/decisionstudy/prepare.go +8cfadad4df5549dab4bf57b130e04f6cf7b77dd12f4fb2de0d37c6ed9815ad65 validation/internal/decisionstudy/review.go +014738e213f7f1e33a11545e0666c7a98c1ad814ec4a398ad7f23fa93acf21c6 validation/internal/decisionstudy/run.go +760c361ce85346058789fa0d0cc747e4646c68573e5948aa78411ca7a3ac790d validation/internal/decisionstudy/snapshot.go +8b3a77e003bbca687968262f72729f547ef4557a6db2941b5ed4508eb1a5c00d validation/internal/decisionstudy/stats.go +c32360bcfc60b7739e499bba726bf439e7d6f7fbe8738aaac41c7921b13bef5e validation/internal/decisionstudy/types.go +12a6886b1d15ae783b1ba0f5d0b54bfefb99ecca1cf30e44ad98ff135164834b validation/internal/decisionstudy/util.go +2267b58f47f8eaade322ee376d51764e34c20267ef723bcb199d01b23d59bf70 validation/v1.0.0/adoption-remediation-2026-09-24.md +04db27ad4bb1fb1ab33f5049107beb5848860a65ed3aea34fa8d022021d80c5e validation/v1.0.0/command-activity-audit-2026-09-25.md diff --git a/validation/retrieval/provenance/2026-09-27/v1_2_reliability.diff.gz b/validation/retrieval/provenance/2026-09-27/v1_2_reliability.diff.gz new file mode 100644 index 0000000000000000000000000000000000000000..6436be38d24d0b871c111e8952d1b19106243a43 GIT binary patch literal 40777 zcmV(>K-j+@iwFP!000021H`@Sa@$yzCi*-16zI(8F1rYj*44HY)iFw^Gs~y*;;~&- zC&H1Df=Ez8nYRW&DOK%?mpDwTHm_ty+KM=UiwVOOqNB0z{b9;>$ko&t!J~W zC@$+RE61bqvRZcSxcyMi=j|AO)sy0KRZLdPuKKwfzx|I_-+ntvHmeevy5%%BpGI zwS(-7>`d;!pS_Sf46;G?HT~uxcgV9zIiF|em#d%j`R6tO{=QybX614^D4J^4RP9w( z&bq3RpN?Ew^ysKxL9qL)2zC#r`2*&<=JvYi%+lJ zG}WYAR&6^NPUdC37|{z9d!y`aS2kUCQ+HR{`9;}Q=RA~bwXCwXn$u5K%}I9tpPQ<= zJel927{Pq{!Eu;TCpVjl~e0V;um;67I)qJy9 z(j_#(=d;yh)Bc1D#`=d&_c;wiAJS%JBez;?x(&_!EsbR|-?a60H5wHAqwLj;W`(A8 zGw*o%%Ed)}xmj&!sg`Ba@SJuTT}~_0(M&g0w`rEuG@EQ_9+w@h=W<$4%dTn%1wEx+ zwp~@yw9V+!byaru!?s*h*+sLW@9@_2@PkqIb-k=!bDZXN}mdvjIO9cVFWG-bwX zPs68em1F-d7W7jLgT7nQdCIQKrYxRa-()0OKJw2yInBSges|B5{)qLvjC{K#E@d)>_o0pemy;SKp+cY#qDg}AfT7LP$ zq+ghY)2zH+)l=N4siq_o%VlR{rzN1xMw5o=w%^jcZI+X(a(T&fQykg}X{rtDkZ6Em zC+VD&ON@wC9iuE|SYb3dq=X4_9yVT`H+Fw3o@Udn4S}2a7912u_b&Pg{b04moa$rj zA+~2u<<#j;5DC`NeQ7(ex)h0rlQiRlv7iO)a6kGHo!*HY=5oEB)4wlB`E?{t-Zgw6 z!o`E()8F-WvXba2i zMk;NTX0^G*@JPIPFgty})ZriPRX(qSzJ zqoxmM&1xZAohHd`K>ml^ZL_3zy{?utFUx9%Z3nY7TafCWw`?WhP!5LgbwOl`X^p0< z*-WP;{;EGmm>53GHa>deZo{_f0EX%H-e`ZcXFG@F(yFC#?;T-3b*t69J;CXsH~9yR z^O`Q*G_(yW);)WNatXcSrp0CDl&6tSOkzOuMKa&MgGaCj$Jcx3Y#`HCUPGZZBn3$= zWsCAdC3l3eWAa#+%d78-gH07a_t2N5>x;>pUU(X* zXr7d|;x(b4!_IAU|7X@!3tG;Oq*Yu0%vQo;%{~Ke7iMTe#v94FhW<;^C(=}OgunPA zd$nXWF;9gERx*|(=?5xKs^jV5^CR-zX3whXxZF#{N$)2jQ+gk=V0o55eVPrQY` z$1?2N?+d!Aw_;qJVw^~HuQ7cM36h;pVYDz+>S+HwGYAL@f zIs^GdT7?fnN#kOZl4k8*9cB7yzFMtGoRV%L^Y4sxWQ!FmWW@%B*}yP(e-^Vs6fjIO zxLoq@VNY{U-Cxl>M-M63v?>TS{<>b@4Kq$dz~sMVamG z;~L(Omn4MP6P~UR5hS~1PWDoAuK=p><13$ZDdw9TYdDcCV^(-d)|Y zh~0ZK-g}bK?dkNB*kW&40vFzkG_(#M*j4AP5(x`Gv9SD z-wOOLEIbE=TbI0+n^tcXtd3XV?MZx*dd;RAmB{=`hX;H4@!?*!pl!@<70to^{z3lq z=ox(k^UjVBJ>i1o(({9VCFOES4*03v5mp!vmt1Okn{59u+>G2w5zj1MV_VhN0_#2J zxn3aBgt79IRRu0C>ZQ#J-VzRp*owKR0T>$mR@YTHjl6f*Z|0wwISd(aYu+`9iIVmz zdE<1;qMUPlXi{$4lGi?mXoQrCBd59P3v%h$h3QrwXwO$OvgT{{MG>8%(^*Y($Wk2s za!%*EzFflQhe8p>I9GoZKeGF-*g!0&d=^_hW^K2bvehr^d$XduO{yhXODmDo=a=iQ zI2s)nb2>HW8TrhM@@KwFyILYf!;&vL2fqDr$3a2POI1PRbhVteHOV7;2L7Ov!J!@z zO=^Fdq=2k$SQC)^V)N_+nWXMv6z?Nj#8+K{0qu{EMx!yU6o)-nPl?$zw#y2=%$$uu zv6{27fBj_i4A=JTYnq$LACXBb5wE(WJFyl{qvC@?qA@i0yb^eqVYD+ka+>p0GHe&zF*d2>T4mo>s>a5Wc}g)zW=hwy9V%$;pOzx_tBc z1*FSCeE#ho;wrG76B8jjAB{(&(Ru1)k^sA^Nvn~r#lz?%uQ%{1Pet_geQnmFv?mcH z)VP~Qv<5pOZKLePoVI_**%gSk1q~%TC=Pv?A|5!xD5A<)B?mu`J%9vffaoG(b~JerDOAfkCjcSianmg}V6dArS0g{5+fW1EI9x9vTUi z_iFAASjv*#8?jUe6Jbjkg=m(-jz09c8V4-qs&Ie!jVry-s5J0(1}W8)`Sh6 z&B)&k-X0qWXJUW3O%okz?BCwOgasiQT73l)ay*_O?8^3LmCl0n~sg8rJNXhgy?y)j_yp$r+P^;;amc=@)5nE1iTwwJ${ys z0fH8{q4&OQ))PJX*b=kn@m)nTA|F6bsV?Y0#HUz^<}UQL`qM6CD>TIPmES6z}u z-~cg6(H)1%#QDIERDPf3phiMCb(tQzFGaH8JktPR(eO%hkn^g%Casi<6Sd(m^<+-_ z5?fR}XcgsQXMLB2D#r|s}aSC;^CHNZrJ&5J_mz$-yD?DS!%lSHq2#!X_7CPs3 z!jJq?Ie{jv)1Y@6QR6d3rb5@xp8P6V1gYUc23ttFi0xJ}ymktW3R@}!mN5F`@v$y! zh}_a=u|UZFQ8gOOY#m6 z3zlI=z=0h`HvI(|{Fe=%ZArD+sQQs5nob%d&gD=>GOOwKbUoc_!EgYMnnY`=mRg)w zg3hKX<<(*N+(97gi}BP!z_@sad)@XkGU>qL#Eb)NcBE)@rpD}&EY#mvss)aHg#Dli z2680_!xRW~_WIS>L*AARlO!f17(~}eV34d<-5+@XpbUl{!AT-ZJ0eLUI3#x5O0Zo( zR6g_M>t0q(!MEW@&_CrAu)Wc)YZV?dJ$xh}imqPJ{*vHWQQk1NX6S3cF_9?^Pe4o> zyTn^@2D@en&8zE{OItw@RP$SnS0fKa>%b`*zZEM5jt}y#s(yTAe~#_S!mcd*DG1aL zhOZ+1tTP=4u`L-SlA6&OVseIz1SaavNTb=blGVO9&N%!vaYl(PzDu%WBz|eRn_HfB zdAcRZ^6j(|M>0Z22`GS&RXSA(A)9G1+?`sx0BA^zFUjH2uG~A4N6O&%Kp-s2X$hNe z+V(c#oV&LOU~b7P36O}6$5CV3$YmPl{UdqYrNn?fW@nPPCsW}gf5Z7G{V)IdLS;uT ziB0Ssg#~ab4Vzvjz-0otzI0G%uphJc-#=Vq=;L;YwmdZ{_-RqxoS*aJ*NJ{S^ULVS zMaRrfX%5{zspssyvVe0p!A@ReAYrL6pt}y+!_QedC6khrJ6R%sXnV6WrgfK;2PYQj z*#^*yXH7-Jr^7-1i(>!D{&Df-d9i;O2J-`=Lmp&eG)z;Amq`!)bI@78BK#z8o1`=Y zQ~^Q%kieve(UGO?UHyXu*Y%UpbIv8o@h$`!!nfq#zjr6Mm)+zC4u=0xRqI4{P0HnJ zS$l|lQPPV1>;hG^hy=-^c=IsWaddY7W=BDMv8g1UA{&OzVns)?3U1nI5@Q{@h8Ek7 z?;+qfv7fG1tk#pM@|1NHh{bMXc?LVH)koAVr++wnLO0>B5ENgE+4#+h4$68SJUDi% z7$ZQlMy9#JC`d^I3sKETh;#y~Fnz#MkQHOawmhCm+7#fS7$E_W;gW$Aq|Bh;YgT7b zZlIiA(^`{eC)ayJ3X-mylj>MhAzI^#LKp9p76Q=`w;?%z5XD9`cly^j64xN;8e$=I z&31nO0oI#k>>qHc;kl2i+4GCT(dha9lNmXyJ1_wcxmko3c*x}dDCK)kvLXHVi1A6q z7uoID&DsB%TrEmY9rtWz$&;3I3lm6? zn!HoC#$u&u?|2o9I7F=>5Vl_^i{9r&n%ie-5hXGeQ1CwXu1IR$uP#r60d#kQId}w_ z=(c+iNjnszF9qI|qn>NQV?;g)tihH|y{6Js{lvgZ7R^G=v_&E`_BFML;6iN~^?brg zr|ymzZI0v>OFqO9;(#n<6EDnQ$WyC!8LcM?wBUOSYX4z7g{-&-ZWTHZBw@ZT8KPY> zdi;yz49)xt5=DRfBT1quS!yTQ_h)b3{O~u%kdsdd^Xepf_H0ZNVf4p8W?ydV`IHj` z43*4dujZT8V26VO+*s_^on#mFGMOaA?g1|p>=op~p$LY< zPds`599^!?Pt)^`MfVNMv1k&j$X#xhn!w6nz1mW9B1UrVDcns9B@`*j8lIKHAcKUq zaA$XAelvc>T7>Ln67k(Az+leLzIplmySI$K$ur9#z;aR8iv|+tRzRamM*Sf$y>s!i z5yH^SLGBzs?}fjfJPqQ6Pt&wg6ezrJby~M@A(1!4}GA(Qpwtd<8hi@ z(X0U0!N|%KIn5{kfB&!lN02c8`+xnvaWG@xmPDx8!2K+aJLXgaEdz&6=H;y)GV&YY zM_pAw-^=D4WE(|yhnOI3XF9e^8TE=@m(0%DEf;&x6Wh*2&})@?Gn8{gJH9WHR6O!5 zo|yr_t90^dOK5uo?6gF-FH;L>y06J&%f3!FxUMi^psI~^vVrY)FE zeET$M_1_ZDbufnFJ_Q-~@Mx|t?_jtc%`KO<*sxO!>a9+oKC7CVJIFs_x3*P#@|SK; zSK?g#l_3x)2^7sa=y{34r<~xmiR+<0UqtJm$rE!m1N48CeaAO0;)Co?q^&JiV32gL z%pI1o1gCf~gf+?W#V#(bAfKYXT%uwlY&4e8?@aF-#521K;Tc>D4`~dt-RM%WR5)7j z0M}!bovA4dfqJl??~}=RaF8FMFu0E5Q;wFf)ko1{YgLN~mPKWq(+zYi4k)LqdM1xV zKulVf7J-_tG@7nuXJU)o(6 zz7f@=a(e%AsX%zkT+B<_I9_>}MQ@NT*h=p>0|3N|U}xU#Fc@yC>1CywinCS^tcn>F z3_Ff}JhSy+$ERgrAC`byxVCdGDnkc4eqJ_xmFN0Ss%CEK2rkS(?aSVPgHqvZkh87o z!ch7c<2M+eKUEhW{!E-wMr2gVlzFBiJ*>vd9bo6|POP&AP6~h>q08P*C-7`j)(+^LJSm}tU%f<$@t%5eQ7|6a*+v>Gvc{n zJuF+>n>VtJj%)~C#xr$-ZP7S&69u;rCS?4GZ##|JX{uqflpPBRZk5S#ibljH;9PT# zUfhZYv+Sg9V))*2MsfD%*Cv1@8{NR66g(^PPT|Zd3=qZ%fP}i=F0rNptFb&8-wYTa zM>W2wB#ouwA7-V#RdSyR(d%5wb@}h;NoZZ_S3J(4xbK&JH(&O)&t5JgHV*3mUBKTKS?_Ed}l37 z#M2Q@0&EZRf2XAGayhIJ*)}f3wZ~siCNEtGGw_~Trln@5|A7sYpSS56U7;m zIB#R=^Y9u~+05&zQ88{jIvP_auO0x*5Fd^1-kwlpHmKCZ z6uUS|9*4~t`)(feL&iWchyrJVR)c=ZY^^PJ$0zee5(9GNhf51Hse+#%a|%NsSbpM$ zaaA6>gY`cc*#=v z8uA!#*tWb)?07nODNV?ksVJ?_Oe)KpD%PhF@XQVy*8uG(IDmeoS0`Tp(W!K1$l{x- zGrJvmb3h=1a~d#;58#Z&t!BAQ`Lz`nxR4qznR}Hq|JcsYn7<>8e{^6Ik5BL#uMC^tr@V|=QbuFLNztN+SDRO zvj4LPPM9PD#i)nlvcixyOR!+*fsha=s%KM_pUNH1n3XNOT8Ul=1WhZvSd&OB+>vc0 zR7HjyALIv|i+Q%E!6O#ARRabNU}WTlc<*3;TKRnxv{$W6%GgO77W@jFRGQZdP|Jxc z=>andJGFNO&QuL8&8W8RdsDN1vm>UF!dGeX9jm~W;QYFZYDsw7TC2RBAwF1JjS&ke zu_`#<1?y;WqShyD2_U$!p&OLj(b>GU2-;nnKB7wdYN+Yv2RV7uUdf`f1ki?|?9%}Qws-@GEKBg@dUItth zwjCNTIrWdF3Rm4_4b-umnAXgA;H4#&>qI*LR1Oj!gicBU5_?*`IY{TAmE~}t_EOvh z-q5V*#4D~_Ey@_l+#i6QiUqG}u#eeGK<7~lHP$jonwB)gI6iD zKhkipy^%~u+o57XiouF%c}-T;I+UUWD8B^QNZg}L1W+^GQ!yu+jQVivT$`x=$XQjH z@`Nr3=k1pF4SC>GmJXF9UKdq&wPLl%<8)pn8V-qiY#0N|I9!U1qsg-E$xLRAQR_6EmXL*>uP$LJ z*(S}jK!6M zW5r1l!CY<)oZ~Cwk1#N{lh{&LZUym-#GXB+4K)=Xpt5?@T@`kAe${&8 zb6#s=%L95sj;e57Io0B0e9ZnQ*c@rT$FMZmjA!W(SleRm&~ysUx)2A;m4ftzg+rv~ z5u}bDebPr8pGyQ~^2g9Ix&|Z%Qt$i)*<_=y`S<6sLC1!b6UA_(F(CHt4^T2@t?oq~ zbz<(6AMbinV(X-jtDj5$@$M7u#|NTPLz)Uv2|3j$0j5Tr2|`Az(?+4aXv8}jJSI&6 zBCkBtLRu91TyWTjjn=qUOJ6a_7KZ0$F+5i=34MXXn}7vkNt0M^Z-5`>S}VzzU5WLa zTG<{7HeaXaHNlEx=x0pYGZd{4r5sZo6r&fiHlm)1D4ERbi)5 z53uibHo=!|U@1)yW3-BEj^J={u+)!oSW-w1CkiGA1Y2$@EI5TvqOP1kogg(P8)PNC zrRI4SEZu7pNF~5AsRN0h*$WBgtpRJOVmo=?-Xhq|2}wSrh5#_&l1qt9q=fiUUPhhx zow5>K2F};UtQKm8a0p(fLI)j9{3wZqCYeoU%j#;t26%;8P3dlvd51E&5xA0S`c$UX zN+C0oDsafh5m$5k*7T<|vPHdd;psYNnL#Epo!P6cvk*$(0JM+|R^7@L|=Go+8>v`K%Y@p>0s9 z$-Dr}O!XBzNA`}6*pc18A^(P%1~LU}UTj!RkX-%Cx@8dALQOXIM@)1W2d7(Qxl!Kc zgKSA+=Q4svMTOtO#3|X+R$v}CS}VDXRzadhqe1bNsw{REf=?&eUnzo5YRHv2GgqG9 zJPsn_c;C$s%Z3#_q33PrfmH9EDZIGt0si}utGOH%#s(R|aTj1XBMc3VyhfS4p7#B1mIguYLc z#+D_xG?3u=O&>UR7b~qX+$KPfsPV)(qKo^&!n}WQnC}DZ>x@DnHXc;A^)*m>u+|Kf zSY=1N!2@Mb$>U@q|H0Nf%RX6PQSQY`viLp7=(FrU>Mz;aEXfEPL}@}BsaM>lleS7+s9959igj!e?F(dt(|Y?kna)r@6Tv^?^ht6O{Sqp+ z6oPD(dqs~aKGaCLgS8{9ItM-B5C=al#{PRfl^pp`!9m}^U z>ecEATAnd2(_rXCdS0F_)DbS#uwmk}sjU*#(E64H`PW7~l zBq1R_*j%0NT$0a1$b(ZVj-eUuNa|?vB&O@-(}X8uFb)rji}|+9-ELPW#L0rm<+h1g0Wx zr&j9$q?gQQ8Hz>%R!Kt{mY!p}bjx^zAal}5V}$NQM%%h15Pd&wGzWbXQdxJ5wRQK4M!QtQrXqLQfu|2ffc@?m9W=3kzLX8)|H0g-zI^Fg-C_D{&kuvPXh4hC^G*>mCPU_^bow4W?GVrQ9|NLQ(2hCjuLC5m!=<8-o=7OXe@lDJTmA>?hSC>^N3$p zJ@s}onZThERx$1Py#BY``*?x*$IJ?uX$oAua>a82poUy~AR{8bjy0`W!mOhaGiwTKc@!4_gd1902xG_Cd*;D*C>;_U zOdYze(YVG5S99YiT`dLkVKk#ViK$t0fPl586hWHSc^tU#)y?s-McJKLw(nyW_+}0ABT@467@dORsX*y{y*S{ z(h4W~@R&{Y1_~^kzV!tRRGN>RSDNmMA5T^UM{hUM10?4EAI=ZDkMx1 z6G6{{!{`4h70qiFU@CM5Lm@$J^(Sn5%jJ7dM@RoEQr6jYd1Qy9XY#Q!569Tup*Nhp zMx2erDuVe;h*Gr3n1^^6r#_Zd=!DrCAGZhrkR;%BcZn2srz^$~(!ULEWzckyWTeTu zPTG2!=mMB^F1R4egi7gY49wKoRl4lDRrkd7N0fSrD03rfG3)*HJMeU8zb8+3_IvPj zqa+L!#ZSP{arnd-IwGcj2d6RZ?$MKc|M`JrDxbf9rci-82pvrmm13xCDY66%({$0{1z&oO3EYQ$%g(H(v8&I;ALc%)j63J17&0L~Mf>4Q4Bt5816gE-b)1oUc zwHv1t@hYhaNjof+6AtCsr+Y-^TMMS;R=jG1;io7ugbU*|DtvWMIVLSBAH*qaS%cjS z7riYs6wlhTk#}v4{1eSu6;Cu6Zt-N@Sr`h7i0N5+@i{9_m|U$^Ue$-K)aSGA5imrA z?OKNfJ{1I!TCvr%#ZnB8jrO?nVDSO}Lliv`m2s+m7TT5!!_H9-!K$*T86#eVq@@=v z3)?W%oWQH^mrLGxJo?9Ct9R;P_^Xb^ori%1*bbVddryNA9^HG9QY>9OEhkU*k4B@} zlfCDYqoW-ZOZVRz*2Rt^~XuO(p0oPS9vkCcGPnI_7FHsaIG?X)Kz^C-1} zeuhaU?&g$;NtAXnC`q|Z+HPf6mOj6W$RUN+bJdL%o>|FCdrw!s?B1q)gF-aIjww6!#A_-I!MZ~=rnHVEIjZt&Z-_&$r zcWcbtAVUb6dIUN&UBHM$J-u{W<9$+?gDWu+Z}T8FrsvA5rEvR z1q+e31`SO(wB1I^=@N1=$ysHP$V1r7vWGDLJHtv}GXKgwHyC<*RNL~(R+Qbb5!1FG z(eAVJAQ9|`ZoR?qVS8`*i}&6o2b7j>eI;c7h)b(=Y(-EoY&3H9%X)?M=!aH9;u+J+ zv3-B0!k&Jwyli0V6Z(xFRbMpwxlzlwDSQi)97;x0K=kS;t=w&TBc@{`Hd;GYh?!8L zGOfTcM#HN0oAAGR1OV^Jt>W;I1<4xjJz{1`HTi({4u-y$46FvhT{B4Kj=}P6vb$<> zB?0|VwJB4M>}md~4kRbc`#W?Tcy5Z=aE2Cnbhm?f=&DOvj=@mzijFqGk*!L` z8?z$;r^VeL4T8l|EI{g65@5+#-Ds!KG{l3QXH1H+1l+ICrX6Muppc=vMTTyH;EPeOg!ZV=UT}nR zQz={1TaG~Tye?xZLmDB#O~7n4b=7eYMZb5dLLk{Bp;;~v;+z3C9WPQ^%45g_#%B8J zlOE*Yo!<6>UX5m!9M6@LO4w4m=NctiQQl!iRC8Wz2@czFRt70x=>XhL4v)fjJkrWY z;Fn-J1V4(UM@6YA6{FOgW>>ds2Bo$sd?^ns3S0?(Yh6JRxup-g)nqj{!S>x&PZB9c zl)7)$l!iKxM3=TEXip^9^lf=tHCoh%Af-R_WA@T^kFXHmku%~CU*Z7pPpq)23n*md zN37(9?lKGs%mPn>_kFo!4TxW2P7ql_cqySUg<%569Rz6(>SuhgU(I6 zl8N^1w@LI|3cr~1N(xXO3(Fq;Tq+n*T9oua6$@q38Q5G8x>T|I3C4$~zW1W~vbkbJ zVzD!dH$?)R=3YC?pq)eoZc-JrS~ARc$H!%BiN=+cnGun1T~R&E~J3n4JBS$~sCFxsZD5R8JAnT3R|#W?OI zp)$(;%>JFbOZfiM5>85uXnSEW46(CCd8^5#QTBQxR;N{}Ijg3ij4c}q@CVom1cWl> zo*@$D;=&qQcx_RL$~YjHfl{oN64uWdftpJz45=)c&y{`{s3ReMGoDwrQvd^u_BE$5 z1IJz&msZLR;X%+B7bOR-=A2dRt`-JRkBHaGrS@A9i>Pc!qcyB(c9m@^NY+8jQgx#D zAfs>Hqo2jwNF2sV?XojvycVRPdQ+Jnz@`@ujnqfr&h&VcKR53W%*XdS4$#l2(Zqw29@@l)atD2Ze>Zd0wSDLu2*2^ z1@>boi{RU>+Mt%%2o4oG7u<+}tPz?TGE<_OiBa~d;jnI0wZS+xqyL@89j`(C&M(E< zRXh6V2n;k4TUBZ8?N@1Whn&p|P+Bb+WMMj9LRIrhU#J$#R%vizuAs0prJ{wzy24Y3 zoGBGSE4G};E^GVKD~7T{s1<6Z(B9{LXc0y0fuaSWvrAfr#iIVlM^|09ZcoPK!;PlZ zwe%wz&#IC)UOQe)){p7T)*40g1a-Gkhf@WnN^wKeyK7VNC(7!A?=hiDO^bnj#Pejw1 ziQ2^o|I&T(=r&REcK?r1XyN%REOwCeh zI?~=Cp1El<#Q!0gwM58{j;4GF;K+;zD=8{(;SMZ!J+hFGB9X) zDT^X%>{inDhv{(<^qg#C^QX3j@r?>*0Ej!-C zp4rf+_fd@U@CXq)muxK_@PV8daDz!Uw_xiFu&_bF(WAX5iYUYn7Uh&JofLecjpWa= zz5PPy7aiKN>n+Q9yZR-AIn7GN8bz%e3;e`hRDmxRNd=NP7g$xA&Ni6GNFi{is-` zZuNmFkJTKWa1FiSpTOM8CIeg0()4WoK43&XJz&^)XBn z;-tJ`dMGAuaAoMD;85gqDWNwbH{+D| ztrTk}Vni4y4N@l(Q2@Hmx*wuFq{Iy`soNH(=%6_Ckqf{XU_SUC`kH#>VR4iMK%n7B z%nk~WDszxt0<1OS3od-yv~5a?Qo*33_vg%>K#ZM^49qKPss!@8z^bb;Sh6-67FMo! zK?z+5M+9AN2OO9gj>0B=9YdMP zjf{GF$C9uuh;DJ5CJ<*$s0tj|Bm6weR536WMkwP&MlOc;A<@QN_&wpN#Z~!9bo@?msFE;g0wkgj!LoEaxdTg8R zV^{|nF9ox>ZIu{AZIivoN^}L7epa-N;F06Fi5H~qtk~<*7=jOaQA_Dn#y^da-maqs z1=1Hx(@amW^o0sG-;`J5EKoZD{_>*-Oxdd%x|SXH--xwVH5QccgCqZ z8LS=MCA%xc8oHMLE$AD0LSm` zseE^; z+anpLR>Y=cwT24v1_~r$+~G}jTH-1zI?rhdH4*M$=MbW8#n?{KioM_r5F}e2l%tT54z@x(o3kn>jr$ta=mdp%3F1j%A*pP|w@Jteh z=q*T(=)M;=;ZJsOYHGc3lxPS5y{`htPtbvDgc~hLxq}O{A6R*@qOCjW?u@L2^m_;e z^mLX?@FE*FXGS%|Le#aH^fsnx?!+D5tm{N7ahy4LPdO_+(NlIUOQu6*% zZun>;NTrIF>`s)ijaX!(kdPSNX+S6{uwXvnxKyH_5tjPsd!Y0rF;)NgIF+Eq+j<*o zSwu-Z^cay)5rg^8GIApN9Tya4R6-CTSY(tVVUzKp&u!iVaiK9*yB=_Yh?JHTCPBMF zdik{w{RBTrcBPhMsUYc9vG~f{z@1blKq++9Eubd*-`fFMNE;v;v;gTm<{fvn)wJ^0 zz?p%p0Z|KkzrU|8BJlEwLyMO|mP<>saAdd$9AL>HwH7s$dc;4%-KZCUkpf&M_&?1S zXu6~B6l8QIB@7AtUIt9@jz3_AdE6J$XD()M~|^+c&gxdnVI zFCG_hd=*4k&MkEUtFwyTA{Jox6}t3Qux6%{1DiB6aDGLjJdWV0U__TnOEfN$3`uAW z5Q9;?D#oajS_2Zj<=|TzNaD*$obJ%ZKZD^r&jYve1GdYtjXl!pT@Ogw3OW~*g)D#- z{26FxU^;qD3>e^b!N+0RVuE7DB3<0}@LO%@BSaUnhA)LUn(I^)!SaVTIgPYW_}xBu zlUO;&n2>@c$qXii=2aw_!8V<$8iX{IAtw}AXTneBK{pWos5;*c-oz><3hFkO8{vq9 zvUPSMi8S!wUaZ!)iRXiH%t@|)4?{?HFbeeO#nG^#sgU!{wfTw$>K7Pa(z??!MqYI8 z880vwY>`MYzBl5@h@4fQ0xmo~~oPT`h+p*>KNiRB&jft|)ONZ18}{un(BT9qJ) z&8c+Ofq9FC9FY5AOGfPr@kk-jCQSz~RYUjS7qAAr?8o5FO#EdFs+3-OqWD&zs4eHZ z4V*pHg^7@5NtUmtgSql*&c$4Kz1Qp)cdTpHYx8CCOrXF6R}IbEbXj!gBya<_x%PxkFDDZl!>_VZL5y+Sfdng7)N=_9eY|YY1<~~#t zIanSi9+cZG4Mf|KHYZ_}lI-y4mnoiZovS5{9KX7{KUgqUONjt4D^F0gEv;mU1 zKy|CfV{fzAaqMpCM<~ABLx;=Z-qDkM50tkNlR&@Yo`-xwB0`_N&%|hcu58lM|dME2u^fBbvP4Hh`aqTN#jZRc>lIK0y@ zK|{-#K#Ck%@$RU>bJqRP=c@1>gpwdeen-5o!0Q+iM&p#WMcI#*4zoH`W~C}*2HFHr z88yF&F(gvObw@`WknYzPm9Qy{JQDTo+t)DNwdF<}1Wtlr(-jYP6j8kDY`sG@wiRbO z;+zit_JnCb+{&`~SpBk&y7uQit#wa(To!HW@c7)#YSQd9zAHRXL!d$sJDs+SBE=j9 zB5&6ET66geCy&v^GJ7~&Uj&+IJGA@H^F6MaK0eI%(Ma?6S*#w4ilIqjG7F3I$Lvds z;RS;n0?EC)K(!p(y-cp0hn`RB*hf9uV$1xaQxqp+L?_m`okK0IkW!{?tlnBVXshW! zj%r%VoCkusG-k5nRtAeR_SapUi^0^_LM0;$@YX#=e0ih-ZEKf7l=NEd>E!**b29)( zK)Am$ZbbFCq=9Cz7Q*0Q-PQML1q*d1kgwKhJGH_LxY^9UP6L#=n?@Q;P%d}0x0}zs zBsYP7>Epg%MUXA6K+rBN(t<7;eG)spdxS|aYpe!yFp|#_zhTwuG5^30PC`VZ-F}yW z5F?E<=zzapZFic|#yDPBiu%le*GDLe<8`N~x(I#lXf|ePi>#wS87j{9sinY53zP$v`EwKCAnAt|AbW;x) znl=zib^yzCdkVYo!JSr*)H_>FnC?v^xV~5E>c8Nbw0?ye?sndOhcPO>HIh8>XM@6X zM&exd39Kd1Li&i&dErwU;U!HZDl*i18x-MdZS25uhr@7q6d8370`zi&V+&z@-BcBf zD}dvj<(juhxfD!g+-5_A$#A$rbt2bp-usTH?y3iGSEkw0C*C%Wdu%XkP37?p^)a!W zWjafrd}F>@iedf@Zyc55x~_j2`=s8sQWgLH;U3eR?LU1cG-p+0Ji_urtR`-mEav&q z9sNvWJh+iOA^5-sd$4Q8l)9^np#_f-Wb{4HvHogsFeUk=0rA{Z! zINFZ9-I{NesDb1>Y7}bss5gtUcWQkF zx9a#x4qBoVQaU%5ba#Z7-nKRdMX&ZAQBm^IPq-HIcOANSe?eosGKNlh8g1mBPG~{v z{TWhI;*`S)Ju<`Oy?b4ALa;hwWU!Yj3XXSCQ zpYJ{6?S1E@nkySOW+t~X;wA~I!}7*8x9oNla^l2mCq=o19$H+E8EJ!)bdz*X&X|T% zU1G#0 zKJsrZf*>=-0NDkZ$Wha;QT7)Lp!$ok&mtMrZnbkH5s@w9S*1Y6<~%l8R5U~#tdU}s zX#l3K-@5LOwwmUzh}jBSNINEG1Pux{U<84yqG*(TTb0+6cJzR!7$r!xUN|j|WFibO zBj$;&A50sx>Md<4wc1nHEjAoO_e#3rM|SOa#m0N!D~ zQFwo}(z7DC$6FZ@=#Jf6EZSPE_f>dpbr}r_r(XE!jkgB3=0(D1ePM`&dlX%631hbH zZ9Uy`-Cnc7@HZ@)shI4QO+T1qe#`We#N*!1aa=I&92&B&QHn;N*ZLxb_B#X7(2N3W zcO3oL{ijbD6Czpv>ne!+;Cli@T3h4UJCynFDgTi|h!%W}Noi|Q;#~pw}YsDCw@W93EO>DMvUWGm4 z-ih6AbzY}aA)FL6u-8=+w2eEnF_=eb_L7d}OkS?sv8VLxTH1OepBX{WV_cF4t`V=a zB)P>3o|@g0#BsNoza%1Yuz#38V{7`D%#Fk62udWf5%C)@3^6aSU1=Gccsxoe0xOq; zVPM5qNWdUMAy%a$j3bA&hkZuLM$B^-;a&$J@|n_aOJgKye=W$i?uOaWJ<_n8W?L|7 zV`cyIcJWk-EzrQQgzxR7X)UhzV)35_V3YI2(xmJ?!{wo`3(UEqY!+N=!94-;UY%4S z^`QxBH*|238rK_mf!f)D;p|^o&xF--cUKZM-KS_tA~gu)CY?%xLs&Bw>uwM#y8H*n zaxrX(K6cffJw+d8eeB#mgIFE{YcDx8KI~X5C2cQ6GcAzaKZv$bum-@!- z0HSiZI7*U7kYAy}P5MzON}(0NN@#vtgpyoo!4t}e{Ak3b)>_G%NnO>{K>C6=qC4G^ z#*ny+#RIg^PYWqfrklr<@VW1 z7Z)c+eF>do{1erke5<7ErZMKqmeKKH19YT1D%308Rev%lM7ld#+X_u0ik)9li%jax zijlt(*yHjRo{d2!(7q<)V96XN)h=>=R1Wd>cz9pP0`lf76aMHm64(A<{gi%{FaDUl z?MhG+=x<F$-1N@=c5fg)BQOnZj5i;-bK(2TOliBm;j^iX2ELD1>-MnW3tKAu9dIfi(0kdKQk9BqQ#qd{iFvVW<|{E`nu= zGZA2H?Ig;<(K~HO=pqkDV5qjFf^=ogd% z)8dLfu)B#((9t_PoS_;|+jdIhACGZ+Yn_-^9}tvYb4~-&a~1_SUyExDhTaT8Ij)t9 z!^G122h$0onG{Fly>*l|wpZ@A1brqvbk4>X(|e_~q|U}IGZ5b$WdqZoNCG1&R8(Xi za!hPj8#rN$*OJ2!2tz5Lgs}o=#GEVXxNMd2k-Z*V!9owpZyT}e`wZYgG)d(Y%TUQl z7~|9RIxKlPp)<+SQcKe$uWq!iQpPEI#r15zWkg)1c`BiQmPkNEcrh>pj?}Cw=rgwPnK`x3EG!Z^7ceMz8ud(m zzYRCwZ&WX`%0;Dv!5UWD@AH%0N|GL0nCxl`GUDYBAyu-o6%CJ9H1IIA0x@DXKvXZ=;0p2T4^VZf9uwAFoXI;?+OCFutQWV?l2RVkg00lrOXj8rb0=v>54PZM zv{rP*eKTl-d}zVhDT3O?DOd1l6`Oh*F>$*4#PTWzw2wRkx}W0KIMmy*LJ$JtWO^Y z_P^?qb)+99_F7MSm8%j4mSC#|ZS)NfN1<#rtqJ>>y)L^ew9JcNh14nUpKn(EcZhmF z$JMF96teht;7`i}o!lgFpl?+v2PQFNt{0ZDj>IHy<~T+Z)li4%B?}P8bRCVNCY#>W zJw!Kbhp~H}rVrXA?vKALk`GXl35&>Cj|)FJ_q$=M0N6zE{wdjIVBHI}IZ7lZJ{A-G z)mpW4R}@TN4?b$lSp3(7phqat^9ZUNAIPMasg~vn-|%H<=DC|868Y;LEM)=KEd*s+ zv1Yv|W0*+9{>%=7(1627s9UYBRKe2kaKLzbZ|@|C2`+RHyA8K!joQZG0DwM-L`u+B z(h|ByP2A^S49xHj3Ev8M3(70Fh{o!#j*=lsw;KPlIWM%dTaO)(h6v$c1cP%Glx%BC zWjP3Df|Y5oS3aSOiwZ@c`Jl*Xf@*%Atg=N&P_<`dtnB|4*;gX5w7F17NqJ2Gz6mv} zJ-a>t?dh4c)z)0=bSgRRfSchRF~!l)vyA-fLb8c07MC1f;{%2_gxF3r3al{;;q6%2Y^Q5JkvQXV#k~4KXkdoe ztPn6B|LgyH`@{FnwUM)+V4p+Kz>Us|s0`Qd+UiE&_C23X;}&4A=wb8XIG_k2p;Nv! zK{RvNdvF%45m?>!T3}BUc9@rtw) ztq~bD>#N@FvDOHEenDp&;c}y2iqDFi}F5+FT7Ak~{ z6I0_NCEfA%AM8pQk?tqkj4C9JZ{~*sVhbY{|CvTi+e;pi#8^De%h=F*8k$Sy$%zX6 zG0_1Ovm=(s^qUiZ3`ki#nOOp$02L(o6w@+B{!&a+)D+S7P94fbdzSA8dz_! zI@^e;exFN~9qDlvW7TYos4;G;g+upLo_fu*XKAhwCT*L{KerpduKN8)x7+Au9wCn# z>MDzTIOCS%*kwzfis_8qLmP2XuIYbo0;bZ<{{`tEl%NXJ9(X_aqIr#D%m%Qh@6EhN z*Iu`u`JJre*G{vSb05;G>;@s_K|!w2e2VfNHaVErW1@kAmXix}>>R6MaILAvA1IZA z;mI_zYI(MaKPEGdXsQ;b)y=MwRe;%@jkDw03iUXUM+_ys!blLonqpEf2Eo=#Z>__R zM*@$#!z#f=)psP2Zu2G4QbZC3vX6pb;f&n2eg>b`n; zz(FK1HYHauk694HQS>he63H*0nj~Y3jT2-VS9ZQZ4NzBBefMUM!!+z>?X0rfuFi_m zP3)d=)L&|}c!JJ$7g<>dh{Gdu2J!>Eq~R_P`NXIO_P(WVwo0Zq>1o|B_p+mTnkjZy zjR6$(M)b5=JMPgXSqrEgSnkzyY_1=y-=v-~K^fYqNNrT*<#ie&X$0;EMGn3b*ei1c zJ5#nM`ytw@^gwjc7*t7Pp+_3`vWP-MvP~!8B`#qD5!YZp^qHev`I_&S?#b-e@xWuv#{SfN?43 zh~lt`iNKL^)b-nIZ8yv5Ss|!bLy1u%?9|I@#)0S-eK=;f!Zfp5YbWwPA(!BhZTM_j zPPHw&$utd*J90(n*WREZI-h%QLuiE&qUrmLiaq{BtG~tD<($lm+Sm8e+E1TB0Hb|v3Ef7l+vMzH-IfLd*$9n)B_KMbm zN@p(@?WtTfd+SsTf}v#JJnc#EQUpdGAz}0GAd=DhLTg2Pfy$kmQAoGGL}J1Vm!!3s z0q&LjDPmp9?;__q6;hzMD*=+msbJkErL@LEY0grJmjFu#0iQ7JX_7`rKhZa`m1ZyZY6|Eg79nV5}AvzMiil~aSUw$}br$$Y$FAG(C$ zcyM55q>N{BZnf?782R9odo+ZDE&My~ob4|UIQn6x_COr(trwx4={*djEVbC>viSr5 z)?1Xc-5Dj*ry&ZgEy~g!dhIr?Sf>HuTk9!n)L^N4*9?^CAct^ZhY6URb!k_3Om#^T zm2^^*g*icEc!PLqvH=7>0lYF_UDgvF9spsGuWI&cK5U2Os|;V3h1IsEO41Y49VH}b zBCAnuFDw{Qt*|N~H zFwBQ#!1|u0b&AM~+a!F$puvr9LO%&qr;R`YAdyqNl^zbo!qMHt_Nk;tb6n`l(Ottg z7XsM}5qD4i2}?Q~{*nrc(A{~zrM%J2@nhw2rBZHr;erOJd45){WoHnl2XP|Wb{Z&N zQ#Uw(oSI+|tbWHBG-DPnQr0w6Y}85k6P6>VlZCu2 zVm&nXc_1j{Z6Bz{2n3bZv9#>t`t zjT9LQV)o;uSM#})yps%$kPcoZJQNmPOds8vt`@Ie+8=o&5~sw*QNZqX)}XJ@Mw`ij z-TpEfY!r?LaD>`gVzR^DU{c0K1dnY4buPVl>U?F&Jp#F8w%rW$#*TgTP zmW06;aWaLD6&9+brlIhFXYk-bHxycPc*mZkVH`Y)3Zmf+hGzn}(+TiK+7Y2D5Rox` zZ*D7E2%ku)`9d<`<;QK16`# z7BNtoygbu($6|s>4M*G+URf*790DMXY!&@dOCdIc`z|?=ves&g1Rxk}$H6e_x-4d& z#1s(&#+{4E#uoxg+C@|A9TvW*|Hq!8#Bat?vn6K|-qbYu-3Lu8Nx{Bv#V5^}wiR+- z+K)5HFjop257)_uSFt1@5@^wmPU_mzFLXM!l(68BU`*J zE5irMXvk4oqZ=QXE?T|IN(yJM$UtP%ANW1moCRSR`6u^n=E|WK~I? z+vY)(EDE{0^I7&Ku^rDI0PY9E?VCrtk=qdn}rpXG zb594R2zP40_LRB_rB4MW;HG3MpO}V;OT_VtcS3YHMMVDSCb;`CVbeB*2kZDr5Kd&D z1=tNA#IkN6trXqEcMBCHqwE`Hw~pHNOzSS8x^of+3;Trr;&@Z#Q4M{_qbw}E`z?%1 zWMH|K&aD|imu_@DN=MZ&)YWBz_vf1Sq7Z0VYPle(s@0tq)n}F{p3@lo5&(n9`;T4y z2z}#Wx&Q-e&Hs{p6@tfVC-++6YMUS0IAlTayFZ-~jFbHz89~6qg()9MfF2W1#64WB??84y3kd zQccV0VL2KdA3QlO%PQM@^5ppNaKHsk>F)=_;jn)Xutw*5$4nZX?;SDG`yl%wipz$1+W<{QhjGEeYrb1H=&QxG}kf_)B*GTtT%>!5#p(3 z>l6^y$Jw9*MotkBhl)7+cezPKoG*cG0Jgqa8TvKk*>~`zqmJ=n9>RuJgR8$D6xZTs z^!qGJoNhI7t>TdRi-4nke}4UBv_E=M zs4Nz&HXZNtA0OTE&Er65?=-u7DT59Ley)3E*HX-0D<#r6;^AbML~I?=t&)C|1-V25Ji-n#HLp{@CYOsuBe%@ zg3RDq(n*$j$hrER)*6Q=FmB<%Lm7?6FMjy`-OK;`oeU%S`!TC^&RvQv3yaYOoZg_E zHsdExp6nIz3DyqWg&lag_d_pzEWl);9J#r=jdF*!Y9?9ySG?1#P0RWUo`_s^dyG3lG?cxH{ zDe`uc#Wa@Cxoq4OZLYpr>ZH3K|2nj5vGh1y!1vZ!j@#{l08>XOPnJSIcS4qih_@DZ z&W?t~xTr2U58Q>GC5n5z-Ye&HZ1$w%Is}C<&zYZGk}Ti;5h)>ZwC+g24^y}V&6g0 zYjA8FkkT?b20a9B>rbol4oZ|zVS0>R#^xS z(4NuZ(9a!u z{O+#P<2B&z-2+ZrLmjZi05xEzoF0DZa--u%){6RiI`#5#3sfTUR*wx}Qb7qCh+rQm&e- z&BqnX_AjY(ouXt<0rgf;%}4D+NOPU?;|%&p&e~SzAT989bvTA66Onk+@5Qt<8#1(9 z9O8Xa*I9X{)N-BpZNw-Mx*zgW zB}OJ?7E{{0j#RW_o;;geu0l{p>oBc~3$udA9eBQAm z%IP&N48v(+cU{%0x9=Gg)@-mSZ`klkJCtn&X87LCRH6YGu%*P>+Ui5cs}j0e zDoTqDeeSF#C;TVbo#!6tBNy?8!;U+#&6oyldZq=Qn!hciIyDaWYA7?$-o8B>|MlG& ztqcF)r>45Bex{Xqd-mql*L2wb{O;Xr`i{gUX@>JWG>?ViqT7LaDfCm8S^OIe6=0Tvo+NpSE7@>RQac*e?uu^p;6nDQYI`Q>rJ#54-3>pO8qEZZCH zAC2ih$NZnCAr`JIsVw`*FYCt2KiHLG@rkPfy}&k0Area=YPhz|{bmVR+6f5RITmiK z+#kSG2;5o*P z1{e39h^*PF5r!{vYCY<~$_Y$3KA3)e2}ezUL==wbLURovDv~gvM16U%L9N*xCo(*Z zE&d)&vH9Mp)#W`nsZQb~y9iBQk|l(mNs@5v-8iSyMkW=vwCKa*#oaJ7V=FfP&B_AQ zmWOy?Y}^uR1MbzTlF#Gi_b*Pr;+=uYp-PweoEKL4V?H!iYDfFa=x_(4+hgtO#e=&N zdnX{gXppj$&KAh&1lRkEnq5oaKLpiX2Z~*K-(<_a3vu71meyGTJ>l_dqV!^N#asp| z*at<3Ow%AA8uC7|J{q4)&PidW*m)ZISVkgiBF#7lh0`ej($;nn2W}`J$h{4A-*|R3 zgoZVf*rGOc_3l0iAyZ8B?@>6VOSnNxcEt42G-c&!b{AxhqNx?q)F%_5$=jNI z{T4nU3Kk56C%TAh-_d>iL$#!xe(j()G!_w-$-rc*1V5GvM9B!P$7TU1T2bPI;rj?6 z#}@>e${!!~$YzT`w4=PUpz4kJX|o5j){zk|co33H6hihM!QpRQxdp2Yi6)#oYF3?6u zWnavypzI!sCk_Z$2D?@AP$m@`l^HYJ@P-1#GX0}50zx9qPeQHxxy@M z{GCgAx#Ftg3UfcJi5}_EI>dTO76iLf@qng~M(G&Se41H+?d}3-g%OADA}FGs1=hRL z-RP~PT$Dn?2~mh%F%$kw4p#Mswj-06lI92GypgGyb=Y)PG3h5?V+o4i67>?h#3V8vse?7iF+1I?!{36Q19WUi3Tq(8uj-X>itFPT->&hiP|k0jE8&Q`qKK ziTO6Q?p^&YtndICUS8mMm)+5dG$Qy=PAtJhD%8j#3?^bAW4H^^4>8pX-Agxw12{R*b2?? z=7;_}T+2TTSc~j??Qi5EF-;SCZFmHAH_7xh%IkI#C`kFFuy$<|qdQhU;zmEgV#N)4 zBOcPoGU`bP4M!Vi%@&RGSYw<{M=VB3#Fb$dNngAK=9Sb=xL}mDw&yY>D1~rQGC!b_ zkdunAvV{jcMdz#P({mpss|P~~@teo~FfIh)4MFBe2}7LhKr99gkhC;&QWCgDa-bVu zTN7&>O6$-Czb~c@-A~y>CR$0gV4@6<)G3L+26I*QV7`S55|V63=lV5ILW ziSnX_KoB5y1&m{%31RYOW7^lY5)ZcTzu@~6M%`23hL5epGot84MSFxySKxT^)yZ^raimLg}yB*J5|Y2&NZ=k`U(DB;K%43!cGv$RDh(*{NvwqI!v!)Oh>h*osQhh#7eeI*;M z%hN)vK$GG0-$_G*3)Xa+tmbsvxhU0ht%slL>AB)J-&h_+o4mFP5VzD}Fc`F- zJ$JO_wD65^>ij)~DD@4z+L6UCw+zvfAb?Q8E3hU(iov>8`eA5&p`^}e$raKXL}+&h z5wJ{Zp2Q#m>_oA*3vrjkVvv5D%+f*%>Y6n6Sj0+c&*WnB5B2&!mKN*VAVoGR==ueu zCP}sS3VkjG4;yvFgLlQcfDYe>YOCp193V?70sEA5S}b8a4tLTx^rEC3m$0OpSM~+M z6mPWUsYh$J#U}>MIs-&5^ven*=9($dOt;02tSeQo6_v~pHh?q@`h_jAtY|&0uKg7y z>}-8UHSAeh!Y(8Sp^}!XWj#R}6@{ZBo5NOGpsZ2lxRI40s{_bz#QgO;n@6D_twY@@ zf{`&QEN&EAFLktB7pB`kK=`~Y6&i$-mBA!+qto!40Hg`Ql5sj1Rf(h)QS;02?%U8I zho{cUaF^3wqvwPGacq$-REk5M+>{MsWK-FUgW;7>Da75%lvRNxF;}rT2TQE0cCkd# zj*2)kjfWyo30$pcp1Y(}Eum1v>G8-Jvp{^BgHd9bpCVs`^C%V1+1&-tzEGYWd(Xj6 zz~@jXrpV8#Y)Avp)K)(}vOmXmWnotqLasn(2jv1oi}dsH=M=GKpBVDhuImUk#GF#4 zKz|zUxuN)?nCv6m8t6{=Iwr$PR~O7YAtXp?Peg&Y_&o+gU>LVsBF6bByw%965p@dO z^ioF_`GBZX8NeK(m7>DzI0VHho89H17A`h<ha+5`(De?~3w9*SKQw1G# z<{3_ZTd(p#eE0qL56(lKgr3(rQwiM;Ti_CqN=@6m*W^x9Gzc&R1}y{&-@o`(DXx>G z2bnEe*duqSb$$Affcer?G9oNYncyD{&o@HHk=9{z7+)sJg5tRzgXw-jQM*l2s zsvE}kjBXshh=;dDjFxM8Z^Jbw!jP`|mprLrUC&ouDOuZVBvm_SU%mMS#Z*31-`=5N>jjHBtAB={Y!G{(f|;Ci zE_)Z-hO9Hi-1qRD&d0#$Odaq!iE#v=B@~NNPm;`gN5V_gs!)XmAKHp35C10Gt$vI^ z>tNDTN~Xa(1sBSI+uA*;XU_{J#33?bFD<7CEDnZoKS3k-;RJ)iyJG$zY2}%gV{i3n zK?W_N2*9efA(sh%ir%?iglOw7w6ZUx!Bg%BFG+S1bp1^L>V)5x+(1<|M^aUq5D>rj z`0bC61@92bmJKd1s^82n%AB1VbcQg?=EF*pY~3?&ZF!O?!@N z5ty{#Q8w@20x=Qk_$v`67RxuMKe(lCz3nK+t+bxi@wkrc0fc-=a~(Jg0+=fs#C7U4 zt-pcE;3VC3J2Ty5)=(pQ)W!YZ8_&bDCx`p9z0v5fJU%*{Ow&Dh{~cs2-hT~tN?TGBFWDoWna@7ITV{nDyNLb#p}#=r;q$@pVOJ*ucOT%v4~zB5gMZ}tc>2X- z%+k>lo~5HbnI$k(zhLh}9L(&$=+0r&@&EnNiuK-dr+NC??vBIrHd z=$~?r;b;E(@^$d4Ua)wMe*ZV-p`vrWoIDcdz&v~ZBey>!`=A|tUC%2H9Y2QMOuYc0F5ekp~@BNR7j(+W5zOgu|?5l$VnyeuS(fk?D84rdF-gw#` zVW?oy|2O`dr!P#PhJo=tymN&tp#|F*1cOY_~-#C`+Gy2D}o)0qGB@dB) z2ngLL{_Fyke^2&~j?0R4@Ws*I{_{!o5c%h~xK9B3McAqPkbwJr3HaIPE&U5ZLkc5$(q$6|>OJ5F_`9HQrcC;Rf`bIvt< zelANl6v}Hj%&pd1$eMY}nPLM$epUuko94Le^5c(9W zOe5#CM+S@jGY+$_laKz%+Iu`+GDS>x!i~#1`ZN1l+jX#wz@(0?M}MXbC+w6CM@t9lBk$O^$H(N zx}PmgFnVD)wzxz3A5E0~;2*O1;!?w$6`+6q*q#izjAKoI{*xu!XUaRv@|JzqmrawM{0XyLHKVuqj9)gl zp^W|~59BdRnef{t-CDhyO-Ax3w($#3lF4n8ZhrFEt!MFQ>sfTufB6%?1(r6vmwz3- zosdp5$#0(dzjavtx16l+e&u+cg8TMK^l8CP~rF z9({gwd(qU>s7gphTlTMTAT+T{f8bx>E=c;ZXkO5u65IiuZ}9onsTsvGbNky^d_Mkt z*P{P<5cB_iYhhAq1i|)`CJTT0YREYQ=_UC3$dvilX_SipoX_shJ<;>a=FeW=Gz)s} zY`=f{b)F@E|JT>tHT@*}JiljRSRkY!&Y&TM=9%g+6wi+5c%skjpd1K-^#Wq+HH#l7#8va*18iP_KosaimF z$GloT3d4My{riqx5%dYxGy0;x)Gwkwj^VEFPqy#*<7hNeKa*Tcb~K+o&7U9sH<=ga zdFg42O!#5%X?JSmU;J2C+irNgQ7x{Glwfi_&K}vryVD$P8w^TDq7F&DR*OwR&8-iBF8&|DOBBQ zK>ep1PP$CN^a`&|U-OI1L7*tQ;3LUf3249 z$sG^)hT~L!6*f)&!9S%{Uh9qXcBqWsKErL#(D{>@5~CO1gvc%mDmYyA&OQs5C2m%@ z@>TS4xHxgZ!o~kzdr!9Awvpwt`zv71aSs{5kVx(DXwFf*B<@aPkCpU1C^`&8LNbrV zB_L&4t?1vU?ounNKtM_->5TIbi-oF-s=}?i)m@H_W?Cc=OEopU18|-;o1Z4dnP>CJ zsAH>{-u({PIz46mG1H0yei!k7S$so4biX9>B9IOY26lUgpA+gUATRw;bf@{^gnGOy z&QG;5C9st+KQJ>76N7!jde|E64xnA02RA;BJ5drAAp? zbUhBMfzTi22^o!(MDj1540V0Wjjzi2^f}ME3r7qSvVEB^D^uiB?zp;R3kFRR&@+qd zK}qz@WM;L(?<9IYzr32x%yw!)+Hfa#!<@LaAF|q|S}dZ6*ah4e2lKG)Xhpc^PEWo< zeE_pEgg;VfMJ`cd~jS-YY9-iAf zs0A^r*%ZR$THxT>@+6Qiuh0jRI}_jYgPhr&CWBQFEhwNF<1iWX1sB!omfsqD zpj**FrH!{IgvL-mC6tRox&IKQD!Q8Rl2MQ?(S%2`<$58-D?JLq@Z?m>6tA#itsw)F zRxQbFwdV;|Kjt(K0B$YKK$Ab)sK>i5|02IQ=?b9z&4qr!IG= zW8FzF=AB*$tu-Ahg7nz{ClR&?PzBK-g$ROw&itEC!ntEEzo(e3?xegOg^ z+OvIX9noRGdKFNf)@xlDLM-n9C8vqrD#k2m-DG+-1yj2?WM9x6)N3t)27eNd|1>Jf z1wQ}5Oh2PeWQ3LiuT=fq@Ssg!u%828TcB!fVzeVrRc1gdr_u(f`wvmun!gr_-AWZSt!02Jzg2MiFf!Bmq*jH$;H*&WTjuC+pbDU z)m=qj0C@1A{9d6BTTmD+nBWPT3(U4+UFkuGf%BMeNSGM|xqhE7En{2ABG<8E!Nu3@ zT+JXf7aYYpG{Z-$^c5u65|qg@$mDXl%vJ^ceexe4(TEGLf;$S98XmLn+0ww>fQi&+ zkY~2ibQ)T&;2PD^_8|dg);#SQFrC*l-Tl@-EQ{sWDAl}>LJfW7V=1bVyy1RnLFyrm zH^_yG6-XOEcLlG#hMQ^0CxByhffUX`B<+CD7jRhdrwr5P1i3%Xew=M=fVU4g^}Wq+ z@U*h8V4*Tf!^47$HwbN;%5mSQ(mH_hKhAK$Hl(hY`yD^|VTQ<3jJdG`vzDuYAS{rx zi)+}iW8ErO(L-*BgkALe>$#9!jx7RzOC%EV-D@5n(mq#xEN79Wn zP2rDDz(&JosVmy0WCN)IOGiFL2eWPu4FYsp?CIDBTJN0xYOUoJ87&dvZt*HpY%Zf` z1v7zd<)5&mWd?PMp*+-=;w+zI(_QnWRcP#?qOn%H{09&YZyp_Ns-;6cW|2e&lL(W% zR?yLUpE19GC57xoK+_i(&T@GZokMn=W^@qkJKu)Lw3U`<( zhyLUj_TV}ia_Uj)Bb@}zn_MZ4j&PS%;c)gKcP8qF8)KZ{GtPZo7E}v{iIGqVDH(w* zEJ~rw&p6_FX+>#6cr?2I%LC16YA6yJwYfPDspO^uj&KAyv!wWLVzcp}+q+YEkn$G= zdhoj;frja9Gv-x*DDetjWYI(MA~n7bZ8=BCgJ&r#nP?$c+)&|do{ap;PXq#4OE6iZO431*Y8V0{{U*$uArv72-i{DFS6qr!((A7w z=7!aE41J|Mq*U>bYrK~nY={9?XOOs;v72OS&Ly^>{jG%bW0Iid@(PnssL1}C4N}7S zYe1~H`Z7>q_NtW8f~aXKs{7fdyAQYe%>TCcHIhDb4CQBYx4icu` zkQJ{nQ*0C3k(h}ND5#l|F+Av88a%pN8f2R%|Jr~hB!RB^`rWO$X<@ySwAFS_WMM&C8{0`lDZIe;b{(@gGyCe=B(EKl&); z07-kjJmQG4qnXn7;GB5Wc#z@XD6?q9zIbe-UA!*=LzRv8~C)(Txl%jtulv1^5NWgkci3BG96bDns>kIt) zVKxyv!uk=4ZMY4LYGfz^i;o1O2ZQI%PJiIqYFM~4(*gj13qcTthB9Ua<-ZPLY$pl9 zCM(X<^|+$BIeadQn?r*o7Ie-mNdOVz0jNgAcRkX{+>k2IKElInI)cSwE{^&2jPMOE zWpzeHaK3<6PMAV@QurACk#~lUnRBwRufvi%s8`BtK7w<~L*_kWRXlM|*(!YrEG2-3 z?g^`bN3^nzElGyxV?+8Gvz>9icXqJ1o2ENO(a*QHYT1r4+RJ8)=STyH1ne78Q_PGF zzk5{bs_WIYL@5)MOoKt?A(@wrW!Ny3qyab+MNB~Im+3tqp@O_{o3$ll8Q!WG#^y)fSOO-O>)EAZm&o8!l=jE(!z{ zx!8ja8yXun6f$fOGSJr-U!7$LLrY_A3>7h34covpgsg;Z7&_R7(!U_@jn=x}PgVgv zi%XAa{b~ayhvC4h4XIphpkf93l+>g$zXIhdPH9D3%1kZk8F`Kg4N2=m2*NxB)dLVW zME+S(j6G%pCR^VX*RD7tNJ4@8#l#B>0~J=`Y$QD#jswZ=v^mNHuj>@gK-1J~tF33+ zt6Y^u=SbLqQmSAj-C*0jpx-r5nLq3f_WD92Nz?vL(eDpNHh-uaPaLAI2SwlQQuy7j z5hp76i2oiuFE5r#kl^@LX0TFo0y%sP0j?;6|*nW<)yr}=K=6cHgSK>00`t2 z`|xjTU5vx3kE&%%an2aEyR&IpZSV6n?iCF}MtUO;rw-V|dN7nDVw#VO=$}P7H;E?@ z(`Bcp{rORy#CsXc{*ns8#LW!6mG9Cho0%A^F3$j$UUB}pFK=Fo4?^zQd#RVb??QFa&&cj zgXmyALdP>m+Xrd^$S41pa4MT+rZ1|b@f*Cj?6|7Tdz*z$VxN!4?|VEdd?Tx{4Z9a} zhpfsl{se~&O6JUj_`c#tydBl%Kliy5ybGq4L_@(YApkx7HZPa&^UH3TYIH_USEl3* zcl`CyU}raye)V7H#PZAZOO!6vV-hy;F);hx-92FTySoQAvp?d@u70M!RGD8>NfSU4 zm?|$rb+*YwM?4*+0wJZEN)tJkf>0me6Mh#{$O&IHX=`sE2?V^YqX#Kk+&$ zr;s6JYSXI+k*kILlfFY1SnfjLp->VGv#RXx-L{019m_tLB~iKSzQoE9;!GIhj>%-A zUtfLJ6$|gl$O;8nPzDo^vgk~-u+>(3HSjYwz$D;ex~e=~U|!rnvq{BQrTM#l;|TA% zr6BA*aAaDI-e;iJgU2fVTC?vM++flG|Vzg)cs zV>Z4`v~=!H@Y9~6@BL>I_?ZNLCV`*j_J0$({m(;tBM`n<;{x+nI65af65_KSiNHT6 z5#EDpTWHGjq#!&Xy-=HaHoh$At_LGv|2Ch@x+K!SL#KBn(b`EKJu7;>XHjA0Ix17W z31H%b$p>x1>H<%m@C_`?SFsL<`^kK%+q>KSJWtcD?a{cO7ePC54Wx?2xTY^CFk6tb zE#BhHZFD!&rEFMKLSN$;e+JKwbT*QGBJ7&&)eNb;Y;N5zoBC@w(o@ zea3`~$dEr8B~^zTo!`!3d{Sm+pQtm68dCLdkScS6LDg0HH(=Jly=40!ip4)W1LX-(?JC~o%XIj+<2T>w zBA4jPyNo#QNMD0fl{z^Op$cfYPaj!YL1h1@vK^}bIPD?jZ6AK5QJtRo;Pw7#iM%COku*Yk^-dh@m8q z3#|~dyg_(1N?;gXAqNc&@j!adIxn#7uZ<^+gUoeXcG{+3$eL?{@oS`Oz_7YHRQV)K zh#nQT0iK9KesEf!t)lJ8DH^-q<<}gUt&YALO6`L+emqZF8t$&=4>09e|de)D$2ISMo{*;e;@#7LFHzUa%M7o3qQw z1y_mG2;{{ee}d%I^KeQ?;WJ{boz9#G4JXDZ1ZSANJZv;TAB_4$2*fZFHOH!-)gS>QZmCzS zRWRa~3P!!A5x3Md!YW7HQswaL9&tYFel_aZB~4sh$(J)N{hh ziUX{P!qY_aiCbzuAr&ZYsRFIhh2oaFPy?kXZmASC*N)AIHPDyhmikgd z#VKy7IJMHC;+7gzttu6_RHYi|RB=n4YOQh=w^XibwXC?MmeoMrid(8%jr6X#rQQ`# zOsWc)F@8(Qg*ZIsElr7~HsRmLr~ z%0}vC+)}-4t7pb7^~@#;C%|;Q9omWZJbG9vX9-70Yo*XO&}`$Dnr*#`8@E((o9eo8 zOI^2K>5W?|y|vmeWf|Gnh`yQUW4Ld-6#sJN<9sG`q7lX+FY!`UT^5#Cvy|iB_Mi-G z+~3+v`nuEy1M6amdsP)fv{>0yeX(v>(oTl4`GeiCiyiM4j&{S&)quB9xSN7)M!W^% z-OS!PzZ*IgJcG5v_{*X6Xn2dT;y}SfRy*SjvoP;JQ&Q#O`+!7C|(EZ&< z`SLxaH^AiJogP;*+$8kh-}RvX(!u|eQ}+ZsDo-Y->rR2uY5hqs zIyXNJ&KuUBi0Xi~r-FyHI~g>n_H>vFu0J70&wHN|YeJ1qiX74Iw9u%BpBR7=J~Vf_ zj#5cUo)M`Jc9H|gd9zIwz6qy~E-uRA0y7Rnw5-77o+*gtvm3qk(YEPV7=RVL0-o}1SG_6IE%hg3H?^LPx zyhh%qC2#TEDl`{CPEnhV5PpYR078_;75 zx~GDKU&^^vrRlhq!aMo>jk9Toejuq(djW%~Tx6*X9y`S&z z@3oB;aB^L?7=SrWu!g$$|6t!rOem$7*URG-cu8Z5OHgRSWyL^Y2jdN_=QItIHPKNnOs*^+)ISpewqZzmXr4r&H^AZh>`11~Lpn zMtD^#s+tr~XCu?80Ani^p!&pa{=cu>p}OjrlYyQ!W)Lo>iz|Vm7|(%H!MBhWf3zmk zyzlU!WC6+iWWmgb%gblMD&{q(1;dLO; zUav5{$QjtDi%UqQH@dtUi}|Q%7lEl)ARjC!gq566mI;E>x9`hD7M`KHQj{zUQ)~Ps z*2rWpLg7^Wh{S&54wP-uD6P;Hft--jLlHRb?NbFqemvYo>M3jyS6$581ErZ%dQm%c z!kR%cEYH3;b5xZ>oq%;X#ioukKvwJJ&wq$TQm{qJVeSYdgtF`oj=&J0p{y|D_Wyn| zE#ScGH>{|o(h)u?)&f%QX#)$`-7K=4&M#|P8DNDFkj0F~+e&b`YbgQ-Ul|3Qf9*&VQ=E|WsJ531 zN%wLI>I+H&Y5lsirV6UZs+!Rw*k;K$%zhKt(M@SooCuXazt+O3PR7L{as!ge2YUx5 z+BDy8FgZ+p=4Gr;8j-mJ_c!(1-R)LZsj$D-0_|#>&jCB7O~l%hn$n|{x!}ikG6|fR zcArzU;lNsSgg+?rHO6_NlvFoHYkyPVqWA@NH>t3nIJjFXN>P1oQ@(l2;xH5qd;NkB zE40&Z+gZnvhi+86E%ZdycFC($yJQ+zkJZ!838i#9Q+l40ZU?s>?rgTp9kt;^k8BSN zTRL(ioa3tzR_-!ead-E2lm1Q=Z}oSQ!KOm+MP4oo^;OjM4y3z?95+Fud?yv_G)^MF zoGkqpR~+{{1WW@b0x=Jre3wKCM!bs|A29im zG`JpOA=OZOpELIAIn$%gspv)3gxz^jH+lvW5&d}gu_F$F%S;mQ@Jds^1`VqC7v)Y6 z1RL3&oGxtatck{i19No`7F zvx4WBxzNaELUIj|hgz>otPSlMH{v5apv;9NLrwY9@C)fzS5^rngh(~@;y0RPOCBoj z4J7eX*RZjGi>edC))#^D3H*NxXlt+4Y9w-k`ZKxzR_e$3?M1! z*Hf|*rPq16Y;=B^Ux3`p*mr()sa~Tl*76Gwj#tXegW#;fi$<>ZLd)Ek3II1n833?C zD0wA;!UWSNXeH+qAPX<#5CLs*0(IQ;i`je$b%aq|)v58Jh;Ye;N0ly!(2xdTo`}o* zlX#5l!IDTu))-Y)J1BhEyITHu`mx-2H=qEsTmTL)|YXWlIrTsk~0$o4*=g#)s0i>2>u|O zjFYIHWAsDgR92rl2T^r!PB`A@d0wY_nn(Z8|8m*aJ^P2j*@zKX7Fl4)m&NM3Cg_lvlp#-@)5r;JkZe66Z03OjRH*ch9Tl~`c&Y)qmLYlYhM zEd?Wmr5c1Z&#FHEP2lh=$+rJO*tdMCjgv$YsEn?$O`wb-?xsM3mY@h^DqYY%dPHec zhnxWFsz^C+nn87;3z6Dw%!a=`*iH7d8dPNxm6tG}PKSx79_O$K8rhY}TBV|WtLoW) zn=g`R{sH%r5@=NfMk-+~`rwYKS3`AO-YrrVuJA(4ge8E3#3mn-t!=jQNu~Bn9Do6M z&+EY5HJLYWmYlT)EnXw-MyL1M5q+o~F;+e~=_q*`Pj$Y^Jq|&48tp$6H@@mhsy35~ z2*EN+`LckE)#+r_(U%s8km6*0nk-jKvlRD<0Dk&RKA@AKriBk9MrZSo$i&5&let)p z?CF!VG{hTURzb?mY?6su8OGTs<8Vm*q!{08JE!?4j@7`bSDEtko#gGa}mAui@ z$&BqvfZpICby3vqxYn+GTh23YzvDd6*zgvcT-DVo+v}wuVtsnVp;3Zbo`~6e2AgEP zgukSGCdL@2^&y(+Gp{COPlyG_(46#R*I%OYK3DI$ldD*BT2%tG4!T}IFc=_~P+2?L zHNpBNOQmU~8cKTQB9kIXRhfr(K=kvk9UmAfg2e`#f5-r4$(Mj7M`F@L&$7Ph$RYsz z;LV3CYOGL)S>)Dz4o@Yz>bduSCiRtm%kyr?7Pg)p-u~#(0`BV#Gq3!1n@KcaZ*rk1 zTS`10RR#Q{gNigBPuWwROU#wFxtDB3@y^a5+1s)uAX~x5J32{xj2S9PO%35o7}CET zeDv%3D^~M;%DwhV_fbHwyp}$X4Ouc;sCs`diUlgB?**lwmk=l-Z@#Dn0dUQd^yT%|8*&%T*CYhyLbePA;t(7}b zv$592ZQ_MS<}?Y@wv4H1s7*gK?RV;$H}rnlAu%q1iK5sX%5w{wn76M1z9jdm_6_PG}|la)>KO))sA=v9m_jtX|b%BGNpM^8ZE{Cc+Hhkeej9X@^062n~vkN z$a4#EtP&d2`>kskIuDuWs^ZS5Rz724%UIOOB<^RY+VuJu`l!VT)=Izai@L)@dk);p zc33wc;ZU7)w#SOsxNqrJ_weyy4L5;p_s#6FcV@lH)O56$JvZY6Sv3eb13&%YXgK^7 z@U@k{9xvv_ z^1J!!_3XST*@*w91k3QNaz1?vPD&m(>^sxq=|X%3S4D4Tmp4@>jtZ#w(Zzh_c;76$ zW>8=ybQo8`BHmpfmf3Uf^dG`~RRv&$M+DA6eq9grE zH_|li^pKl7ZwBDtpdny`R+Z|Ii3A%unR;$`I*EsTU7mIHh`YO==pA({soFc#v}+bM zuog9N7j>T(4r*VxRr|ubJ2ur`K*__ZUymR(eWEJtrGMVl(I!!8`2Rc+fUCzJ^%320cw=tf4uEZ6oVUuU4S;J%wL5URsfK@f;JD zI|IBifn=Plid^GNg#IDEGTyRegUv+Vvt+v_X4TAaJ=FeQUxV!<*o*l>X#MZrK94rD z$TG=9=MXORuw_^;+%*yQf$JE!V6AoNu%b_k%lWt%%YzWmY6|0kP*NOmCbmHTh88nf zj@z#N(!+ZEDU2GdA0=(Mr(VHZRuP)goI&W0Khj|{%*c@b#CQwak$N=m07PvHP!Qem z`UTa8C!42fLu-bAu!_`rT(w#HN3eq`^!Jz^Vcg@;Kr88l-V1F5m0=0t-4KB4H>CfN z4e5KA7qBmemfEjbsvh28sr@xe9R!xDE-w6w99WBVZZ7J^1lFVu{Q4CQ)QTuLdyI*+ zDmTdfLqPYry1t;EL>KEDhK=5Q;xtc@tdHlV>~pm=}-?%tT6rjT3>O_OHD@M zo?MWY`~$LhaP^UZ*YWgs#o{u*5z-Ebr_xyoQV9qdY_yW*Q{`auFh>8IGsJ&m4&1*F zOY8?&qGE0;2M>nt^u-C{_^tX7R6~R)toM6=x z|J}G><}*BE)b%U%*i;?k_G z)LxJ4q9Vj>xwCazrJthl+&H0*TT19xlwd(4HH}$alAR%jTLp zAbIHVLJV7-w=jJ`eGZi#Nj!$XMYGQ3;5UdQMp$I_WSPi_Ljl<1qAptX;SnOio%ru3 z!)Z}2+v85_GksLXC6-4tdQAVxcQDZ|_2QAyiVp1jLp!l&omBib{n#`7g`CZ;pK`=?V47AV4#O~u5eo%$>@VpR9zT*})>I_+6b@)xkT^Vv zCiRHJqwR9U;Sy3cAEHGS+%S}C5!`SUoiG%C?b~26voIi|#B#U>l&1oeztnBGN>j}Y z8kDD9GrCzh(=ZUHnrXNKQ|&$mtI~#xMAaahuU>Aricyobd8@yM%ME#IRot&|tEV)_ zed@M`X-ng>tzi%?tG2EK)S_ssr^3%eYPQzQ?4BiCJyk|pS8NT#XkD;%9jN=&YhAxc zyIgCylv?;Cdg@GvE82t)P4iN%;YHo2QfrtVG%wT|o{?3lrmk%E$*FGR%sA_+W{2w$ z)<{g=Duk`X!26Yrtq0V!Tx=Z_vrKF&VbCiM#_hzx_QhfACeoxRY~6Uf7;Gz{@Lu&q z>)=%Dh_(_8>&q&(5)FOzZCeS4p~9f8#KTs_JJ*hHR8+^Ox{_oal-3m_*8{s><;W&WZCWw1p3kqV6xm9at}92}N}9e$UC6pE zyH7pHI Date: Sun, 27 Sep 2026 12:34:25 +0530 Subject: [PATCH 18/20] docs(product): record retrieval limits and next gates --- docs/README.md | 114 ++-- docs/continuation/astra-understanding.md | 24 +- docs/continuation/go-intelligence.md | 243 +++++++-- docs/go-intelligence-north-star.md | 508 +++++++++++------- docs/plan.md | 318 ++++++----- docs/research/codebase-indexing-retrieval.md | 502 +++++++++++++++++ docs/v1.0.0-roadmap.md | 6 + .../v1.0.0/adoption-remediation-2026-09-24.md | 65 +++ .../command-activity-audit-2026-09-25.md | 73 +++ 9 files changed, 1351 insertions(+), 502 deletions(-) create mode 100644 docs/research/codebase-indexing-retrieval.md create mode 100644 validation/v1.0.0/adoption-remediation-2026-09-24.md create mode 100644 validation/v1.0.0/command-activity-audit-2026-09-25.md diff --git a/docs/README.md b/docs/README.md index e6d495a..8e8fc53 100644 --- a/docs/README.md +++ b/docs/README.md @@ -8,77 +8,55 @@ compatibility baseline. Shared interfaces and invariants live in [`contracts.md`](contracts.md). The completed v1 implementation stages and their evidence live in [`v1.0.0-roadmap.md`](v1.0.0-roadmap.md). The architectural rationale is -recorded in [`decision-memo.md`](decision-memo.md); [`plan.md`](plan.md) is the -concise product plan and routing summary. +recorded in the historical [`decision-memo.md`](decision-memo.md); +[`plan.md`](plan.md) is the current product summary and routing guide. ## Next-generation Go intelligence -After the contributor instructions, a new agent should start with the -[Astra analysis handoff](continuation/astra-understanding.md): product judgment, -source evidence, confirmed observation defects, deferred work, and the next -decision. Its quick start routes the next task without repeating the broader -investigation. The [continuation handoff](continuation/go-intelligence.md) owns -implementation status and the next action; the -[Go intelligence north star](go-intelligence-north-star.md) owns the approved -architectural direction and acceptance criteria. -`v1.2.1` (tag `67f54b7`) is the latest released baseline. The current -`codex/v1.2-reliability` branch contains unreleased post-v1.2.1 work. Its -product direction is a deterministic, snapshot-bound evidence compiler that -supports the edit, refresh, verify, inspect loop. Slice 1A adds next-action -guidance to existing MCP text, and Slice 1B refines private evidence projection -guidance. Both preserve the frozen public MCP inventory and -`agentic.focus/v1` schema. - -Private local Luna evaluations now include a 20-run discoverability pilot and -a 12-run integrated adoption follow-up. In the focus arm, capability delivery -was healthy in 10/10 runs, but no run called `go_context`. All six integrated -runs discovered the shipped skill, called `go_context`, refreshed after -editing, and used the evidence. Integrated median duration was 431,641 ms -versus 223,758 ms for baseline, about 93% higher; median tool calls were 24 -versus 26. Astra judged workflow adoption locally demonstrated for the -combined skill and MCP surface, with comparative -product value still unproven. They do not establish MCP-alone causality, -improved correctness, productivity, token efficiency, or speed, statistical -significance, generalization, or production readiness. Both studies used two -scenarios and `gpt-5.6-luna` at max reasoning. The integrated follow-up used -three repetitions per scenario per arm. Reports remain private and are not -tracked. See the -[historical adoption results](../validation/v1.0.0/adoption-results.md) for -older v0.8/v1.0 evidence. The two current studies are regression evidence, not -fresh proof of product superiority. Review of the six integrated traces is -complete: no transcript establishes that context improved the necessary code -edit. In gRPC run 3, refreshed context prompted broader `./...` verification, -which hit the output cap; focused verification later passed after a stale -snapshot rejection. Client-go runs made 3-4 context calls each, and gRPC runs -made 4-5, including ambiguous or unhelpful selections. The bounded guidance -improvement is to narrow ambiguity using returned candidates, finish each batch -of edits and formatting before refreshing, and refresh again after further -edits or stale-snapshot rejection while avoiding redundant refreshes when the -snapshot is unchanged. This observation does not establish causal edit-quality or product -value. External and multi-model evaluation remain pending. Delta refresh -remains deferred. These documents do not override frozen v1 contracts. - -A post-guidance regression used six integrated Luna runs, with three -repetitions on each of two scenarios. All six qualified, passed acceptance, -stayed within scope, and required no operator intervention. All six called -`go_context`, refreshed after edits, and recorded evidence use. Five focus -calls failed: three client-go calls returned `invalid_input` for invalid -symbol references, and two calls in grpc-go run 1 returned `stale_snapshot` -because an observed semantic location was absent from the snapshot manifest. -grpc-go runs 2 and 3 had no failed focus calls. This shows workflow adoption -continued while exposing failure categories; it does not establish improved -quality, correctness, productivity, speed, or product value. Before changing -provider behavior, the audit classified the client-go failures as malformed or -reconstructed refs and the gRPC failures as unbound workspace locations. The -provider now omits unbound locations with bounded uncertainty while preserving -strict stale rejection. External and multi-model evaluation remain pending. - -After the failure remediation, six integrated Luna reruns completed with 6/6 -qualification, 6/6 acceptance, zero scope violations, zero operator -interventions, zero failed focus calls, and complete refresh and evidence-use -signals. The median duration was 411,537 ms and the median tool-call count was -31. This is diagnostic regression evidence only and does not establish product -value or comparative engineering benefit. +After the contributor instructions, read the +[Go engineering north star](go-intelligence-north-star.md) for product direction, +model/skill requirements, the next delivery cycle, and acceptance criteria. +Then read the [continuation handoff](continuation/go-intelligence.md) for +implemented behavior, current evidence, and the exact unfinished step. +The [Astra source review](continuation/astra-understanding.md) is dated +historical background, not current implementation sequencing. + +The first customer is a Go engineer using coding agents on real repositories. +The intended workflow covers understanding, implementation, debugging, +refresh, verification, and review/resume. The next product cycle makes one +cross-package API/interface change dependable from start to handoff, using the +existing deterministic evidence compiler and concise shared skills. + +The user-directed aspiration for install-time, branch-aware repository +indexing and repeatable retrieval is recorded separately in the +[codebase indexing research note](research/codebase-indexing-retrieval.md). +The current development branch adds an exact branch source-view preview and +records the first retrieval screen. It does not yet provide persistent +indexing or demonstrate comparative product value. + +`v1.2.1` (tag `67f54b7`) remains the released baseline. The current +`codex/agentic-go-retrieval-2026-09-27` topic branch contains unreleased work. +The frozen v1 registry +remains 14 tools, seven resources, one template, and six prompts; the existing +additive `go_context` brings the server to 15 tools. The revised plan does not +change that inventory, schemas, or strict freshness behavior. + +| Read for | Authority | +| --- | --- | +| Customer, model independence, skills, and workflow outcomes | [North star](go-intelligence-north-star.md) | +| Current implementation, remediation, and next action | [Continuation handoff](continuation/go-intelligence.md) | +| Branch source-view preview, retrieval findings, and next gates | [Codebase indexing research](research/codebase-indexing-retrieval.md) | +| Current campaign's implementation/check status | [Selector remediation record](../validation/v1.0.0/adoption-remediation-2026-09-24.md) | +| Older adoption experiments | [Historical results](../validation/v1.0.0/adoption-results.md) | +| Compatibility and execution invariants | [v1 freeze](v0.9.0-release-scope.md) and [contracts](contracts.md) | + +Keep the current canonical Luna matrix separate from older integrated +diagnostics. Workflow use and task acceptance have been observed; comparative +engineer value and broad model/host compatibility remain unproven. The plan +requires comparison against equipped native Go/gopls workflows, plus actual +review/rework effort. It does not promise equal competence across models or an +unreproducible capability advantage. Full-replacement refresh remains current; +delta delivery stays deferred. ## Verification report contracts diff --git a/docs/continuation/astra-understanding.md b/docs/continuation/astra-understanding.md index 66ef9b4..920feed 100644 --- a/docs/continuation/astra-understanding.md +++ b/docs/continuation/astra-understanding.md @@ -1,10 +1,15 @@ -# Astra analysis handoff +# Historical Astra source review -Status: authoritative continuation note, reviewed 2026-09-05 at HEAD -`284df97`. This records source inspection and product reasoning. It is not a -test report, benchmark, implementation approval, or proof of runtime behavior. +Status: historical review from 2026-09-05 at HEAD `284df97`. Current product +direction is in the [north star](../go-intelligence-north-star.md); current +implementation status and sequencing are in the +[continuation handoff](go-intelligence.md). This review is optional background. +Its scores, defect statuses, and next-slice instructions describe that dated +inspection, not current work. Do not restart completed observation work from +the prompts below. This is not a test report, benchmark, implementation +approval, or proof of runtime behavior. -## Fast path +## Historical fast path **Verdict: NARROW. Confidence: 88/100.** Build a small, dependable Go evidence compiler for coding agents: bind observations to an explicit @@ -41,10 +46,9 @@ semantic_gopls,snapshot,snapshot_test}.go`; continuation and north-star docs were untracked at inspection. Inspect current status and only the relevant diff because this dated state can drift. -After `AGENTS.md`, fresh agents should read this file and -`docs/continuation/go-intelligence.md` first, then the -north-star and only the frozen scope/schema or source files relevant to the -slice. Do not repeat the full historical review unless touching that area. +For current work, follow `AGENTS.md`, the north star, and +`docs/continuation/go-intelligence.md`. Consult this historical review only +when a relevant source rationale is needed. Its old sequencing is superseded. ## Product judgment and comparison @@ -189,7 +193,7 @@ through the observation; how to expose applicability of “latest” verificatio contract coordination; provider/editor overlay semantics; and whether agents use the evidence often enough to justify its cost. -## Prompts for a fresh agent +## Historical prompts Analysis only: diff --git a/docs/continuation/go-intelligence.md b/docs/continuation/go-intelligence.md index d12d50c..8feac89 100644 --- a/docs/continuation/go-intelligence.md +++ b/docs/continuation/go-intelligence.md @@ -6,12 +6,13 @@ This document preserves the current product understanding for a future agent working in another account or session. It is independent of conversation history, account identity, private memory, and previous tool output. -The product direction is a deterministic, snapshot-bound evidence compiler for -Go coding agents. Its central workflow is edit, refresh, verify, inspect -evidence, then reconsider what the evidence requires. The goal is to help an -agent understand what to revisit before treating a Go change as complete. -Effectiveness claims require comparative evidence. The private Luna studies -record workflow adoption, but do not establish comparative product value. +The first customer is a Go engineer using coding agents on real repositories. +The product should support understanding, implementation, debugging, refresh, +verification, and an inspectable handoff across supported models and hosts. +Its deterministic, snapshot-bound evidence compiler should reduce reliance on +model memory and repeated engineer investigation. Comparative benefit remains +unproven. Model-independent contracts do not guarantee equal model competence +or prevent an agent from ignoring evidence. The full approved architectural direction is in the canonical [Go intelligence north-star plan](../go-intelligence-north-star.md). Approval @@ -32,20 +33,38 @@ is implemented. This handoff points to the plan instead of duplicating it. Do not repeat broad documentation exploration unless a source fact below has become stale. -## Current status - -`v1.2.1` (tag `67f54b7`) is the latest released baseline. This branch, -`codex/v1.2-reliability`, contains unreleased work after that release, including -commits `8e65ce1` and `71099dc`. Do not describe this branch as a release -candidate or assign a new release label before a reliability milestone passes. +The [Astra source review](astra-understanding.md) is historical background from +2026-09-05, not current sequencing authority. Read it only for a relevant +source rationale; its old next-slice instructions must not restart completed +observation work. -The current product wedge is to make the edit, refresh, verify, and inspect -loop dependable enough that a coding agent knows what to reconsider before -declaring a Go change complete. The system provides deterministic, -snapshot-bound context and verification evidence; it does not decide that the -engineering task itself is complete. +## Current status -The public MCP inventory and `agentic.focus/v1` schema remain frozen. Slice 1A +`v1.2.1` (tag `67f54b7`) is the latest released baseline. The integration +topic branch `codex/agentic-go-retrieval-2026-09-27` contains unreleased work +based at `7b5111c`, including preserved dirty-worktree changes, branch +source-view support, retrieval evaluation infrastructure, and updated evidence +records. Do not describe it as a release candidate or assign a release label +before the applicable release gates pass. + +The 2026-09-24 north-star revision selects a complete cross-package API or +interface change as the first product workflow: understand obligations, edit +and debug, refresh, verify, and hand back current evidence. Offering focused +declaration candidates from the observed diff and completing the review +handoff are proposed next increments. They are not implemented by this +documentation change. The existing system does not decide task completion. + +At the earlier `7b5111c` inspection, separate uncommitted selector-remediation +work was already present. Its +[campaign record](../../validation/v1.0.0/adoption-remediation-2026-09-24.md) +reported focused checks complete, with full post-change gates and evaluation +reruns then pending. The current selector-screen result and rerun boundary are +recorded under Current next action. Do not overwrite existing work or infer +that a documented plan has been executed. + +The frozen v1 registry remains 14 tools, seven resources, one template, and six +prompts. The existing additive `go_context` brings the server to 15 tools; +`agentic.focus/v1` and the frozen v1 schemas remain unchanged. Slice 1A adds `next_action` to existing MCP text using the existing structured `Verification.NextAction`. Slice 1B refines the private projection with cause-specific guidance for stale or unavailable evidence, truncation, budget @@ -109,11 +128,12 @@ unavailable. A passing requested check is not a declaration that the task is complete. The follow-up guidance is bounded, provenance-linked, and non-mutating. -## Findings recorded from the documentation and targeted source inspection +## Historical source inspection Use symbol names to relocate sections if line numbers change. These are -recorded static inspection findings from the earlier review, not a runtime -audit or source inspection repeated during the documentation revision. +findings from the earlier architecture review. Several proposed improvements +in this table are now implemented in focus or observation, as recorded below. +Do not use the table as a list of unfinished work or as a current runtime audit. | Source and entry point | Observed behavior | Consequence for the plan | | --- | --- | --- | @@ -151,11 +171,11 @@ replay from model outcomes. The reviewed [v1 release evidence](../../validation/ does not establish a paid model pilot or comparative agent advantage. These are historical document statements, not checks rerun in this session. -Comparative evaluation is not a product milestone for this reliability work. -Prioritize material product behavior: context selection, coherent reads, -Go-specific relationships, explicit refresh, and trustworthy verification -applicability. Additive post-v1 interfaces are documented separately from the -frozen v1 contracts. +The original reliability work did not require comparative product claims. +Its implemented foundations remain useful regardless of later evaluation. +The revised north star separately requires evidence of engineer value before +widening usefulness or model-support claims. Reliability release gates and +comparative product decisions must not be conflated. These are repository observations and design opportunities, not claims that the current runtime is fast, relevant, or superior to other tools. Those @@ -163,37 +183,37 @@ properties have not been verified in this documentation task. ## Important limits -The approved plan is architectural direction, not an already-frozen wire -specification. Implementation must resolve and record these local facts in the -stage that needs them: +The following boundaries apply to the implemented foundations and the next +product cycle. Resolve additional design decisions within the slice that +needs them; do not silently expand a frozen contract. -| Stage | Facts to inspect and decisions to record | +| Area | Current boundary and remaining decision | | --- | --- | -| 2. Coherent observation | Implemented: request-scoped ownership, atomic retain-and-pin, release-once leases, captured-or-manifest-verified reads, guidance identity, strict admission, and existing provider ordering. The provider still observes the live disk workspace through gopls; this is attributable evidence with final validation, not transactional filesystem isolation. | -| 3. Useful context | Exact input/output fields, existing budget ceiling, combined MCP rendering costs, required-envelope failure behavior, and interoperable current Symbol Refs. | -| 4. Implementation support | Provider capabilities and typed predicates for method sets, embedding, aliases, generic instantiations, enclosing tests, lifecycle sites, and partial-source results. | -| 5. Refresh | Private delivered-manifest storage and retention, logical identity versus content revision and locator, current-reference issuance, and evidence that can confirm deletion. | -| 6. Integration | Explicit additive inventory expectations and verification references tied to the observed snapshot. | +| Observation and execution | Request-scoped observations, retained manifests, and final validation are implemented. gopls and verification commands still observe the live workspace. Require stability during checks; endpoint equality does not establish isolation from transient concurrent edits. | +| Context entry | Query/ref/position/file/package selectors are implemented. At `7b5111c`, `focusContext` returns no focused context without an explicit selector, even though `Core.Focus` separately computes the diff. Current declaration candidates from that diff are proposed. | +| Implementation support | Typed predicates, enclosing tests/examples, and partial-source evidence exist. They expose supported source facts, not behavioral equivalence, complete dispatch, or intended business requirements. | +| Refresh | Full replacement and private delivered-pack metadata are implemented. Old refs stay stale; failed resolution does not confirm deletion. Delta delivery and semantic before/after obligation differences remain deferred. | +| Review handoff | Reports and applicability assessments exist. `CurrentVerification` retrieves the latest stored report; `assessVerificationApplicability` decides whether it applies. The proposed handoff must carry both result and applicability. | +| Model/host support | The recorded adoption campaigns use Luna/max. Shared instructions and real delivery/recovery checks across other clients/models remain product work, not demonstrated compatibility. | +| Branch source view | The development preview creates a visible detached Git worktree at an exact branch commit. It supports explicit local/remote refs, defaults to `main` then configured `origin/HEAD`, binds optional dirty overlays only to the exact source HEAD, and rejects moved refs during snapshot capture. Configure the agent at the exact returned view root. It does not persist retrieval state or capture external `go.work`/local `replace` inputs; submodules are reported as partial. | The behavioral decisions are already fixed by the north star: selection occurs before expensive expansion; required identity and uncertainty survive budgets; incomplete source does not license stale semantics; and response omission is distinct from confirmed source removal. Cache limits must cover bytes and -entries, and active observations must survive ordinary eviction. Applying a -delta to the previous delivered pack must reconstruct the current selected -pack, including locators, omissions, and uncertainty. Historical snapshot-bound -artifact cursors remain stale even when retained pack metadata is used for -refresh. +entries, and active observations must survive ordinary eviction. A refresh +returns complete current selected evidence; no delta reconstruction is required +for this cycle. Historical snapshot-bound artifact cursors remain stale even +when retained pack metadata is used for refresh. Do not advertise inferred ownership, complete dispatch reachability, semantic goal enforcement, or a measured speedup. Disk snapshots do not cover unsaved editor buffers. Upstream gopls MCP already exposes workspace, package, navigation, and diagnostic tools and has its own internal snapshot model. -Differentiate through cross-operation evidence lineage, impact, selection, and -verification applicability rather than duplicate navigation wrappers. The -reviewed upstream source disables its broad `go_context` tool because of -context-size/redundancy concerns; this supports bounded explicit context, not -a claim of unique navigation. +Compare against upstream gopls with its actual workflow instructions. +Cross-operation lineage, impact, selection, and reviewable verification +applicability are candidate sources of value, not established differentiation. +Another harness can reproduce these mechanisms; measure engineering outcomes. The minimal read-only change-consequence and verification-applicability slice is now implemented. The additive `go_context` MCP tool and `agentic-go context` @@ -217,6 +237,25 @@ before relationship expansion. Selection reasons and uncertainty distinguish absent, unavailable, unexamined, and budget-omitted evidence. The 8 KiB default is bounded before optional expansion, and CLI/MCP summaries share one renderer. +The current working tree adds a private, process-local Go retrieval cache for +query selection. It parses observed `.go` files into declaration fragments and +combines deterministic lexical scoring with symbol, receiver, package, and +declaration-kind signals before resolving candidates through the current +gopls observation. It is advisory discovery only: `Core.Search`, the MCP/CLI +surface, schemas, exact snapshot validation, and semantic evidence contracts +are unchanged. Focused retrieval and focus tests are present; repository-wide +validation has been requested. The model-free screen and private text-candidate +ablation are now recorded in the research note; they show weak relevance and +do not qualify this slice for product-value claims or persistent indexing. + +The longer-term branch-aware indexing aspiration, historical GPT-6 Sol +findings, current GPT-6 Luna Max verdicts, and retrieval-specific evaluation +plan are recorded in the +[codebase indexing research note](../research/codebase-indexing-retrieval.md). +The current cache is not a durable or branch-aware repository index. The next +source-view preview selects an exact branch commit, but branch selection is not +part of the live MCP retrieval contract. + Full-replacement refresh is now implemented. Delivered focus evidence carries an opaque pack ID backed by private content-addressed metadata containing the original selection, logical declaration identity, and digests for the evidence @@ -275,6 +314,97 @@ refresh. Raw artifacts remain private and ignored. ## Current next action +The six-cell selector screen is complete. Its safety gate passed and its +efficiency promotion gate failed. This is selector-guidance reliability +evidence, separate from retrieval evaluation. Do not run or direct the stale +18-cell rerun. Keep that result separate from the historical 18-run adoption +campaign and from retrieval evidence. + +The model-free retrieval screen is complete; results and limits are recorded in +the [research note](../research/codebase-indexing-retrieval.md). The corrected +medium screen is weak (Recall@10 0.35, Precision@10 0.20) and indexes none of +the 150 supported text files. The small screen has no usable retrieval score. +The Kubernetes capture is incomplete. Its single-run warm p95 was 10.52 s, +above the 5 s screening target; sampled Go heap was about 703 MB, while process +RSS was not measured, so the 512 MiB process target is unresolved. The fresh +four-question text-candidate ablation found complete candidate-pool Recall of +1.00 but weak top-10 macro Recall 0.1625, Precision 0.075, and MRR 0.28125. +The fixed `rg` scorer is not a competent native-agent baseline, and gopls +completeness is unknown. These results do not support a product-value or +arbitrary-scale claim. + +The exact branch source-view preview is implemented as +`agentic-go source-view`. It creates a visible detached worktree, selects +local `main` by default (then configured `origin/HEAD` if absent), binds a +dirty overlay only to an exact matching source commit, and rejects stale branch +refs at snapshot capture. It preserves `go_context.base` as change context. +It is a source-view setup command; persistent indexing, automated MCP branch +selection, and full external Go workspace capture are not implemented. + +The one bounded retrieval redesign allowance has been used for an evaluation- +only text-candidate ablation. It established that text evidence can enter a +complete candidate pool, while the unchanged scorer still ranks too little of +it in the top 10. Do not add persistent indexing or claim retrieval value from +this result. A future retrieval proposal needs a new bounded scope and fresh +held-out cases. Repair and freeze the native `rg`/Go-tools/gopls workflow before +any comparison. The small and large source-coverage gaps and unknown gopls +completeness remain open evaluation limits. + +Persistence still requires useful retrieval and a measured parsing or ranking +bottleneck; its gates remain at least 2x warm-query improvement with no +freshness or relevance loss, against the 5 s warm p95 and 512 MiB process +screening targets. The separate 32-run Luna Max engineering screen remains +gated because useful retrieval has not been demonstrated. + +Only after useful retrieval is demonstrated should the separate engineering +screen run: eight held-out tasks across three repositories, randomized paired +order, and two fresh repetitions per arm (32 GPT-6 Luna Max runs). Freeze the +source, binary, task, prompt, and evaluator hashes first; use GPT-6 Luna Max +only and keep the study separate from the model-free results and historical +GPT-6 Sol/high work. Require zero accepted stale or wrong-branch evidence, no +accepted-patch quality loss, and a practical gain such as 15% lower median time +to an accepted patch. Track review effort and actual token usage, and report +the same-model reviewer limitation. The screen is directional, not a +statistically powered or general claim. + +The proposed R1-R4 workflow work remains a separate product cycle. Follow +[R1 and R2](../go-intelligence-north-star.md#next-delivery-cycle) when that +reliability milestone is resolved: current declaration entry from the diff, +then the complete API-change and review handoff. Use existing typed evidence, +renderers, and schemas. The [north star](../go-intelligence-north-star.md) +owns their acceptance cases and model/skill contract. + +## Historical canonical adoption campaign + +The maintainer-supplied 2026-09-24 canonical summary records 18 Luna/max runs: +two pinned scenarios, three conditions, and three repetitions per cell. All +18 qualified and passed acceptance. These reported results are not rescored +by the documentation revision. + +| Condition | Runs | Context evidence-use signal | Refresh | Median duration (ms) | Median tool calls | +| --- | --- | --- | --- | ---: | ---: | +| Baseline | 6 | 0/6 | 0/6 | 206,861 | 17 | +| MCP-only discoverability | 6 | 0/6 | 0/6 | 249,512 | 22 | +| Guidance | 6 | 6/6 | 6/6 | 308,899 | 26 | + +Guidance recorded zero scope violations and zero accepted stale evidence. +There were five failed focus calls: one low-level `invalid_input` and four +low-level `provider` labels. The private transcript audit classified all five +as selector misuse: an altered Symbol Ref and invalid declaration coordinates. +No provider defect was established. The current remediation preserves those +raw outcome categories and adds bounded private cause/recovery classifications. +See its [record](../../validation/v1.0.0/adoption-remediation-2026-09-24.md) +for implementation details; the later selector-screen result and current +18-cell rerun boundary are recorded above. + +The pooled guidance/baseline duration difference is about 49%, but per-scenario +medians differ by about 17.5% for client-go and 4.8% for grpc-go. These are +descriptive, unpaired small-sample summaries; none identifies causal tool cost. +The evidence-use signal is temporal ordering, not a scored better decision. +The older campaigns below must not be pooled with this matrix. + +## Historical integrated diagnostics + The post-guidance regression used six integrated Luna runs, with three repetitions on each of two scenarios. All six qualified, passed acceptance, stayed within scope, and required no operator intervention. All six used @@ -309,7 +439,7 @@ transcript evidence was 450,293 bytes. These are diagnostic regression values, not evidence of improved speed, correctness, productivity, or product value. External and multi-model evaluation remain pending. -## Earlier trace review +## Historical integrated trace review The six integrated traces have been reviewed. No transcript proves that `go_context` improved the necessary code edit. In gRPC run 3, refreshed context @@ -349,10 +479,19 @@ correctness. Use this prompt only when the user separately authorizes further work: ```text -Read docs/continuation/astra-understanding.md and this handoff first. Observation, -verification applicability, declaration selection, and full-replacement refresh -are implemented. Continue the current reliability milestone from the current -next action in this handoff. Delta refresh, general derived caches, expanded -refactoring, and speculative test selection remain deferred. Preserve existing -tags and public history. +Read AGENTS.md, docs/go-intelligence-north-star.md, and this handoff. The first +customer is a Go engineer using agents; the first complete workflow is a +cross-package API/interface change through an inspectable handoff. Observation, +typed context, full-replacement refresh, and verification applicability exist. +The six-cell selector screen completed with safety passed and efficiency +promotion failed. Do not run or direct the stale 18-cell rerun. The model-free +retrieval screen and one private text-candidate ablation are complete; their +weak ranking does not qualify persistence or an agent-value study. Repair the +native baseline before future comparative claims. After useful retrieval is +demonstrated, use a separate matched native-versus-Agentic-Go screen with +GPT-6 Luna Max only. Preserve both dirty source worktrees. Do not repeat +completed work or treat plans as shipped +capabilities. Continue only the user's requested scope; preserve frozen v1 and +agentic.focus/v1. Delta refresh, general caches, expanded refactoring, and +speculative test selection remain deferred. Preserve tags and public history. ``` diff --git a/docs/go-intelligence-north-star.md b/docs/go-intelligence-north-star.md index b7ac025..f019c55 100644 --- a/docs/go-intelligence-north-star.md +++ b/docs/go-intelligence-north-star.md @@ -1,99 +1,133 @@ -# Go intelligence for navigating, generating, and changing code - -Status: approved product direction, revised 2026-09-23. `v1.2.1` (tag -`67f54b7`) is the latest released baseline. The current -`codex/v1.2-reliability` branch contains unreleased post-v1.2.1 work. The -product is a deterministic, snapshot-bound evidence compiler for Go coding -agents. Its wedge is a dependable edit, refresh, verify, inspect loop that -helps an agent know what to reconsider before treating a change as complete. - -The `go_context`, `agentic-go context`, and `agentic.focus/v1` capabilities are -post-v1 additions. Slice 1A adds next-action guidance to existing MCP text -using existing structured verification data. Slice 1B refines the private -evidence projection with cause-specific guidance. Both preserve the frozen -public MCP inventory and `agentic.focus/v1` schema. See the -[continuation handoff](continuation/go-intelligence.md) for current milestone -status and next action. - -Start with the [continuation handoff](continuation/go-intelligence.md) for -implementation status, source pointers, and the exact next step. Existing -[v1 interfaces](v0.9.0-release-scope.md), [shared contracts](contracts.md), and -[verification behavior](v0.2.0-release-scope.md) remain authoritative for shipped -behavior. +# Go engineering workflows across coding agents and models + +Status: maintainer-directed product plan, revised 2026-09-24. This revision +updates direction and delivery criteria; it does not implement capabilities or +establish comparative product value. `v1.2.1` (tag `67f54b7`) remains the +released baseline. The [continuation handoff](continuation/go-intelligence.md) +owns the current branch, implementation status, evidence, and exact next action. +The [v1 interfaces](v0.9.0-release-scope.md), +[shared contracts](contracts.md), and +[verification behavior](v0.2.0-release-scope.md) govern shipped behavior. + +## Customer and intended outcome + +The first customer is a Go engineer using a coding agent on a real repository. +Agentic-Go should help that agent understand, implement, debug, and verify a +change, then leave an inspectable handoff. The engineer should spend less time +reconstructing the investigation, correcting avoidable mistakes, and checking +whether the reported evidence still applies. + +The technical foundation is a local, deterministic evidence compiler: Go +semantics, change impact, bounded context, guarded operations, and verification +lineage. The product earns its place through better engineering decisions or +less effort to obtain an acceptable change. More calls, larger reports, and +successful instruction following are insufficient evidence of that value. + +The ambition is to reduce dependence on what a model remembers or guesses. +Small open models and stronger proprietary models should receive the same +defined operations, source facts, and failure semantics. Each should be able +to use the product through a supported host without learning repository +internals or reconstructing opaque identifiers. Equal task success across +models is a hypothesis to evaluate, not a property of the tool protocol. + +## Model independence and limits + +Separate three promises: + +| Layer | Required property | Evidence needed | +| --- | --- | --- | +| Engine | Equivalent explicit inputs and repository state receive consistent contracts, grounded facts, limits, and rejection behavior. | Deterministic contract and failure-path evidence. Test results, provider availability, and timing need not be deterministic. | +| Agent integration | Supported hosts deliver usable text/structured evidence, instructions, exact references, and recovery paths to the model. | Real client checks, including a compact-context model and a stronger model; protocol delivery alone is insufficient. | +| Engineering outcome | An engineer obtains an acceptable, inspectable change with fewer consequential omissions or less total work. | Comparative tasks and observed use in real repositories. | + +The engine must not change truth, freshness, or check semantics based on a +model name or confidence score. Host-specific configuration and instruction +placement may differ. Publish support only for combinations actually exercised; +do not turn a Luna result into an open-model, Sol, Astra, or all-agent claim. + +A model can ignore evidence, misunderstand a requirement, or author a bad +algorithm. Selected checks cannot prove an arbitrary business requirement, +and advisory tools cannot stop an agent from claiming completion. A future +completion gate would need an explicit host/CI enforcement point and a +machine-checkable repository policy; even that would prove policy compliance, +not universal safety. + +Another sufficiently capable harness can compose the same Go tools and build +similar mechanisms. Irreplicability and guaranteed superiority over every +model/harness are not engineering requirements. Differentiation must be earned +through reliable composition, useful Go-specific evidence, low interaction +cost, and less engineer effort. Upstream gopls already offers an +[experimental MCP server and workflow instructions](https://go.dev/gopls/features/mcp); +it is a baseline to compare with and a semantic provider to build upon. ## End goal and observable workflows -The core workflow is: +The complete workflow is: ```text -context -> edit -> refresh -> verify -> inspect evidence -> reconsider or continue +understand -> implement -> inspect diagnostics/failures -> refresh + -> verify -> inspect applicable evidence -> revise or hand off ``` -A coding agent should be able to answer four questions from compact, -source-grounded context: what matters for this change, what an implementation -must satisfy, which existing implementations and tests provide examples, and -what needs reconsideration after an edit. +The agent and engineer retain the implementation decision. Agentic-Go provides +the facts and bounded operations that decision requires. Context never runs +verification implicitly, and ordinary navigation requires no Change Contract. -| Workflow | Required behavior | +| Workflow | Required experience | | --- | --- | -| Enter unfamiliar code | Resolve a query, reference, or position into declarations and relevant relationships; expose ambiguity before expansion. | -| Implement new code | Select a file, package directory, or interface and obtain obligations, typed API usage, and existing tests or examples with explicit selection reasons. | -| Change an API | Relate changed declarations to method sets, direct call sites, related tests, and conservative package impact. | -| Continue after an edit | Refresh the previous selection with current locations and evidence differences, without mandatory Change Contract setup. | -| Verify a change | Request executed verification explicitly and identify the snapshot to which that evidence belongs. | - -The historical private 20-run Luna feasibility pilot and 27-run adoption -follow-up record workflow and instruction-use observations. A new private -20-run discoverability pilot qualified all runs and kept them scope-safe, but -its 10 focus runs made no `go_context` calls. A 12-run integrated adoption -follow-up also qualified all runs and kept them scope-safe; all six integrated -runs discovered the shipped skill, used `go_context`, refreshed after editing, -and used the evidence. The integrated arm combines the skill and MCP surface. -These results do not establish MCP-alone causality, improved correctness, -productivity, token efficiency, speed, statistical significance, or broad -model generalization. External and multi-model evaluation remain pending. A -review of all six integrated traces found no transcript proving that context -improved the necessary code edit. In gRPC run 3, refreshed context prompted -broader `./...` verification, which was incomplete due to the output cap; -focused verification later passed after a stale-snapshot rejection. Client-go -runs made 3-4 context calls each and gRPC runs made 4-5, including ambiguous -or unhelpful initial selections and repeated verification after incomplete or -stale results. The bounded guidance improvement is to narrow ambiguity with -returned current candidates, finish each batch of edits and formatting before -refreshing, and refresh again after further edits or stale-snapshot rejection -while avoiding redundant refreshes when the snapshot is unchanged. This trace review does -not establish causal edit-quality or product value. The historical adoption -report is in -[validation/v1.0.0/adoption-results.md](../validation/v1.0.0/adoption-results.md). - -A post-guidance regression comprised six integrated Luna runs, with three -repetitions on each of two scenarios. All six qualified, passed acceptance, -had zero scope violations and zero operator interventions, and used -`go_context`, refreshed after edits, and recorded evidence use. Five focus -calls failed: three client-go calls reported `invalid_input` for invalid -symbol references, and two calls in grpc-go run 1 reported `stale_snapshot` -because an observed semantic location was absent from the snapshot manifest; -grpc-go runs 2 and 3 had no failed focus calls. Workflow adoption remained -possible while the failures exposed categories to investigate. The result does -not establish that guidance improved quality, correctness, productivity, -speed, or product value. The failure audit classified the three client-go -invalid-input calls as malformed or reconstructed Symbol Refs and the two -gRPC stale calls as workspace-symbol locations outside the active observation -manifest. Such locations are now omitted with bounded uncertainty; changed or -missing manifest entries still fail closed as stale. External and multi-model -evaluation remain pending. - -The follow-up six-run integrated Luna rerun passed qualification and acceptance -in every run, with zero scope violations, zero operator interventions, zero -failed focus calls, and complete refresh and evidence-use signals. Median -duration was 411,537 ms and median tool calls were 31. This remains diagnostic -regression evidence, not a claim of improved quality, speed, or product value. - -Keep Go-only, local, deterministic operation; pinned gopls; source provenance; -explicit uncertainty; and existing containment and guarded-refactor guarantees. -The intelligence implementation owns context selection. gopls supplies semantic -facts, and MCP and CLI deliver the result. A passing verification report remains -executed evidence, never a safety verdict. +| Enter unfamiliar code | Resolve queries, files, packages, or current declarations into bounded context; expose ambiguity and reasons for each inclusion. | +| Implement a feature | Return relevant method sets, implementation relationships, typed API usages, tests/examples, and scoped repository guidance. Distinguish observed patterns from requirements. | +| Change an API | Connect changed declarations to supported interface relationships, direct call sites, tests, and conservative package impact. | +| Debug and revise | Make current diagnostics, executed failures, and useful source locations inspectable; expose unsupported evidence and an actionable recovery path. | +| Continue after edits | Refresh the selection, issue current references, and reassess verification applicability. Missing evidence is not proof of source deletion. | +| Verify | Explain selected scope and checks, execute them explicitly, and distinguish pass, failure, omission, and incomplete execution. | +| Review or resume | Present current change scope, report applicability, executed checks, findings, and limitations without requiring the engineer to replay the agent conversation. | + +First make one complete job dependable: change an existing API or interface +across packages, migrate relevant consumers, debug the change, verify it, and +hand it back. Adding cancellation support is a representative case. Type facts +and existing tests can guide that change; they do not prove cancellation +propagates correctly or that every external consumer was found. + +Use the existing implementation before inventing capabilities. Declaration, +file/package, typed-relationship, full-replacement refresh, guarded-refactor, +and verification foundations already exist. The delivery cycle below names +the remaining integration work and proposals. + +## Skills and interaction requirements + +Tools and skills form one workflow contract. Maintain concise shared guidance +with only the installation/placement differences needed by supported hosts. +Do not encode a particular model's conversational quirks in the engine. + +- Prefer a query, file, or package when exact coordinates are unnecessary. + Return current candidates and useful follow-up arguments from actual results. +- Copy Symbol Refs exactly. Never invent, decode-and-rebuild, or modify them. + The current position selector requires the declaration identifier; a line + start, comment, or local variable is not an equivalent selection. +- Recover from ambiguity using a current candidate or a fresh selection. + Explain selector errors without disguising them as usable evidence. + Repeated misuse is an interface problem to investigate as well as a correct + rejection; it is not automatically a provider defect. +- Finish a batch of edits and formatting, then refresh with `base` and + `previous_pack_id` only. Refresh again after further edits or stale + rejection. Avoid repeating unchanged-state context calls. +- Treat impact as planning evidence. It does not authorize editing every + affected package or exceeding the task's supplied scope. +- Inspect failures and missing evidence before continuing. Skills must not + equate a passing requested check with task completion or bypass strict + rejection to keep a workflow moving. +- Make useful evidence and limits available in text as well as structured + responses. Keep output small enough for limited-context clients without + concealing uncertainty or requiring private implementation knowledge. +- Permit skipping context compilation for trivial, familiar changes or when + current evidence already answers the question. Appropriate non-use is valid. + +Broader fresh-position resolution is a possible ergonomic improvement, not a +license to snap an invalid selector to a guessed symbol. Resolve its semantics +and compatibility explicitly before implementation. Existing guarded edits +retain exact preimages and approval boundaries; the external agent authors +feature code with its own editor. ## 1. Documentation and implementation continuity @@ -116,12 +150,13 @@ Historical release evidence remains unchanged. ## 2. One useful context interface -Add the read-only `go_context` MCP tool and the equivalent +Maintain the read-only `go_context` MCP tool and equivalent `agentic-go context --format text|json` CLI command over the same intelligence -implementation. Introduce the separate `agentic.focus/v1` response contract. -Existing v1 inputs, schemas, and behavior remain compatible; register the new -tool as an explicit additive interface and update inventory expectations -deliberately. MCP and LSP types stay outside the intelligence domain. +implementation. Their post-v1 contract is `agentic.focus/v1`. The frozen v1 +registry remains 14 tools, seven resources, one resource template, and six +prompts; `go_context` is the existing additive fifteenth tool. This plan adds +no MCP operation or schema. MCP and LSP types stay outside the intelligence +domain. ### Selection @@ -129,19 +164,21 @@ deliberately. MCP and LSP types stay outside the intelligence domain. | --- | --- | | Symbol query | Bounded candidates with package, receiver, declaration kind, and location. | | Current Symbol Refs | Focused context interoperating with existing search results. | -| Source positions | Enclosing declarations and relevant relationships. | +| Source positions | Context for the current declaration identifier. More general enclosing-declaration resolution remains a separate design decision. | | Workspace-relative paths | File or package orientation and implementation examples. | -| Changes against a local base | Changed declarations, structural consequences, callers, and related tests. | -| No selection | Compact workspace orientation. | +| Changes against a local base | Current change/impact evidence. Automatically offering focused declaration candidates from that diff is proposed in the next delivery cycle. | +| No explicit selector | The current base-selected change view; use existing brief/file/package entry points for orientation. | Reuse existing package scope, contained paths, and one-based UTF-8 byte -locations. Multiple explicit anchors share one observation. Ambiguous queries +locations. Accept one mutually exclusive selector group. File and package +selection may gather several anchors within one observation. Ambiguous queries return candidates for selection; a candidate is not silently chosen. Accept a response-byte budget and an optional previous pack ID. A previous pack -supplies its original selection when no new selection is given. A changed -selection produces a full pack. Ordinary navigation requires no Change -Contract. Free-form goals remain explanatory context, not executable semantics. +supplies its original selection; refresh does not accept changed selectors. +Choose a fresh request for a new selection. Ordinary navigation requires no +Change Contract. Free-form goals remain explanatory context, not executable +semantics. ### Evidence and rendering @@ -241,7 +278,7 @@ existing v1 tool error contracts. ## 5. Coherent observation and bounded derived state -Introduce one request-scoped observation consumed by discovery, source +Preserve the request-scoped observation consumed by discovery, source extraction, context assembly, and semantic synchronization. Carry captured source, package/build inputs, and snapshot identity through internal helpers instead of recursively recapturing state. Reuse package discovery within the @@ -269,6 +306,14 @@ relevant derived state. Concurrent requests must not combine evidence from different observations. Preserve shared concurrency limits and cancellation; do not introduce unsafe provider concurrency to increase throughput. +Snapshot identity is not execution isolation. Current verification runs Go +commands against the live workspace and validates state before and after the +operation. Require a stable worktree during checks; endpoint equality cannot +exclude transient A-to-B-to-A edits during execution. Stronger support for +concurrent writers needs a separately demonstrated execution/coordination +design. External services and unrecorded runtime inputs are not made immutable +by a source snapshot. + ## 6. Explicit refresh of delivered evidence Store the original selection and a manifest of the evidence actually delivered @@ -276,9 +321,10 @@ with each private pack. Reuse the private artifact infrastructure and its containment and permissions. A refresh reads retained comparison metadata; it does not make old snapshot-bound artifact cursors valid against new source. -An explicit previous pack ID refreshes that selection against a new observation. -A changed selection or incompatible build/scope context produces a full pack. -Separate three concepts: +An explicit previous pack ID refreshes that selection against a new observation +and returns a complete replacement. Changed selectors require a fresh request; +preserve existing build/scope validation and rejection semantics. Separate +three concepts: | Concept | Meaning | | --- | --- | @@ -286,26 +332,18 @@ Separate three concepts: | Content revision | Whether the evidence for that item changed. | | Current locator | Where the item exists in the new snapshot. | -Report added evidence, replacement content for changed evidence, confirmed -source removals, updated locations and current references, previously delivered -evidence omitted by selection or budget, and relationships that became -unavailable. Unchanged content may refer to previously delivered evidence while -its locator and snapshot-bound references are updated. - -Compare against the previous delivered response, not its complete internal -overflow artifact. Leaving a response must never be mistaken for deletion from -source. Confirm a removal from current observation evidence; inability to -resolve or examine an item is not confirmation of deletion. +Return self-contained current evidence with updated locations and references, +explicit omissions, and unavailable relationships. Retained metadata describes +the evidence actually delivered, not an undelivered internal overflow artifact. +Leaving a response must never be mistaken for deletion from source. Inability +to resolve or examine an item is not confirmation of deletion. Old Symbol Refs remain stale. Refresh explicitly resolves current identities and issues current references. Ambiguous renames or moves require selection; -expired packs require a fresh request. Normal calls remain self-contained, and -delta delivery requires an explicitly supplied prior pack. Reconnection or -switching agents must not silently omit needed context. - -The reconstruction invariant is: applying a delta to the previous delivered -pack reconstructs the current selected pack, including updated locators, -omission state, and uncertainty. It does not reconstruct unexamined evidence. +expired packs require a fresh request. Reconnection or switching agents must +not silently omit needed context. Full replacement is the current policy. +Per-evidence semantic differences and delta delivery are deferred; neither is +an acceptance requirement for the next workflow. ## 7. Change consequences and verification @@ -324,89 +362,149 @@ evidence with its observed snapshot so later edits cannot make an older passing report appear current. Preserve verification result semantics, conservative package selection, and the distinction between context and executed evidence. -## Historical adoption evidence and next decision - -The 27-run Luna/max adoption follow-up contains an 18-run canonical three-arm -matrix, six integrated-skill diagnostics, and three scope-wording reruns. -Description-only discoverability produced 0/6 focus use; generic prompt -guidance produced 6/6; and the shipped skill produced 6/6. Initial integrated -safety was 5/6, while the scope-wording rerun was 3/3 acceptance-pass, -qualifying, and scope-safe. These are historical observations, not proof that -the workflow improves task outcomes. - -These records establish observed instruction-surface use and safety only. They -do not support causal speed, token, reliability, adoption, performance, or -generalization claims. Keep focus and full-replacement refresh; defer delta -refresh. Trace review is complete and its bounded guidance improvement is -recorded in the project skill. Keep the studies as regression evidence, not -fresh proof of product superiority. External and multi-model evaluation remain -pending. Defer any new release label until a reliability milestone passes. -Raw artifacts remain private and ignored. - -## Delivery order and completion criteria - -| Stage | Deliverable | Completion criterion | +A review handoff must present outcome and applicability independently. Include +the observed change, relevant build/scope/check policy, checks actually run, +unexecuted or incomplete checks, findings, and limitations. A stored latest +report may be historical; it becomes evidence for the current change only +through the existing applicability comparison. Use the current reports and +renderers before proposing another artifact or wire format. Free-form intent +and unresolved business questions remain agent/engineer judgments. + +## Evidence boundary + +The [historical adoption report](../validation/v1.0.0/adoption-results.md), +later integrated diagnostics, and the 2026-09-24 canonical Luna matrix are +different campaigns. Preserve their identities and do not pool them. Current +campaign status belongs in the [handoff](continuation/go-intelligence.md). +None establishes broad model compatibility or comparative engineering value. + +The current matrix establishes guided use and successful task acceptance on +two pinned scenarios. Its failed selectors were rejected correctly; audited +selector misuse does not establish a provider defect. Evidence-use counters +record event ordering, not improved decisions. Pooled duration medians across +heterogeneous tasks do not isolate the cost of the tooling. Unmeasured value +is unknown, not zero; equal final correctness may still conceal differences +in investigation, review, or rework. + +## Next delivery cycle + +Keep this cycle focused on the cross-package API/interface change. These are +ordered implementation proposals after the current reliability work, not +completed features or permission to run model campaigns. + +| Order | Deliverable and status | Completion criterion | | --- | --- | --- | -| 1. Documentation | Revised north star, handoff, authority clarifications, and index links | Another engineer can recover the direction and current status without conversation history. | -| 2. Coherent observation | Shared discovery, captured source, safe cache identities, and active-manifest lifetime | Relevant helpers reuse one observation while preserving validation and cancellation. | -| 3. Useful context | MCP/CLI entry points, query/ref/position selection, excerpts, direct relationships, and useful text | One request provides grounded context to begin a focused edit; ambiguity and budgets are explicit. | -| 4. Implementation support | Path/base selection, richer Go relationships, existing examples, and incomplete-source behavior | The agent can inspect implementation obligations and examples together, including explicit unavailable facets. | -| 5. Refresh | Delivered-evidence manifests, current locators, replacements, removals, and omission distinctions | Refresh communicates what must be reconsidered after an edit and satisfies the reconstruction invariant. | -| 6. Integration | Verification lineage, compatible discovery, documentation, and client guidance | The complete workflow works without mandatory continuity setup and preserves existing v1 behavior. | - -Each stage is one coherent implementation slice. Do not build all later stages -while introducing shared observation. Record consequential implementation -decisions and the evidence actually gathered in the handoff. - -## Focused acceptance cases for future implementation - -- Ambiguous names, identical methods on different receivers, embedding, - aliases, and generic instantiations. -- Direct test references, external test packages, and unresolved interface - dispatch; each extraction predicate includes a meaningful near miss. -- Temporarily invalid source, unavailable semantic facets, and the distinction - between missing evidence and examined empty results. -- Tight budgets, large declarations, multiple anchors, Unicode byte locations, - and preserved provenance and omission markers. -- Same-size rewrites, edits during observation, deletion, module/build changes, - cancellation, sidecar restart, concurrent requests, and cache pressure. -- Refresh across shifted locations, changed content, deletion, ambiguous - renames, expiry, and evidence omitted by budget. -- Reconstruction of a current selected pack from its previous delivered pack - and delta, including current references and unavailable relationships. -- Equivalent evidence across CLI text, CLI JSON, and MCP, accounting for - rendered budgets and deliberate compatibility checks for existing v1 tools. - -These cases specify future focused correctness work. This documentation change -introduces no new tests or benchmarks. Complete the agreed repository gates and -scope checks before committing. - -## Boundaries and implementation decisions - -- The local north-star document is the planning target. No separate Notion - document was found in the repository documentation reviewed for this plan. -- Preserve Go-only, local, deterministic operation, pinned gopls, stdio MCP, - contained access, disk snapshots, and explicit stale rejection. -- Retain the existing dependency stack and private artifact infrastructure. - Unsaved editor overlays, hosted services, and general graph infrastructure - remain outside this plan. -- Additional languages, an embedded LLM, an agent framework, autonomous - editing, broad analyzer expansion, and speculative test selection are not - introduced by this work. -- No broad comparative evaluation, paid model run, or benchmark campaign is a - release gate. The private Luna focus and adoption follow-up are recorded - evidence only and do not establish causal engineering improvement. Local - Conventional Commits are authorized after the agreed gates and scope checks - pass. Pushes, tags, and releases require separate authorization and must - preserve existing public history. -- The documentation stage and the implemented observation, context, focus, and - verification slices are recorded above. Continue only from the current - handoff next action with an explicit bounded implementation slice; later - roadmap rows are not complete merely because they are listed here. - -The behavioral decisions above govern future implementation. Exact new wire -fields, cache capacities, serialized budget accounting, and typed extraction -predicates must be recorded alongside their implementation contracts before -those stages ship. Resolve them against the existing constraints and record -their rationale in the handoff. They are not frozen or verified by this -documentation change. +| R0 | Selector guidance, private cause classification, and recovery replay. Remediation is in progress. | Malformed refs, invalid coordinates, genuine provider errors, stale results, and recovery remain distinguishable. Finish the recorded reliability gates and reruns; do not relabel failed calls as successes. | +| R1 | Offer current declaration candidates from the observed diff. Proposed. | A base-selected change request can lead to relevant focused evidence without invented coordinates. Preserve explicit selector behavior; bound and explain candidates, retain ambiguity, and mark deleted/unresolvable declarations unavailable. Resolve compatibility and representation in existing fields before editing code. | +| R2 | Complete the edit/debug/verify/review handoff using current evidence and renderers. Proposed integration work. | The engineer can inspect the current change, check scope, execution result, applicability, findings, and omissions together. A subsequent edit makes old evidence visibly non-applicable. Agent judgment is distinguishable from machine facts. | +| R3 | Exercise the shared skill/workflow in real clients and different model families. Planned. | Text/structured delivery, exact reference use, refresh, failure recovery, and handoff work in the named combinations, including a usable lower-capacity open model and a stronger model. Record unsupported cases instead of claiming universal compatibility. | +| R4 | Decide whether the complete workflow earns repeated use. Planned. | Compare against equipped alternatives on fresh tasks and observe engineer effort. Continue only for a recurring benefit; simplify or stop expanding capabilities that repeatedly add work without useful outcomes. | + +R1 and R2 should reuse the typed relationships, tests/examples, impact engine, +refresh, and applicability checks already present. A feature list is not a +reason to rebuild them. Any design that requires a new public field or changes +a frozen behavior must return to an explicit compatibility decision; this +cycle does not pre-authorize that expansion. + +### Acceptance cases for the complete workflow + +- Begin before any diff exists using an interface, query, file, or package; + after editing, use current changed declarations to continue investigation. +- For an API migration, expose at least one source-supported implementation or + caller obligation and relevant existing tests/examples. Validate those facts + independently; the presence of output alone is not usefulness. +- Preserve pointer/value method sets, embedding, generic/type facts, internal + and external test imports, and scoped repository guidance where supported. + State unavailable relationships and excluded build configurations. +- Recover from temporary compile errors using current source and diagnostics; + never substitute earlier type facts as current. Do not guess business logic. +- Reject altered refs and invalid coordinates, then recover through an exact + current candidate or fresh file/query selection. Record recovery effort. +- After an edit, formatting pass, or relevant build/module change, refresh and + reassess the report. A historical pass is not current evidence; an applicable + failure remains a failure. Missing race checks or incomplete output remain + visible under the requested policy. +- Keep provenance, currentness, essential uncertainty, and next actions usable + under a small response budget. A host receiving only text must not receive + a misleadingly stronger conclusion than a structured-content consumer. +- Permit a routine local edit to bypass unnecessary context work. Allow a + different supported agent to resume from current evidence without reusing + stale refs; private continuity remains bound to its documented environment. + +Retain existing stale-state, cancellation, containment, interruption, and +schema evidence. New focused checks should cover the actual behavioral change, +not duplicate the existing corpus. A capability described here is complete +only after its implementation and applicable evidence are recorded. + +### Finite product evaluation + +Keep historical corpus/scorer contracts unchanged. Start a separately named +screening campaign only when its clients, permissions, and budget are available. +Preselect four tasks across at least three repositories: two ordinary Go +changes, a cross-package migration, and a controlled stale/incomplete-evidence +case. Keep challenge results separate from ordinary-work results. Define +acceptance and review obligations before observing treatment outcomes. + +Use two model families in one host for the first screen, including a usable +smaller open model and a stronger model when available. Compare three conditions: + +1. Native editing, Go tools, and upstream gopls with its workflow guidance. +2. The same baseline with Agentic-Go available through its normal descriptions. +3. The same baseline with Agentic-Go and concise operational skill guidance. + +All conditions receive equivalent task requirements, outcome-oriented workflow +instructions, permissions, and budgets. Availability measures the delivered +integration; the guided condition additionally measures instruction effects. +This is 24 screening runs, not a statistically conclusive efficacy study. +Randomize order and block comparisons by task, model, and initial repository +state. A missing model/client is a coverage gap, not permission to substitute +Luna results and label them model-independent. + +Blind-review candidate changes and standardized handoffs using independent +behavioral and review criteria. Hidden oracles must not enter instructions or +tool guidance. Substring matches and tool-call counters cannot establish +obligation satisfaction. Inspect traces separately to explain a result. + +Measure accepted changes, consequential omissions, engineer review/rework +effort, setup friction, total time, tool execution time, retries, and real +token/cost data when available. Bytes are not tokens, calls are not inference +steps, and total elapsed time is not tool overhead. Report per-task/model +results and dispersion; do not present pooled median ratios as causal effects. +Account for recurring operational cost as well as time saved in review. + +Use the screen to choose one concrete improvement, then confirm that hypothesis +on fresh held-out tasks and a second host before widening claims. Observe real +Go engineers on subsequent ordinary changes without requiring tool use. Their +choice to keep the workflow and concrete saved work are useful evidence; +arbitrary retention fractions or invented product scores are not proof. + +Continue when independently checked benefits recur across different changes +and repositories without an unacceptable correctness or interaction regression. +Set acceptable cost and the primary outcome before a campaign, based on the +engineer's task. If a capability repeatedly adds effort without improving +decisions or reviewability, make one bounded redesign around the observed cause; +stop expanding it if that fails. A correct baseline alone is never a kill +criterion. Benchmark growth is not a substitute for product improvement. + +## Boundaries and release decisions + +- Preserve Go-only, local operation, pinned gopls, stdio MCP/CLI, contained + access, disk snapshots, and strict stale rejection. Keep the frozen v1 + inventory, existing additive focus interface, and schema semantics intact. +- Retain the dependency stack and private artifact infrastructure. Delta + refresh, general caches/graphs, speculative individual-test selection, + broader analyzer domains, and expanded refactoring await demonstrated need. +- Unsaved editor overlays, distributed agent coordination, hosted services, + additional languages, an embedded model, and autonomous repair are outside + this cycle. Guarded supported edits remain explicitly requested operations. +- Source facts and selected checks do not establish behavioral equivalence, + complete runtime reachability, guaranteed cancellation, or universal safety. +- Reliability maintenance can ship after its applicable engineering and + release gates. Comparative usefulness, model generalization, and default + workflow placement require their own evidence; a maintenance release must + not imply those claims. No giant benchmark campaign is a blanket release + prerequisite. +- This revision changes documentation only. Continue implementation when the + user requests it; that request is sufficient authorization for its bounded + scope. Paid runs, external outreach, commits, pushes, tags, and publication + require their respective authorization. Preserve existing public history. diff --git a/docs/plan.md b/docs/plan.md index cedad4a..6306aae 100644 --- a/docs/plan.md +++ b/docs/plan.md @@ -1,169 +1,153 @@ # agentic-go product plan -## Product thesis - -Agentic-go is source-grounded Go change intelligence for coding agents. Its -currently shipped workflow is language-native change verification: - -> Given a local base and the final worktree, explain what changed, what may be -> affected, what evidence was executed, which findings appear introduced, and -> what remains uncertain. - -Go is the reference implementation. The durable product boundary is the -versioned verification report, not MCP and not a count of tools. The CLI, -GitHub Action, and MCP server are adapters over one engine. Agentic-go remains -deterministic developer tooling; it embeds no LLM and performs no agent -orchestration. - -The personal v1 module path is `github.com/ashwingopalsamy/agentic-go`. A later -`github.com/agentic-mcps/go` repository is an independent module identity with -an explicit migration, not a transfer or alias. - -## Release authorities - -- [`v0.2.0-release-scope.md`](v0.2.0-release-scope.md) is the executable v0.2 - specification. -- [`v0.1.0-release-scope.md`](v0.1.0-release-scope.md) is the compatibility - baseline. -- [`contracts.md`](contracts.md) owns shared protocol and execution - invariants. -- [`v1.0.0-roadmap.md`](v1.0.0-roadmap.md) owns the staged v0.3 through v1 - direction without retroactively changing the v0.2 contract. -- [`../CONTEXT.md`](../CONTEXT.md) defines the domain language. -- [`adr/0001-verification-report-boundary.md`](adr/0001-verification-report-boundary.md) - records why the report is the durable boundary. - -Broader phase documents are design material only. They do not add release -scope merely because a possible tool or rule is described there. - -## Architecture - -```text -Change Request - -> Change Snapshot - -> Affected Package Closure - -> Verification Plan - -> Executed Evidence - -> Verification Report -``` - -- `internal/verification` owns portable types, policy, orchestration, and - report assembly. -- `internal/changeimpact` owns Go, Git, module, package, declaration, and diff - discovery behind the verification interface. -- workspace, execution, parser, audit, and analysis packages are infrastructure - adapters. -- `cmd/agentic-go`, the root Action, and MCP tools adapt the same report. MCP - types, workflow concepts, and adapter names do not enter the engine. - -The report began at `agentic.verify/v1alpha1` for v0.2 and is frozen as -`agentic.verify/v1`. Go-specific entities use -namespaced kinds such as `go.package`; top-level concepts remain portable so a -future TypeScript implementation can produce the same semantics. Extraction -into an organization-level specification waits until a second implementation -exists and proves the common boundary. - -In v0.7 the CLI and MCP adapters invoke the same unified report path. The -report adds semantic and compiler diagnostics, exact snapshot lineage, -bounded context and refactor provenance, provider capabilities, and optional -Change Contract compliance. These additions preserve the existing result and -exit-status semantics. - -## v0.1 trust seed - -The compatibility baseline provides seven stdio MCP tools, four resources, -four prompts, and `agentic-go-vet`. Its active concurrency and error rules were -calibrated against a pinned ten-repository corpus. That evidence is -corpus-specific, not a universal precision guarantee. - -Those analyzers remain the only v0.2 policy-finding domains. A new or changed -predicate requires a positive fixture, a meaningful near miss, a documented -limitation, production-path coverage, reviewed external findings, and an -acceptable false-positive rate. - -## v0.2 usage seed - -The primary workflow is: - -```sh -agentic-go verify --base origin/main -``` - -It provides: - -- a final-worktree snapshot covering committed, staged, unstaged, renamed, - deleted, and untracked changes; -- changed Go declarations and module/workspace/embed metadata; -- directly changed packages plus transitive reverse importers within scope; -- one whole-package test and changed-statement coverage run, with optional race - detection; -- base/current comparison for calibrated analyzer findings; -- source-grounded risk facts and targeted review guidance; -- explicit uncertainty for generated code, build constraints, cgo, external - consumers, generated inputs, and unmodelled non-Go behavior; and -- a deterministic report with `pass`, `findings`, or `incomplete` automation - status. `pass` is never a safety verdict. - -Delivery order is CLI first, a thin advisory GitHub Action second, and one -approval-aware MCP operation for coding agents third. The existing seven MCP -tools remain compatible, but clients should use `go_verify_change` when they -want the complete workflow. - -Selective tests, call-graph reachability claims, SSA/VTA, `test_regex`, SARIF, -HTTP, `doctor`, automatic toolchain installation, and Windows support claims do -not ship in v0.2. - -## Evidence before publication - -The v0.2 release record must contain: - -- golden contract reports; -- local CLI and ephemeral stdio MCP dogfood against agentic-go; -- three reviewed historical changes from already-cloned projects showing - reverse impact, changed coverage, and analyzer baselining; -- commands, pinned commits, timings, limitations, and observed usefulness; -- the Go 1.25/1.26/1.27 release matrix, race/vet/build/static analysis, - four-target cross-builds, release configuration checks, signatures, history, - and a clean worktree. - -This is self-serve implementation evidence, not an adoption or product-market -fit claim. - -## Expansion rule - -Security, observability, API design, naming/maintainability, and performance -remain relevant review lenses. In v0.2 they report only change-grounded facts -and guidance. They become analyzers only after repeated repository evidence -shows that an actionable defect class can be detected precisely enough to pass -the same calibration gate as the existing rules. - -Navigation, profiling, build analysis, fuzz orchestration, and other ideas are -also proposals, not a promised catalog. Expansion follows demonstrated user -pain and retained workflow value; it does not aim at a predetermined MCP tool -count. - -## Multi-language direction - -Future repositories may use paths such as `github.com/agentic-mcps/go` and an -equivalent TypeScript package, but each language implementation owns its native -change discovery and evidence execution. They share only semantics proven -portable in practice: snapshots, impacted units, checks, evidence, findings, -risks, uncertainty, and policy status. - -The personal Go module remains `github.com/ashwingopalsamy/agentic-go` for its -v1 release. A later `github.com/agentic-mcps/go` implementation starts from the -same source lineage but remains a separate repository and module identity. The -two paths are not interchangeable and no automatic mirroring is implied. - -## v1 direction - -The v0.2 compatibility authority remains -[`v0.2.0-release-scope.md`](v0.2.0-release-scope.md). The implemented product -direction is [`v1.0.0-roadmap.md`](v1.0.0-roadmap.md): -source-grounded Go change intelligence combining semantic navigation, compact -context, persistent change continuity, guarded deterministic refactoring, and -verification with explicit provenance and uncertainty. - -The roadmap was staged to prove useful workflows and reliable contracts before -v1. The implemented surface and evidence do not establish a universal -model-reliability or token-saving claim. +Revised 2026-09-24. This is the routing summary for the +[Go engineering north star](go-intelligence-north-star.md). +The [continuation handoff](continuation/go-intelligence.md) records what is +implemented, the current evidence, and the next unfinished step. + +## Customer and product thesis + +Build for Go engineers using coding agents on real repositories. Help the +agent understand, implement, debug, refresh, and verify a change, then leave +an inspectable handoff. Reduce the engineer's investigation, avoidable rework, +and effort to determine whether the evidence still applies. + +The technical foundation is a local, deterministic evidence compiler for Go. +It combines semantic facts, change impact, bounded context, guarded operations, +and executed verification. Skills explain when and how to use those operations. +The external agent authors feature code; the engineer retains judgment about +requirements and completion. + +The product should reduce dependence on model memory and guesswork. It cannot +guarantee that every model reasons equally well or that another capable harness +cannot reproduce its mechanisms. Reliable integration and recurring saved work +are sufficient sources of value, if demonstrated. + +## Existing foundation + +- `internal/intelligence` owns observations, focused context, typed + relationships, refresh, verification applicability, continuity, and guarded + refactoring. +- `internal/changeimpact` discovers changed declarations and conservative + affected packages from Go, Git, module, workspace, and embedded-file inputs. +- `internal/verification` owns check policy, execution planning, and portable + evidence reports. +- The CLI, advisory GitHub Action, and MCP adapters expose the same domain + behavior. MCP/LSP transport types stay outside intelligence domain contracts. +- Current focus supports query/ref/position/file/package selection, + source-supported Go relationships, full-replacement refresh, and report + applicability. Current verification distinguishes requested checks passing + from findings or incomplete execution. + +These mechanisms exist. Their usefulness across the complete workflow and +different model/client combinations remains to be established. The durable +verification boundary is `agentic.verify/v1`; focused context uses the separate +`agentic.focus/v1` contract. Neither contract proves business correctness. + +## First complete workflow + +Start with a cross-package API or interface change: find relevant declarations, +implementations and consumers, inspect existing examples/tests, edit and debug, +refresh, verify affected packages, and hand the evidence back to the engineer. + +An interface migration or cancellation-support change should expose Go-specific +facts such as pointer/value method sets, embedding, typed usages, test imports, +and build assumptions where supported. Their limits remain visible. A compiler +already catches many signature errors; measure whether this workflow reduces +discovery/repair cycles, consequential omissions, or review effort. + +The next cycle is ordered in the [north star](go-intelligence-north-star.md#next-delivery-cycle): + +1. Finish the existing selector-remediation reliability work and its recorded + checks/reruns. Preserve cause classification and strict rejection. +2. Design current declaration candidates derived from the observed diff so an + agent can enter focused context without guessing coordinates. Reuse existing + fields only after resolving compatibility and budget behavior. +3. Complete the edit/debug/verify/review handoff using existing reports, + applicability assessments, and renderers. A historical pass must remain + distinguishable from evidence applicable to the current change. +4. Exercise shared instructions, text/structured delivery, exact refs, and + recovery in named real clients with different model families. +5. Use a finite comparison and real engineer use to decide which capability + deserves further investment. + +Items 2-5 are planned product work. Updating this plan does not implement them, +authorize paid runs, or qualify a release. + +## Model and skill contract + +Keep engine semantics independent of model identity. Maintain concise shared +workflow instructions, with thin host-specific installation and placement. +Support claims must name the combinations actually exercised, including a +usable smaller open model and a stronger model before making broader claims. + +Selectors and opaque references must be usable without guessing. Recover from +errors with current candidates or fresh queries/files; do not correct refs +silently or accept stale evidence. Refresh after an edit batch with `base` +and `previous_pack_id` only. Repeated mistakes count against interface quality, +even when rejection is correct. + +Context helps plan scope and checks; it neither authorizes broader edits nor +executes verification. Skills must expose uncertainty and passing-check limits. +Permit trivial/familiar edits to skip unnecessary context work. A client must +deliver useful evidence to the model, not merely complete an MCP handshake. + +## Evidence and product decisions + +Compare against native Go tools and upstream gopls with useful workflow +guidance. Keep the existing canonical adoption matrix and older diagnostic +campaigns separate. They record bounded use and task acceptance, not broad +compatibility or comparative engineering value. + +Measure independently accepted changes, consequential omissions, engineer +review/rework effort, setup and interaction cost, and per-task/model time. +Tool use and event ordering are diagnostics. Equal final correctness can still +leave meaningful differences in effort; a passing baseline is not a kill rule. + +The north star specifies a small initial screen, followed only by a focused +redesign and confirmation on fresh tasks/another host when warranted. Keep +hidden acceptance oracles out of treatment instructions. Record unsupported +clients and missing cost data honestly. Continue capabilities with recurring +benefit; simplify or stop expanding those that repeatedly add work without +improving decisions or reviewability. + +The separate [codebase indexing research note](research/codebase-indexing-retrieval.md) +records the user-directed aspiration for reusable, branch-aware repository +retrieval. Its next product action is to evaluate the current local retrieval +path before proposing a persistent index. This research does not change the +current release scope or product contracts. + +Reliability maintenance and comparative product claims have separate gates. +A maintenance release can satisfy its engineering contracts without claiming +that it makes models better. Evidence for a narrow task/model must remain a +narrow claim. + +## Authority and compatibility + +- [North star](go-intelligence-north-star.md): current product requirements, + model/skill contract, delivery cycle, and value evaluation. +- [Continuation handoff](continuation/go-intelligence.md): current state, + evidence identities, and sequencing. +- [v0.9 freeze](v0.9.0-release-scope.md) and [contracts](contracts.md): + frozen interfaces and shared invariants. +- [v0.2 scope](v0.2.0-release-scope.md) and + [v0.1 scope](v0.1.0-release-scope.md): compatibility baselines. +- [v1 roadmap](v1.0.0-roadmap.md): completed stages and historical evidence. +- [v0.8 evaluation scope](v0.8.0-evaluation-scope.md): historical corpus, + scorer, replay, and paid-pilot boundaries. +- [Verification ADR](adr/0001-verification-report-boundary.md) and + [Context Pack ADR](adr/0002-context-pack-boundary.md): architectural rationale. + +The frozen v1 registry remains 14 tools, seven fixed resources, one template, +and six prompts. The existing additive `go_context` brings the server to 15 +tools. This plan preserves those surfaces and current schemas. + +Require a stable worktree during verification. Source snapshots and endpoint +validation are not transactional execution isolation. Preserve contained access, +cancellation, bounds, guarded edit preimages, and explicit uncertainty. + +The module is `github.com/agentic-mcps/go`; the former personal module is a +separate identity with an explicit [migration](module-migration.md). +The current cycle is Go-only. Delta refresh, broad graphs/caches, additional +analyzer domains, speculative test selection, autonomous repair, distributed +agent orchestration, hosted analysis, and other languages remain outside it. diff --git a/docs/research/codebase-indexing-retrieval.md b/docs/research/codebase-indexing-retrieval.md new file mode 100644 index 0000000..0801f36 --- /dev/null +++ b/docs/research/codebase-indexing-retrieval.md @@ -0,0 +1,502 @@ +# Branch-aware codebase indexing and retrieval + +Status: user-directed product aspiration and research proposal, recorded +2026-09-27. This document captures the requested direction and current findings. +It does not change the frozen v1 contracts or claim that full repository +indexing and retrieval exist. + +The pre-integration source worktree diffs, statuses, and untracked-file hashes +are preserved in the +[`2026-09-27 source worktree record`](../../validation/retrieval/provenance/2026-09-27/README.md). +The separate 72-run Sol/high decision-value harness remains parked in its +source worktree and is excluded from retrieval evidence. + +The current release and reliability sequence remains in the +[continuation handoff](../continuation/go-intelligence.md), +[product plan](../plan.md), and +[north star](../go-intelligence-north-star.md). This proposal is a separate +research track. A model-free retrieval screen has run and exposed weak +relevance, zero document retrieval, and an incomplete large-corpus capture. +The exact branch source-view preview is implemented. There is no persistent +index or measured agent-product advantage yet. + +## Product aspiration + +The product owner wants a developer to install Agentic Go, point it at a local +directory or Go repository, and have it build reusable repository knowledge. +The first setup should index the repository's default branch, using `main` when +that branch exists. A developer should also be able to select a development or +feature branch and keep its index separate. + +After setup, an engineer or coding agent should be able to ask where a behavior +is implemented, how a product flow works, which packages enforce an invariant, +or what needs to change for a difficult requirement. The tool should return +current code, tests, documentation, and Go relationships that help answer the +question. Repeated questions against an unchanged branch should reuse the +index instead of reparsing and rediscovering the repository from the beginning. + +The intended gains are faster orientation, less repeated search, lower context +and token cost, fewer missed relationships, and less effort to make and review a +change. The ambition is to make Agentic Go a game changer for people and AI +agents navigating large Go codebases. That is a product goal; it is not a +measured claim. + +The tool currently provides deterministic, source-grounded context to an +external coding agent. It does not embed an LLM or independently guarantee a +correct natural-language explanation of business logic. The index can give an +agent better evidence quickly. The agent still has to interpret that evidence, +and the engineer still judges requirements and behavior. + +## Current findings + +The benchmarked source was commit `7b5111c6365a2a12806a561d86745d5bc50d7c9e` +with an explicit dirty-diff fingerprint in each report. The integration topic +branch preserves two dirty source worktrees' changes. The working tree contains +a private, process-local retrieval cache in +[`internal/intelligence/retrieval/index.go`](../../internal/intelligence/retrieval/index.go). +It parses observed Go files into declaration fragments and ranks them using +deterministic lexical scoring with symbol, receiver, package, and declaration +kind signals. `go_context` resolves candidate source positions through the +active gopls observation. The public CLI/MCP surface and schemas are unchanged. + +This is useful candidate discovery, not the proposed persistent repository +index. The cache targets an estimated 64 MiB cap for retained parsed fragment +metadata in process memory and is lost on restart. That estimate does not bound +actual allocation or files and fragments materialized while searching. Each +query still gathers every observed Go file and scores all declaration fragments. +The current retrieval remains process-local and indexes Go declarations only; +it does not index repository text or build a complete cross-package behavioral +graph. The corrected medium screen and repository-scale coverage limits are +recorded below and in the [continuation handoff](../continuation/go-intelligence.md#current-next-action). + +Exact source freshness remains a core property. Current snapshot capture reads +the scoped inputs twice. It retains at most 32 manifests and 8 MiB of manifest +metadata, plus 4 MiB of captured source per observation. A future index can +avoid reparsing unchanged committed content, but it cannot assume that an +uncommitted worktree stayed unchanged because a file timestamp or watcher was +quiet. The existing snapshot and stale-reference rules remain the authority +for current semantic evidence. + +The current focus interface uses `base` to describe change context. It is not +a branch selector. The preview command now creates a detached Git worktree at +the selected exact commit. With no branch supplied, it selects local `main` +when present, otherwise the configured `refs/remotes/origin/HEAD`; it does not +silently use the currently checked-out branch. The JSON result names the ref, +commit, tree, view path, overlay digest, and checkout limitations. Configure the +agent against that exact returned view path so snapshot validation can reject +a moved branch ref. This is explicit source-view setup, not persistent +indexing or automatic retargeting of an already-running MCP server. + +An optional dirty overlay is accepted only when the source checkout's `HEAD` +equals the selected branch commit. It captures staged, unstaged, and regular +untracked files up to 8 MiB, including additions and deletions; it rejects +cross-branch overlays, merge conflicts, unsupported untracked paths, and +concurrent overlay drift. The detached view does not change the source checkout +branch or files. Git submodules are reported as incomplete. The preview does +not capture external `go.work` or local `replace` inputs, so a complete Git +tree checkout is not a claim that every Go build input is present. Use the view +root itself as `--workspace`; nested workspace roots do not validate the +branch-view marker. + +## Historical GPT-6 Sol council findings + +Three GPT-6 Sol council members examined the existing implementation from +index architecture, retrieval experience, and product/release perspectives. +They initially differed on whether persistent indexing should start now. Their +cross-review converged on measuring the current path before adding persistence. + +**Index architecture.** A local, content-addressed disk index with per-file +segments and bounded top-K retrieval is the likely scale path. Publish complete +generations atomically, and resolve candidates through the exact current +semantic observation. + +**Retrieval experience.** Keep the agent flow progressive: question, ranked +candidates with reasons and coverage, selected source, then bounded related +evidence. Use existing Go relationships for semantic expansion. Consider +embeddings only if held-out natural-language queries expose lexical misses. + +**Product and release.** The current cache has not demonstrated held-out +relevance or large-repository benefit. Compare it with guided `rg`, Go tools, +and upstream gopls before adding storage, invalidation, corruption recovery, or +a new public interface. + +The strongest argument for a disk index is repeated work: the process-local +cache disappears between sessions, and warm queries still walk all observed +Go files and declarations. The strongest argument to wait is that snapshot +capture may dominate the request, so a disk index could make the wrong stage +faster while adding a second freshness and recovery system. + +Council members suggested possible screening values such as a two-to-five +second warm retrieval p95, 512 MiB peak memory, a large-repository stratum of +10,000 Go files or 250 MiB of Go source, and at least a two-times retrieval +improvement from a persistent prototype. These are planning hypotheses, not +measured targets or supported limits. Freeze actual thresholds before running +the evaluation. + +## GPT-6 Luna Max council verdicts (2026-09-27) + +These are the current verdicts and remain separate from the historical +GPT-6 Sol findings above: + +- Preserve both dirty source worktrees while preparing or running research. +- Do not use the historical GPT-6 Sol/high 72-run harness or combine results + from different models. +- First run a model-free held-out retrieval and repository-scale benchmark. + This measures retrieval quality and costs without an agent model. +- Only after exact branch selection and branch-matched source views exist, run + a separate matched native-versus-Agentic-Go engineering screen with GPT-6 + Luna Max only, using the same model settings in both conditions. + +These are sequencing and study-design decisions, not evidence that retrieval +quality, repository-scale performance, branch support, or comparative value +has been demonstrated. + +## Product behavior to design + +### Setup and indexing + +The target first-run flow should: + +1. Resolve and show the repository root, selected branch, commit/tree identity, + Go build configuration, index format, and file coverage. +2. Build the first local index for `main` when present, otherwise use and name + the repository's configured default branch. Do not change the developer's + checked-out branch as a side effect. +3. Let the developer select another branch. Keep its index generation separate + from `main`, and identify results by the resolved commit/tree rather than + relying on a mutable branch name alone. +4. Track staged, unstaged, and untracked worktree changes as an exact overlay + over the selected branch. A lost or overflowed filesystem watcher requires + reconciliation before the index can be treated as current. +5. Show partial indexing, unsupported files, generated/vendor policy, and + resource limits. Never describe skipped files as searched. + +The first full index necessarily reads the selected source once. The target for +later questions is to reuse the unchanged branch index, update changed files, +and read exact source spans needed for the response. For committed trees, Git +tree and blob identities are promising stable inputs. Dirty worktrees still +need exact change detection; watcher state, timestamps, and file sizes alone +cannot prove content freshness. + +### What to index + +Start with a visible support matrix instead of claiming to index arbitrary +repository content: + +- Go source: declaration identity, signatures, comments, package names, source + terms, tests, examples, build constraints, and supported typed relationships. +- Repository text: README and design documents, local instructions, module and + workspace files, configuration, and other supported text that explains + behavior or build assumptions. +- Other languages and binary formats: report them as unsupported or add them + only through a separate, evaluated parser. Text matches must not be presented + as Go type or call-graph evidence. + +Generated files, vendored dependencies, very large files, embedded assets, +build tags, cgo, multiple Go modules, and `go.work` files need explicit +inclusion and coverage rules. Reflection, dynamic dispatch, runtime data, +external services, and business requirements remain outside what static source +retrieval can prove. + +### Retrieval and delivery + +The index should return a short list of candidates with current branch/commit, +source locations, match reasons, and coverage. A follow-up operation should +expand a chosen candidate into a bounded context pack with signatures, relevant +source, tests, callers, implementations, documentation, build facts, and +uncertainty where available. Ambiguity should remain visible. The response +should provide a clear way to narrow or continue retrieval without placing the +entire repository into the model's context. + +Keep candidate discovery separate from evidence. Lexical or vector ranking can +find a likely location; exact snapshot validation and the active semantic +provider establish which source and relationships may be returned as current. +Every result should name its branch, commit/tree, build configuration, indexed +coverage, and omissions. + +Keep the existing `go_context` behavior intact. If an MCP branch selector is +needed, make it a separately versioned additive contract. The existing v1 +registry and frozen schemas stay unchanged. The current CLI preview selects a +ref and creates a visible detached worktree; it does not add branch selection +to the MCP protocol or make an existing server switch workspaces. A future +branch retrieval operation must bind candidates and semantic reads to the +selected view rather than the live checkout. + +### Branch source-view review (GPT-6 Luna Max, 2026-09-27) + +Two Luna Max reviewers agreed that the live-workspace `go_context` and its +`SnapshotRef` must not silently become branch selectors. Existing snapshots, +source reads, gopls sessions, and mutating tools are rooted in the configured +workspace. Branch retrieval therefore needs a separate, versioned read-only +result identity that binds the requested branch alias to its resolved commit, +tree, build inputs, and overlay digest. Resolve branch names again for a new +request; if the ref moves during a request, discard the result and require a +new view. Branch-view references must not be accepted by live-workspace +refresh or edit operations. + +Dirty changes have a base. Include staged, unstaged, and untracked changes only +when their captured parent commit/tree is the selected branch view. Reject an +explicit overlay request against a different branch; do not use three-way or +fuzzy patch application to imply that feature-branch edits also describe +`main`. When no overlay is requested for an inactive branch, return its exact +committed source and label the overlay as absent. Recheck source contents and +the branch ref before returning results. Never fall back to live-workspace +bytes if the selected view or its gopls session fails. + +An inactive read-only branch view can answer questions about that commit, but +it does not prove the coding agent is editing the same tree. The response must +identify whether its source view matches the agent's configured workspace and +make its source root available for a deliberately selected edit workspace. +The agent outcome study must direct edits to that exact view or reject the run; +patches written to a different checkout do not qualify branch-aware value. + +The preview chooses a **visible detached Git worktree**. This preserves Git's +normal checkout and filter behavior and leaves the source checkout on its +current branch, while making the managed worktree visible and opt-in. It adds +metadata under the source repository's Git worktree area and must be removed +with normal `git worktree remove` lifecycle handling when no longer needed. +Uninitialized submodules are reported as partial. External `go.work` and local +`replace` inputs, and agent edits made outside the returned view, remain outside +the current source-view contract. The CLI tests cover default `main`, explicit +feature selection, source-checkout isolation, staged/unstaged/untracked +overlays, additions/deletions, wrong-base rejection, branch movement, missing +or corrupt markers, and failed-worktree cleanup. This is a branch source-view +foundation, not a retrieval or accepted-patch qualification. + +The index should remain local by default. Store only the derived material +needed for search where possible, use private cache permissions, enforce disk +and memory quotas, and never put source text, prompts, paths, or branch content +into telemetry. Hosted embeddings or remote indexing would change the privacy +and product boundary and need a separate decision. + +## Design roads + +### Road 1: qualify the current retrieval path + +The process-local Go declaration cache has a model-free relevance screen, but +the corrected medium result is weak and the large result is partial. Improve +coverage and the native comparison before deciding whether a single ranking +change is justified. Keep public contracts stable until a separately versioned +interface is supported by evidence. + +### Road 2: persistent local branch index + +If evaluation shows that repeated parsing or ranking is a material cost, build +a local content-addressed index with per-file records and inverted postings. +Reuse identical files across commits, update changed blobs, keep a bounded +memory cache, and publish a branch generation atomically. A crash or corrupt +generation must leave a known usable generation or produce an explicit rebuild +state. A quota, garbage collection, index-version migration, lock strategy, and +cancellation behavior are part of the design, not later polish. + +Separate retrieval cost from snapshot and gopls cost. Persistent storage helps +only if retrieval work is the bottleneck. Preserve strict freshness and do not +serve a result from an older branch or worktree overlay as current. + +### Road 3: add semantic retrieval selectively + +Use Go syntax and gopls-supported relationships after an anchor is selected. +Add broader graph expansion only when held-out tasks show a missed relationship +that changes the answer. Test local embeddings only if lexical retrieval +repeatedly misses natural-language questions; vector scores remain discovery +signals and must be checked against source. Do not embed a hosted model or +expand to other languages in the current Go-only product cycle. + +## Retrieval evaluation: model-free screen results + +The initial model-free screen of the existing process-local Go declaration +retrieval path is complete. It used pinned Git snapshots, reviewed gold spans, +and three repetitions on the medium corpus; the small and large corpora were +single-run screens. This evaluates candidate discovery, not natural-language +answers, accepted patches, or persistent indexing. Each report records the +source, manifest, retrieval-code, and dirty-diff hashes. + +| Corpus | Pinned source and coverage | Retrieval result | Interpretation | +| --- | --- | --- | --- | +| Small Go 1.27 Interactive Tour | `61ca6068c067756caf8d98b0a933163b39894416`; 17 Go files / 9,444 Go bytes; two Go files parse-incomplete. Git archive is complete, but retrieval metrics are unusable. | No usable retrieval or native metrics; gopls reports unknown completeness. | Too little valid Go evidence for a relevance conclusion. Most reviewed evidence is documentation, and retrieval indexes no text files. | +| Medium Agentic Go | `7b5111c6365a2a12806a561d86745d5bc50d7c9e`; all 420 supported source files captured (270 Go files / 1,343,953 bytes; 150 text files / 1,638,309 bytes); no parse-incomplete Go files. | Four questions, three repetitions: Recall@5 0.20, Precision@5 0.20, MRR@5 0.2375; Recall@10 0.35, Precision@10 0.20, MRR@10 0.26875. | Weak candidate relevance. Text exists in the corpus but `text_indexed_by_retrieval` is zero. This is not an agent-outcome or product-value result. | +| Kubernetes | `dfd7b93a1783878be367e1fc4a780318330cb3bf`; 17,856 Go files / 188,240,197 Go bytes. It meets the proposed large stratum by file count, not by the 250 MiB source threshold. Two supported symlinks were not indexed and two blobs (5,588 bytes) were transformed by `git archive`; package-inventory output reached its 16 MiB cap. | Relevance metrics are unusable. One run observed warm retrieval p50 10.39 s / p95 10.52 s, cold p50 10.48 s, and peak sampled Go heap 702,613,528 bytes. Parsing was about 9.43 s warm. | Partial-run operational signal only. Warm p95 exceeds the 5 s query target. Sampled Go heap was about 703 MB; process RSS was not measured, so this does not establish whether the 512 MiB process-memory target was met. Do not claim arbitrary-large support. | + +The corrected medium report is +[`medium-agentic-go-anchor-v2.json`](../../validation/retrieval/results/2026-09-27/medium-agentic-go-anchor-v2.json) +(SHA-256 `f4c8f8d8a9546b24d1251829b96acbf8bcbed18dd5bd4d8d0cf2133a167a893e`). +The original `medium-agentic-go.json` artifact is retained, but its relevance +score is superseded: some gold declaration ranges began inside a declaration +body while candidates are declaration-name anchors. The corrected spans +include the relevant symbol-name anchor lines. The corrected manifest SHA-256 +is `ef352f80ddf9a31c5db7e68465e0c208290313ebd315245ca3ec85d6d1ecba4b`. + +The capped large rerun is +[`large-kubernetes-capped-v2.json`](../../validation/retrieval/results/2026-09-27/large-kubernetes-capped-v2.json) +(SHA-256 `f95231a4bcf9f3942a89c815f73c4be727a627fc0221858ecbaf6d0d50c29efa`). +The earlier large artifact recorded package-inventory output above the stated +limit. The bounded writer and regression tests now cap it at 16 MiB and mark +the inventory `partial_output_limit`; the corrected rerun records 736 package +objects before truncation. It still does not repair the separate Git archive +coverage gaps, so its relevance scores remain unusable. + +The fixed `rg` arm scored zero on all four corrected medium questions because +the harness tokenizes natural-language queries and ranks matching lines by +token overlap. That is not a competent native `rg` + Go tools + gopls workflow +and cannot support a superiority claim. The gopls `workspace/symbol` arm is +directional only: exhaustive-result completeness is unknown. The package +inventory is separate and has no relevance score. Improve and freeze the +native workflow before making comparative claims. No GPT-6 Luna Max task runs +have been performed; this screen is model-free and is not mixed with +historical GPT-6 Sol/high findings. + +The separate six-cell selector screen completed with safety passed and the +efficiency promotion gate failed. That is selector-guidance reliability +evidence, not retrieval evidence; it neither blocks nor qualifies retrieval. +Do not run the stale 18-cell rerun. Keep its outcome separate from retrieval. + +### Private text-candidate ablation + +A fresh four-question manifest was reviewed before scoring. It is pinned to +the same Agentic Go commit and tree as the medium screen and covers coordinate +handling, audit-rule evidence, artifact pagination, and execution trust +boundaries. The manifest hash is +`3983e5ffdbd8155ba8aedbbc70eb0b251b35629c81005f4fb9b71e2feef9e2fb`. A +GPT-6 Luna Max evaluation-methodology review removed query hints and tightened +gold anchors before the run. The scoring itself is deterministic and model-free. + +The ablation captured all 150 supported text files (1,638,309 bytes) from the +complete 420-file archive, and its capped mixed candidate pools remained +complete at 6,125–9,231 candidates per query. Query-matched candidate-pool +Recall was 1.00 for all four questions (15/15 gold spans). The current Go-only +retrieval path on these fresh questions had macro Recall@10 0.05, Precision@10 +0.025, and MRR@10 0.25. Adding text line fragments to the evaluation-only +candidate set raised macro Recall@10 to 0.1625, Precision@10 to 0.075, and +MRR@10 to 0.28125; it retrieved 3/15 gold spans in the top 10. This is a +small-sample coverage and ranking diagnosis: candidate generation can surface +the reviewed evidence, while the existing scorer does not rank enough of it +near the top. The result remains weak and does not demonstrate useful agent +outcomes or justify a public retrieval change or persistent index. + +The ablation emitted at most 50,000 eligible text-line fragments per query; +all four candidate-pool audits were below the 10,000 mixed-candidate audit cap. +The retrieval report records capture counts, coverage, per-question metrics, +timings, and source hashes at +[`medium-agentic-go-text-ablation-v1.json`](../../validation/retrieval/results/2026-09-27/medium-agentic-go-text-ablation-v1.json) +(SHA-256 +`b276711e4cfdde2c997a67c227fcb52a25ec6596b2deb253f1896abd0e9628ab`). Text +capture and ranking were evaluation-only; live `SearchProfiled` remains +Go-declaration-only. Its roughly 86–94 ms per-question ablation timings include +the mixed-candidate work and are not a warm-query comparison. + +### Evaluation question + +Does the current retrieval path surface independently judged relevant, +source-grounded candidates at useful quality and cost across repository +scales, compared with native `rg`, Go tools, and upstream gopls? How much of +the measured request time belongs to candidate retrieval rather than snapshot +capture or semantic resolution? + +### Study plan + +1. Freeze three representative Go repositories at pinned commits: small, + medium, and large. Treat at least 10,000 Go files or 250 MiB of Go source as + a proposed large-repository stratum, not a claim of existing support. Include + a repository with multiple packages/modules where available. +2. Prepare held-out questions for symbol discovery, a multi-package behavior + path, and a cross-package API change. Record independently reviewed relevant + declarations, call sites, tests, docs, and evidence limitations before + seeing ranked results. Use a branch-matched source view and mark incomplete + archives and unsupported evidence unusable rather than pooling them. +3. Compare the existing retrieval path with native `rg`, Go tools, and upstream + gopls using equivalent instructions and task budgets. Keep ordinary tool + availability separate from extra workflow guidance so integration effects + are visible. +4. Measure candidate Recall@5/10, reciprocal rank, source-span precision, + omitted relevant tests/relationships, and explicit coverage. Measure cold + and warm query p50/p95, snapshot capture, candidate retrieval, gopls + resolution, peak memory, disk size, and response bytes. Report snapshot, + retrieval, and semantic-provider time separately. This is a model-free + benchmark; it makes no claims about agent decisions, accepted changes, or + token savings. + +### Later branch-supported engineering screen + +After the product can select a branch and resolve evidence against that exact +branch's source view, run a separate matched native-versus-Agentic-Go screen +using GPT-6 Luna Max only. Keep the model version, settings, tasks, instructions, +and budgets matched across conditions. Measure answer correctness, accepted +changes, consequential omissions, engineer review/rework effort, and total +time. Do not use the historical GPT-6 Sol/high 72-run harness or mix models. +This later screen tests engineering outcomes; it is not part of the model-free +retrieval and repository-scale benchmark. + +### Decision gates + +- The current medium relevance screen is weak and the large screen is partial. + Before the one allowed retrieval redesign, repair coverage and freeze a + competent native baseline plus fresh held-out questions. Make one isolated + change against a diagnosed failure, then evaluate it on different fresh + cases. If relevance does not improve or coverage cannot be established, stop + this retrieval-value path and do not add persistence. Engineering outcomes + belong to the later matched model screen. +- Prototype a persistent index only when the current approach misses a + predeclared interactive latency or memory budget on representative large + repositories, profiles attribute most of that cost to candidate retrieval, + and retrieval quality is useful enough to preserve. Require a persistent + prototype to improve warm-query time by at least 2x without lowering + held-out relevance or freshness. The current screening targets are 5 s warm + query p95 and 512 MiB peak process memory; they are not support promises. +- If snapshot capture dominates, address that cost or narrow the observed + scope before investing in disk indexing. If retrieval is useful but lexical + misses dominate, test one local semantic-ranking option on held-out queries + before expanding the index format. +- The branch source-view preview exists, but the engineering-outcome screen + stays gated on useful retrieval and an exact task workspace. Then compare + native Go tools and Agentic Go on eight held-out tasks across three + repositories, randomized paired order and two fresh repetitions per arm + (32 GPT-6 Luna Max runs). Freeze source, binary, task, prompt, and evaluator + hashes first. Require zero accepted stale/wrong-branch evidence, no accepted + patch-quality loss, and a practical gain such as 15% lower median time to an + accepted patch; track review effort and actual token use. Report the + same-model blind-reviewer limitation. A screen is directional, not a + statistically powered or general claim. + +The measurements qualify only the repositories, questions, builds, branches, +models, and hosts included. They cannot prove complete understanding of every +codebase or guarantee that an agent will follow retrieved evidence. + +## Decision record + +- 2026-09-27: The product owner set branch-aware, reusable codebase indexing + and fast, accurate retrieval as a long-term product aspiration. +- 2026-09-27: The GPT-6 Sol council recommended finishing and evaluating the + existing process-local retrieval path before persistent storage. Persistent + local indexing remains the conditional scale path; embeddings and broader + language support remain deferred research. +- 2026-09-27: The six-cell selector screen passed safety but failed its + efficiency promotion gate. This result is separate from retrieval evaluation; + do not run or direct an 18-cell rerun absent a new bounded intervention. +- 2026-09-27: GPT-6 Luna Max verdicts set the model-free held-out retrieval and + repository-scale benchmark as the first research action, followed only after + branch support exists by a separate matched GPT-6 Luna Max native-versus- + Agentic-Go screen. Preserve both dirty source worktrees; do not use the Sol/ + high 72-run harness or mix models. At the time of that decision, retrieval + evaluation and branch-aware product behavior were still pending. +- 2026-09-27: Luna Max source-view review separated branch retrieval from live + `go_context`, required exact-base overlays and agent/edit alignment, and left + the source materialization mechanism open pending a fidelity fixture. +- 2026-09-27: The model-free screen completed. The corrected medium corpus + scored 0.35 Recall@10 and 0.20 Precision@10 with zero document retrieval. + The small screen had no usable retrieval metrics. Kubernetes coverage was + partial; the capped package inventory records its 16 MiB cutoff. No + persistent index or retrieval superiority claim is justified. +- 2026-09-27: A fresh, reviewed four-question text-candidate ablation recovered + all 15/15 gold spans in complete candidate pools, but the top-10 mixed + ranking reached only macro Recall 0.1625, Precision 0.075, and MRR 0.28125. + This evaluation-only candidate augmentation diagnoses both a text-coverage + gap and a remaining ranking failure. It does not qualify live text retrieval + or persistence; the one bounded retrieval redesign allowance is consumed. +- 2026-09-27: Luna Max architecture, evaluation, and product reviewers agreed + that persistence and the 32-run engineering screen remain gated. A private + text-candidate ablation measured full-pool coverage but weak ranking, so no + persistence or 32-run agent-value study is justified. The source-view CLI is + a preview foundation with exact branch identity, not a branch-aware + persistent retrieval product. Repair the native workflow before any future + comparative claim; keep the reliability screen and historical Sol studies + separate. diff --git a/docs/v1.0.0-roadmap.md b/docs/v1.0.0-roadmap.md index e913566..e94ce21 100644 --- a/docs/v1.0.0-roadmap.md +++ b/docs/v1.0.0-roadmap.md @@ -1,5 +1,11 @@ # Road to v1.0.0 +This document records the completed v1 stages and their compatibility evidence. +For current product direction and future work, use the +[Go engineering north star](go-intelligence-north-star.md) and +[continuation handoff](continuation/go-intelligence.md). Their broader workflow +does not retroactively change the release contracts below. + ## Product direction Agentic-go v1 is source-grounded Go change intelligence for coding agents: diff --git a/validation/v1.0.0/adoption-remediation-2026-09-24.md b/validation/v1.0.0/adoption-remediation-2026-09-24.md new file mode 100644 index 0000000..db176b4 --- /dev/null +++ b/validation/v1.0.0/adoption-remediation-2026-09-24.md @@ -0,0 +1,65 @@ +# Selector remediation evidence + +Campaign identity: `luna-guidance-remediation-20260924` + +Status: implementation and focused verification complete. The post-change six-cell guidance screen completed, but its overhead gate failed. The canonical 18-cell matrix was not run. + +## Pre-remediation baseline + +The preceding Luna canonical campaign contained 18 runs across the two pinned scenarios, three arms, and three repetitions. The source hashes were: + +- `7354d9c8debb4bcf2225bf429857078de310c176` +- `8c9ee70637600318f1cc4e3931da78f084e41123` + +The evaluated binary hash was `bd7535d8c7172219d48ac45b873939af7d3cac004f018230e85db030c6302cc9`. + +The guidance arm passed acceptance and qualification in 6/6 runs, had zero scope violations, used focus evidence in 6/6 runs, and completed refresh in 6/6 runs. It produced five failed focus calls. Client-go produced one low-level `invalid_input` failure. gRPC produced four low-level `provider` failures. Transcript audit classified all five as selector misuse: one mutated Symbol Ref and four invalid declaration coordinates. No provider defect was established. + +Pre-remediation guidance medians were 215664 ms and 22 tool calls for client-go, and 330537 ms and 32 tool calls for gRPC. These are baseline evidence for the rerun, not a product-value claim. + +## Remediation gates + +- Context lineage: passed. The post-edit refresh used `base` plus the previous pack ID and returned `refresh.status: replaced`. +- Guidance and digest coverage: passed. +- Private failure classification and deterministic replay tests: passed. +- MCP inventory and schema goldens: passed. +- `go test ./...`, `go test -race ./...`, `go vet ./...`, `go build ./...`, `git diff --check`, and task validation: passed. + +## Post-remediation guidance screen + +The six-cell screen used the same two pinned sources and three Luna repetitions per scenario. The client-go records are retained at `private://agentic-go-luna-selector-remediation-20260924/runs`. The gRPC records are retained at `private://agentic-go-luna-selector-remediation-20260924-b/runs`. The rebuilt server binary SHA-256 was `020a7b4ca2e9fdf940e2131d63ec4cc90544071ab859feb6147693f10a9c914e`. The evaluator binary SHA-256 was `3f1d6513c9ccc1d09be6ee8aef62788045ab4b6282df16fd63c0ce80e3e66022`. + +All six runs passed acceptance and qualification, used evidence, completed refresh, and had zero scope violations, failed focus calls, selector misuse, repeated rejected selectors, and stale evidence acceptance. The reliability behavior therefore passed this screen. + +The promotion gate did not pass: + +- client-go median focus calls: 3, above the limit of 2; +- gRPC median focus calls: 4, above the limit of 2; +- client-go median duration: 282291 ms versus 183559 ms baseline, 1.54x; +- gRPC median duration: 367794 ms versus 315501 ms baseline, 1.17x; +- client-go median total tool calls: 27 versus 16 baseline, 1.69x; +- gRPC median total tool calls: 36 versus 24 baseline, 1.50x. + +This is an overhead and workflow-efficiency failure, not evidence of a provider defect. The result blocks the canonical 18-cell rerun and any product-value or speed claim. The next fix should reduce redundant exploration and focus calls before another guidance campaign. + +## Rejected efficiency probe + +A follow-up wording probe required one initial context call and one post-edit refresh by default. It was run once on client-go with the rebuilt guidance and retained at `private://agentic-go-luna-selector-remediation-20250925-c/runs`. The run passed acceptance and qualification, but still made three focus calls and produced one `selector_misuse` failure. The probe was reverted and is not part of the shipped guidance. This supports treating call reduction as an agent-behavior problem requiring a better intervention design, not more repeated prose. + +## Private efficiency diagnostic + +Astra reviewed the paired trace audit and recommended no provider, guidance, threshold, or public-surface change. The evaluator now records a bounded `redundant_refreshes` diagnostic when a successful refresh repeats with unchanged snapshot, scope, and evidence requirements without an intervening edit or stale rejection. Focused tests cover valid narrowing, edits, stale rejection, and changed evidence requirements. The diagnostic does not suppress calls or relax freshness checks. + +The six-cell efficiency gate remains failed. No new Luna campaign or canonical 18-cell matrix is justified by this slice. + +## Bounded classification contract + +The historical low-level categories remain unchanged. New private records contain only a bounded root cause, selector kind, normalized reason, recovery status, and repeated-selector status. They never contain raw Symbol Refs, source paths, prompts, transcripts, or provider error text. + +The accepted root causes are `selector_misuse`, `provider_failure`, `stale_snapshot`, `timeout`, `transport`, and `unknown`. + +## Limitations + +This campaign does not establish improved correctness, speed, productivity, token usage, adoption, or product value. It does not justify provider fallback, relaxed stale checks, automatic Symbol Ref correction, a new MCP surface, or a release claim. External decision-value evaluation remains pending until real frontier and open-model clients are available. + +Private raw artifacts are retained outside the repository at `private://agentic-go-luna-causal-20260924-c/canonical-only.json`. diff --git a/validation/v1.0.0/command-activity-audit-2026-09-25.md b/validation/v1.0.0/command-activity-audit-2026-09-25.md new file mode 100644 index 0000000..5b65bc6 --- /dev/null +++ b/validation/v1.0.0/command-activity-audit-2026-09-25.md @@ -0,0 +1,73 @@ +# Command Activity Audit + +Date: 2026-09-25 +Campaign: selector remediation guidance screen +Model: Luna +Scope: six guided traces and six paired baseline traces across client-go and grpc-go + +Guided source campaigns: selector-remediation-20260924 and selector-remediation-20260924-b + +Baseline source campaign: causal-20260924-c + +The private source manifest retains the SHA-256 digests for all twelve run records. + +## Decision + +The guided workflow increased downstream shell activity, but this audit does not identify a safe product intervention. + +Guided traces issued 172 completed shell commands compared with 111 in the paired baselines. The increase was concentrated in source inspection, not repeated selector failure or recovery: + +| Arm | Commands | Discovery | Source inspection | Verification | Editing | Recovery | Unknown | +| --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | +| Guided | 172 | 37 | 134 | 1 | 0 | 0 | 0 | +| Baseline | 111 | 43 | 66 | 2 | 0 | 0 | 0 | + +The traces contain no confirmed stale-evidence acceptance, unchanged rejected-selector retry, selector misuse, provider failure, or recovery loop. The command activity therefore does not justify changing provider behavior, relaxing freshness checks, suppressing refreshes, or adding more generic selector guidance. + +## Trace-linked accounting + +The table reports only bounded counts. Raw commands, transcripts, local paths, prompts, and opaque references remain in private operator-held artifacts. + +| Task | Arm | Run | Commands | Before edit | After edit | File changes | MCP calls | Focus calls | Duration ms | Exact duplicate commands | +| --- | --- | --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | +| client-go | Guided | r1 | 33 | 29 | 4 | 1 | 3 | 3 | 284919 | 1 | +| client-go | Guided | r2 | 23 | 19 | 4 | 1 | 3 | 3 | 282291 | 1 | +| client-go | Guided | r3 | 21 | 18 | 3 | 1 | 3 | 3 | 222703 | 1 | +| grpc-go | Guided | r1 | 37 | 25 | 12 | 1 | 4 | 4 | 367794 | 1 | +| grpc-go | Guided | r2 | 25 | 21 | 4 | 2 | 5 | 5 | 1026486 | 0 | +| grpc-go | Guided | r3 | 33 | 26 | 7 | 1 | 2 | 2 | 334137 | 1 | +| client-go | Baseline | r1 | 12 | 10 | 2 | 1 | 0 | 0 | 206861 | 0 | +| client-go | Baseline | r2 | 16 | 14 | 2 | 1 | 0 | 0 | 183559 | 0 | +| client-go | Baseline | r3 | 15 | 12 | 3 | 1 | 0 | 0 | 157568 | 0 | +| grpc-go | Baseline | r1 | 33 | 22 | 11 | 2 | 0 | 0 | 315501 | 0 | +| grpc-go | Baseline | r2 | 23 | 19 | 4 | 1 | 0 | 0 | 205915 | 0 | +| grpc-go | Baseline | r3 | 12 | 9 | 3 | 2 | 0 | 0 | 374857 | 0 | + +## Activity interpretation + +- Client-go runs narrowed from an ambiguous function query to the exact current candidate before refreshing. The additional inspection was relevant to candidate selection, not a repeated rejected selector. +- grpc-go runs narrowed broad symbol searches to the concrete RBAC normalizer before refreshing. The second refresh in guided r2 followed a further edit. Under strict snapshot lineage, that refresh is required and cannot be suppressed safely. +- Five exact duplicate command invocations occurred across the six guided traces. Every duplicate was discovery-class activity. No repeated source-inspection sequence or unchanged selector retry was established. +- Editing was represented by seven file-change events in guided traces and eight in baselines. No shell editing command was observed. +- The classifier was deterministic and bounded. Discovery includes repository state, history, file listing, module metadata, and environment lookup. Source inspection includes content search, file reads, and diff review. Verification includes explicit test, build, vet, and diff-check commands. No unclassified command remained. + +## Recoverable overhead + +There is no defensible time estimate because the traces do not attribute elapsed time to individual shell commands, and command duration is not total task duration. + +The mechanical upper bound is five exact duplicate discovery invocations. A separate private focus audit identifies one additional refresh in grpc-go guided r2, but it followed an edit and is required by the current freshness contract. These six invocations are therefore a hypothetical counterfactual, not a safe optimization target. The defensible recoverable overhead is zero. + +## Product and gate implications + +- The guided runs still exceeded the focus-call target: client-go median 3 and grpc-go median 4 against a target of 2. +- The audit does not establish that command reduction would improve correctness, speed, productivity, token usage, adoption, or decision quality. +- No new Luna campaign or 18-run matrix is justified by this audit. +- CLI and GitHub Action remain the primary verification surfaces. MCP remains an agent-facing adapter whose value must come from evidence quality and decision relevance, not call-count reduction alone. + +The next defensible evaluation would instrument agent action reasons and distinguish necessary source inspection from avoidable exploration before testing an intervention. That is outside this slice. Stop MCP efficiency remediation here unless a reproducible mechanism is identified. + +## Privacy and provenance + +Private raw JSON traces and the command classifier inputs remain outside the repository. This report stores aggregate counts, bounded categories, run labels, and conclusions only. No raw source, prompts, transcripts, filesystem paths, Symbol Refs, or provider errors are included. + +Frozen MCP inventory and public schemas were not changed by this audit. From 718d7c920578e471ad09abe4002eba896387d5f0 Mon Sep 17 00:00:00 2001 From: Ashwin Gopalsamy <47941624+ashwingopalsamy@users.noreply.github.com> Date: Sun, 27 Sep 2026 13:03:32 +0530 Subject: [PATCH 19/20] fix(ci): resolve branch retrieval lint findings --- internal/intelligence/focus.go | 6 +- internal/intelligence/retrieval/index.go | 32 +++---- internal/intelligence/retrieval/index_test.go | 4 +- internal/sourceview/sourceview.go | 86 ++++++++++--------- validation/cmd/retrievalbench/main.go | 4 +- validation/internal/pilot/runner.go | 1 + validation/internal/retrievalstudy/archive.go | 9 +- validation/internal/retrievalstudy/golist.go | 8 +- validation/internal/retrievalstudy/gopls.go | 2 +- .../internal/retrievalstudy/identity.go | 34 ++++---- .../internal/retrievalstudy/manifest.go | 6 +- validation/internal/retrievalstudy/native.go | 12 +-- validation/internal/retrievalstudy/run.go | 60 +++++++------ validation/internal/retrievalstudy/types.go | 32 +++++++ 14 files changed, 172 insertions(+), 124 deletions(-) diff --git a/internal/intelligence/focus.go b/internal/intelligence/focus.go index b5129ce..6f8cd6b 100644 --- a/internal/intelligence/focus.go +++ b/internal/intelligence/focus.go @@ -184,12 +184,12 @@ type storedFocusSymbol struct { } type focusRetrievalStatus struct { - used bool - complete bool - truncated bool fallback string indexedFiles int skippedFiles int + used bool + complete bool + truncated bool } //nolint:govet // Field order keeps persisted evidence readable. diff --git a/internal/intelligence/retrieval/index.go b/internal/intelligence/retrieval/index.go index e3afa62..8f210b9 100644 --- a/internal/intelligence/retrieval/index.go +++ b/internal/intelligence/retrieval/index.go @@ -60,28 +60,28 @@ type File struct { // Line and Column are one-based UTF-8 byte coordinates. type Candidate struct { Path string - Line int - Column int Name string Qualified string Kind string Package string Score float64 + Line int + Column int } // Result reports ranked candidates and whether every eligible Go file was // structurally indexed. A partial result is still useful for discovery, but // callers must preserve the incompleteness as uncertainty. type Result struct { - Candidates []Candidate - CandidateCount int - IndexedFiles int - SkippedFiles int - TextIndexedFiles int - TextSkippedFiles int + Candidates []Candidate + CandidateCount int + IndexedFiles int + SkippedFiles int + TextIndexedFiles int + TextSkippedFiles int TextIndexedFragments int - Complete bool - Truncated bool + Complete bool + Truncated bool } // SearchProfile reports non-overlapping work performed by one SearchProfiled @@ -108,18 +108,18 @@ type fileKey struct { type cacheEntry struct { key fileKey index indexedFile - complete bool size int + complete bool } // Cache reuses parsed per-file fragments across observations while retaining // only bounded derived metadata in memory. Eviction only causes reparsing. type Cache struct { + entries map[fileKey]*list.Element + order *list.List mu sync.Mutex maximum int bytes int - entries map[fileKey]*list.Element - order *list.List } // NewCache constructs the process-local retrieval cache. @@ -419,8 +419,6 @@ type indexedFile struct { type fragment struct { path string - line int - column int name string qualified string kind string @@ -433,6 +431,8 @@ type fragment struct { packageTerms []string kindTerms []string length int + line int + column int } func (c *Cache) fileIndex(key Key, file File) (indexedFile, bool, bool, time.Duration) { @@ -574,7 +574,7 @@ func receiverName(fields *ast.FieldList) string { if fields == nil || len(fields.List) == 0 { return "" } - var expression ast.Expr = fields.List[0].Type + expression := fields.List[0].Type for { switch typed := expression.(type) { case *ast.StarExpr: diff --git a/internal/intelligence/retrieval/index_test.go b/internal/intelligence/retrieval/index_test.go index 45c34ca..212e702 100644 --- a/internal/intelligence/retrieval/index_test.go +++ b/internal/intelligence/retrieval/index_test.go @@ -118,11 +118,11 @@ func TestSearchProfiledReportsColdAndWarmWorkWithoutChangingResults(t *testing.T func TestSearchWithTextProfiledAddsBoundedTextCandidatesWithSameScorer(t *testing.T) { goFiles := []File{{ - Path: "position.go", + Path: "position.go", Contents: []byte("package fixture\n\n// Position converts source offsets.\nfunc Position() {}\n"), }} textFiles := []File{{ - Path: "contracts.md", + Path: "contracts.md", Contents: []byte("Public locations use one-based UTF-8 byte columns. UTF-16 positions exist only inside the pinned LSP adapter.\n"), }} key := Key{Workspace: "repo", Scope: "commit", Build: "retrievalbench", Provider: "lexical-declaration-index"} diff --git a/internal/sourceview/sourceview.go b/internal/sourceview/sourceview.go index 3c34ea0..da7d183 100644 --- a/internal/sourceview/sourceview.go +++ b/internal/sourceview/sourceview.go @@ -26,6 +26,7 @@ import ( "github.com/agentic-mcps/go/internal/workspace" ) +// SchemaVersion identifies the branch source-view metadata contract. const ( SchemaVersion = "agentic.branch-view/v1" metadataConfigKey = "agentic-go.branch-view" @@ -33,9 +34,8 @@ const ( maximumOverlaySize = 8 << 20 ) -var ( - ErrStale = errors.New("branch source view is stale") -) +// ErrStale indicates that a branch source view no longer matches its selected ref. +var ErrStale = errors.New("branch source view is stale") // Marker is private per-worktree provenance used to reject a moved branch ref. // It is stored in the linked worktree's Git metadata, not in repository files. @@ -52,25 +52,25 @@ type Marker struct { // Request selects a branch source view and an optional exact dirty overlay. type Request struct { - Branch string - OutputPath string - IncludeDirty bool + Branch string + OutputPath string + IncludeDirty bool } // Result identifies the materialized source tree and any known checkout gaps. // //nolint:govet // Field order matches the CLI JSON response. type Result struct { - SchemaVersion string `json:"schema_version"` - RequestedBranch string `json:"requested_branch"` - Branch string `json:"branch"` - Ref string `json:"ref"` - Commit string `json:"commit"` - Tree string `json:"tree"` - ViewPath string `json:"view_path"` - OverlayIncluded bool `json:"overlay_included"` - OverlayDigest string `json:"overlay_digest,omitempty"` - CheckoutComplete bool `json:"checkout_complete"` + SchemaVersion string `json:"schema_version"` + RequestedBranch string `json:"requested_branch"` + Branch string `json:"branch"` + Ref string `json:"ref"` + Commit string `json:"commit"` + Tree string `json:"tree"` + ViewPath string `json:"view_path"` + OverlayIncluded bool `json:"overlay_included"` + OverlayDigest string `json:"overlay_digest,omitempty"` + CheckoutComplete bool `json:"checkout_complete"` CheckoutLimitations []string `json:"checkout_limitations"` } @@ -84,14 +84,14 @@ type selectedBranch struct { type overlayFile struct { path string - mode fs.FileMode content []byte + mode fs.FileMode } type overlay struct { - patch []byte - files []overlayFile digest string + patch []byte + files []overlayFile } // Create resolves a branch to an exact commit, creates a detached worktree at @@ -150,8 +150,8 @@ func Create(ctx context.Context, runner *execution.Runner, source *workspace.Wor } }() - if _, err := gitBytes(ctx, runner, "worktree", "add", "--detach", "--", outputPath, branch.commit); err != nil { - return Result{}, fmt.Errorf("creating detached branch view: %w", err) + if _, addErr := gitBytes(ctx, runner, "worktree", "add", "--detach", "--", outputPath, branch.commit); addErr != nil { + return Result{}, fmt.Errorf("creating detached branch view: %w", addErr) } created = true @@ -252,7 +252,7 @@ func ReadMarker(ctx context.Context, runner *execution.Runner) (Marker, bool, er return Marker{}, false, fmt.Errorf("resolving worktree Git directory: %w", err) } if !filepath.IsAbs(gitDir) { - return Marker{}, false, errors.New("Git returned a non-absolute worktree directory") + return Marker{}, false, errors.New("git returned a non-absolute worktree directory") } configPath := filepath.Join(gitDir, "config.worktree") value, exitCode, err := gitRun(ctx, runner, "config", "--file", configPath, "--get", metadataConfigKey) @@ -278,8 +278,8 @@ func ReadMarker(ctx context.Context, runner *execution.Runner) (Marker, bool, er return Marker{}, false, fmt.Errorf("decoding branch source view metadata: %w", err) } var marker Marker - if err := json.Unmarshal(data, &marker); err != nil { - return Marker{}, false, fmt.Errorf("parsing branch source view metadata: %w", err) + if unmarshalErr := json.Unmarshal(data, &marker); unmarshalErr != nil { + return Marker{}, false, fmt.Errorf("parsing branch source view metadata: %w", unmarshalErr) } markerCommit, found, err := resolveCommit(ctx, runner, metadataRef) if err != nil { @@ -333,7 +333,7 @@ func resolveBranch(ctx context.Context, runner *execution.Runner, requested stri tree, err := gitText(ctx, runner, "rev-parse", "--verify", "--end-of-options", commit+"^{tree}") if err != nil || !validObjectID(tree) { if err == nil { - err = errors.New("Git returned an invalid tree object ID") + err = errors.New("git returned an invalid tree object ID") } return selectedBranch{}, fmt.Errorf("resolving branch tree: %w", err) } @@ -369,7 +369,7 @@ func resolveCommit(ctx context.Context, runner *execution.Runner, ref string) (s } commit := strings.TrimSpace(string(output)) if !validObjectID(commit) { - return "", false, errors.New("Git returned an invalid commit object ID") + return "", false, errors.New("git returned an invalid commit object ID") } return commit, true, nil } @@ -431,7 +431,7 @@ func checkoutLimitations(ctx context.Context, runner *execution.Runner, commit s return []string{fmt.Sprintf("%d Git submodule entries are not initialized in this branch view", modules)}, nil } -func captureOverlay(ctx context.Context, runner *execution.Runner, sourceRoot, commit string) (overlay, error) { +func captureOverlay(ctx context.Context, runner *execution.Runner, sourceRoot, commit string) (captured overlay, returnErr error) { head, err := gitText(ctx, runner, "rev-parse", "--verify", "HEAD^{commit}") if err != nil { return overlay{}, fmt.Errorf("resolving source checkout HEAD: %w", err) @@ -472,7 +472,11 @@ func captureOverlay(ctx context.Context, runner *execution.Runner, sourceRoot, c if err != nil { return overlay{}, fmt.Errorf("opening source workspace root: %w", err) } - defer sourceDirectory.Close() + defer func() { + if closeErr := sourceDirectory.Close(); closeErr != nil { + returnErr = errors.Join(returnErr, fmt.Errorf("closing source workspace root: %w", closeErr)) + } + }() for _, path := range paths { if err := ctx.Err(); err != nil { return overlay{}, err @@ -541,23 +545,27 @@ func applyPatch(ctx context.Context, runner *execution.Runner, viewPath string, } patchPath := file.Name() defer func() { _ = os.Remove(patchPath) }() - if _, err := file.Write(patch); err != nil { + if _, writeErr := file.Write(patch); writeErr != nil { _ = file.Close() - return fmt.Errorf("writing temporary overlay patch: %w", err) + return fmt.Errorf("writing temporary overlay patch: %w", writeErr) } - if err := file.Close(); err != nil { - return fmt.Errorf("closing temporary overlay patch: %w", err) + if closeErr := file.Close(); closeErr != nil { + return fmt.Errorf("closing temporary overlay patch: %w", closeErr) } _, err = gitBytes(ctx, runner, "apply", "--binary", "--whitespace=nowarn", patchPath) return err } -func copyOverlayFiles(root string, files []overlayFile) error { +func copyOverlayFiles(root string, files []overlayFile) (returnErr error) { viewRoot, err := os.OpenRoot(root) if err != nil { return fmt.Errorf("opening selected worktree root: %w", err) } - defer viewRoot.Close() + defer func() { + if closeErr := viewRoot.Close(); closeErr != nil { + returnErr = errors.Join(returnErr, fmt.Errorf("closing selected worktree root: %w", closeErr)) + } + }() for _, file := range files { if err := makeContainedParents(viewRoot, file.path); err != nil { return err @@ -600,8 +608,8 @@ func makeContainedParents(root *os.Root, relative string) error { } info, err := root.Lstat(current) if errors.Is(err, fs.ErrNotExist) { - if err := root.Mkdir(current, 0o755); err != nil && !errors.Is(err, fs.ErrExist) { - return fmt.Errorf("creating overlay directory: %w", err) + if mkdirErr := root.Mkdir(current, 0o755); mkdirErr != nil && !errors.Is(mkdirErr, fs.ErrExist) { + return fmt.Errorf("creating overlay directory: %w", mkdirErr) } info, err = root.Lstat(current) } @@ -625,11 +633,11 @@ func writeMarker(ctx context.Context, runner *execution.Runner, marker Marker) e return err } if !filepath.IsAbs(gitDir) { - return errors.New("Git returned a non-absolute worktree directory") + return errors.New("git returned a non-absolute worktree directory") } encoded := base64.RawURLEncoding.EncodeToString(data) - if _, err := gitBytes(ctx, runner, "config", "--file", filepath.Join(gitDir, "config.worktree"), "--replace-all", metadataConfigKey, encoded); err != nil { - return err + if _, writeErr := gitBytes(ctx, runner, "config", "--file", filepath.Join(gitDir, "config.worktree"), "--replace-all", metadataConfigKey, encoded); writeErr != nil { + return writeErr } _, err = gitBytes(ctx, runner, "update-ref", metadataRef, marker.Commit) return err diff --git a/validation/cmd/retrievalbench/main.go b/validation/cmd/retrievalbench/main.go index 926a66a..c1b2b1d 100644 --- a/validation/cmd/retrievalbench/main.go +++ b/validation/cmd/retrievalbench/main.go @@ -52,5 +52,7 @@ func main() { fmt.Fprintln(os.Stderr, "retrievalbench:", err) os.Exit(1) } - fmt.Fprintln(os.Stdout, "retrieval report written") + if _, err := fmt.Fprintln(os.Stdout, "retrieval report written"); err != nil { + os.Exit(1) + } } diff --git a/validation/internal/pilot/runner.go b/validation/internal/pilot/runner.go index b3de555..4e778c2 100644 --- a/validation/internal/pilot/runner.go +++ b/validation/internal/pilot/runner.go @@ -201,6 +201,7 @@ type FocusFailureRecord struct { Repeated bool `json:"repeated"` } +// FocusFailureSelectorMisuse and related constants classify failed go_context calls. const ( FocusFailureSelectorMisuse = "selector_misuse" FocusFailureProvider = "provider_failure" diff --git a/validation/internal/retrievalstudy/archive.go b/validation/internal/retrievalstudy/archive.go index df51409..180818d 100644 --- a/validation/internal/retrievalstudy/archive.go +++ b/validation/internal/retrievalstudy/archive.go @@ -37,6 +37,7 @@ type sourceMeta struct { goFile bool } +//nolint:govet // Keep extracted source, file inventory, and coverage together. type archivedSource struct { workspace string goFiles []retrieval.File @@ -59,9 +60,9 @@ type archiveLimits struct { indexTextCandidates bool } -// ExportCommit uses git archive for the exact manifest commit. It never reads +// exportCommit uses git archive for the exact manifest commit. It never reads // from the checkout's working tree, index, or current branch. -func ExportCommit(ctx context.Context, repositoryPath, workspace string, manifest Manifest, limits archiveLimits) (Repository, archivedSource, error) { +func exportCommit(ctx context.Context, repositoryPath, workspace string, manifest Manifest, limits archiveLimits) (Repository, archivedSource, error) { repositoryRoot, err := gitText(ctx, limits.timeout, repositoryPath, "rev-parse", "--show-toplevel") if err != nil { return Repository{}, archivedSource{}, fmt.Errorf("locating Git repository: %w", err) @@ -321,7 +322,7 @@ func readArchive(reader *tar.Reader, workspace string, expected map[string]archi if header.Typeflag == tar.TypeSymlink || header.Typeflag == tar.TypeLink { continue } - if header.Typeflag != tar.TypeReg && header.Typeflag != tar.TypeRegA { + if header.Typeflag != tar.TypeReg { continue } if !supportedTextPath(name) { @@ -525,8 +526,8 @@ func sortGoFiles(files []retrieval.File) { } type limitedBuffer struct { - mu sync.Mutex buffer bytes.Buffer + mu sync.Mutex limit int truncated bool } diff --git a/validation/internal/retrievalstudy/golist.go b/validation/internal/retrievalstudy/golist.go index 268312d..216ae2f 100644 --- a/validation/internal/retrievalstudy/golist.go +++ b/validation/internal/retrievalstudy/golist.go @@ -82,10 +82,10 @@ func parseGoPackageInventory(data []byte) (int, int, error) { decoder := json.NewDecoder(bytes.NewReader(data)) for { var record struct { - ImportPath string `json:"ImportPath"` - Incomplete bool `json:"Incomplete"` - Error *packageError `json:"Error"` - DepsErrors []packageError `json:"DepsErrors"` + Error *packageError `json:"Error"` + ImportPath string `json:"ImportPath"` + DepsErrors []packageError `json:"DepsErrors"` + Incomplete bool `json:"Incomplete"` } if err := decoder.Decode(&record); err != nil { if err == io.EOF { diff --git a/validation/internal/retrievalstudy/gopls.go b/validation/internal/retrievalstudy/gopls.go index d025298..a8734ec 100644 --- a/validation/internal/retrievalstudy/gopls.go +++ b/validation/internal/retrievalstudy/gopls.go @@ -23,8 +23,8 @@ type goplsLocation struct { type goplsSymbol struct { Name string `json:"name"` ContainerName string `json:"containerName"` - Kind int `json:"kind"` Location goplsLocation `json:"location"` + Kind int `json:"kind"` } type goplsSession struct { diff --git a/validation/internal/retrievalstudy/identity.go b/validation/internal/retrievalstudy/identity.go index 530e8bf..5ecdb3f 100644 --- a/validation/internal/retrievalstudy/identity.go +++ b/validation/internal/retrievalstudy/identity.go @@ -28,14 +28,14 @@ func sourceIdentity(parent context.Context, sourceRepository string, timeout tim return Reproducibility{}, fmt.Errorf("reading study source commit: %w", err) } dirtyDigest := sha256.New() - if _, err := dirtyDigest.Write([]byte("agentic-go-dirty-diff/v1\x00")); err != nil { - return Reproducibility{}, err + if _, writeErr := dirtyDigest.Write([]byte("agentic-go-dirty-diff/v1\x00")); writeErr != nil { + return Reproducibility{}, writeErr } - if err := hashGitOutput(parent, timeout, root, dirtyDigest, "diff", "--binary", "HEAD", "--"); err != nil { - return Reproducibility{}, fmt.Errorf("hashing tracked study-source changes: %w", err) + if hashErr := hashGitOutput(parent, timeout, root, dirtyDigest, "diff", "--binary", "HEAD", "--"); hashErr != nil { + return Reproducibility{}, fmt.Errorf("hashing tracked study-source changes: %w", hashErr) } - if _, err := dirtyDigest.Write([]byte("\x00untracked-files\x00")); err != nil { - return Reproducibility{}, err + if _, writeErr := dirtyDigest.Write([]byte("\x00untracked-files\x00")); writeErr != nil { + return Reproducibility{}, writeErr } untracked, err := gitTextBytes(parent, timeout, root, "ls-files", "--others", "--exclude-standard", "-z") if err != nil { @@ -64,28 +64,28 @@ func sourceIdentity(parent context.Context, sourceRepository string, timeout tim return Reproducibility{}, fmt.Errorf("inspecting an untracked study-source file: %w", err) } if info.Mode()&os.ModeSymlink != 0 { - target, err := os.Readlink(filename) - if err != nil { - return Reproducibility{}, fmt.Errorf("reading an untracked symlink: %w", err) + target, readlinkErr := os.Readlink(filename) + if readlinkErr != nil { + return Reproducibility{}, fmt.Errorf("reading an untracked symlink: %w", readlinkErr) } - if _, err := dirtyDigest.Write([]byte("symlink\x00")); err != nil { - return Reproducibility{}, err + if _, writeErr := dirtyDigest.Write([]byte("symlink\x00")); writeErr != nil { + return Reproducibility{}, writeErr } - if err := writeField(dirtyDigest, []byte(target)); err != nil { - return Reproducibility{}, err + if fieldErr := writeField(dirtyDigest, []byte(target)); fieldErr != nil { + return Reproducibility{}, fieldErr } continue } if !info.Mode().IsRegular() { return Reproducibility{}, fmt.Errorf("untracked study-source entry is not a regular file") } - if _, err := dirtyDigest.Write([]byte("file\x00")); err != nil { - return Reproducibility{}, err + if _, writeErr := dirtyDigest.Write([]byte("file\x00")); writeErr != nil { + return Reproducibility{}, writeErr } binaryLength := make([]byte, 8) binary.BigEndian.PutUint64(binaryLength, uint64(info.Size())) - if _, err := dirtyDigest.Write(binaryLength); err != nil { - return Reproducibility{}, err + if _, writeErr := dirtyDigest.Write(binaryLength); writeErr != nil { + return Reproducibility{}, writeErr } file, err := os.Open(filename) if err != nil { diff --git a/validation/internal/retrievalstudy/manifest.go b/validation/internal/retrievalstudy/manifest.go index a448c63..1a23aa6 100644 --- a/validation/internal/retrievalstudy/manifest.go +++ b/validation/internal/retrievalstudy/manifest.go @@ -33,11 +33,14 @@ func LoadManifest(filename string) (Manifest, string, error) { if err != nil { return Manifest{}, "", fmt.Errorf("opening manifest: %w", err) } - defer file.Close() data, err := io.ReadAll(io.LimitReader(file, (4<<20)+1)) + closeErr := file.Close() if err != nil { return Manifest{}, "", fmt.Errorf("reading manifest: %w", err) } + if closeErr != nil { + return Manifest{}, "", fmt.Errorf("closing manifest: %w", closeErr) + } if len(data) > 4<<20 { return Manifest{}, "", fmt.Errorf("manifest exceeds 4 MiB") } @@ -57,6 +60,7 @@ func LoadManifest(filename string) (Manifest, string, error) { return manifest, hex.EncodeToString(digest[:]), nil } +// Validate checks the manifest fields and their source-independent constraints. func (manifest Manifest) Validate() error { if manifest.Version != ManifestVersion { return fmt.Errorf("manifest version must be %q", ManifestVersion) diff --git a/validation/internal/retrievalstudy/native.go b/validation/internal/retrievalstudy/native.go index 30a0e18..026ba94 100644 --- a/validation/internal/retrievalstudy/native.go +++ b/validation/internal/retrievalstudy/native.go @@ -14,11 +14,6 @@ import ( "time" ) -const ( - nativeOutputLimitDefault = int64(64 << 20) - nativeLineLimitDefault = 100000 -) - var nativeGlobs = []string{ "*.go", "*.md", "*.rst", "*.txt", "*.yaml", "*.yml", "*.json", "*.toml", "*.proto", "*.mod", "*.sum", "*.work", "Makefile", "GNUmakefile", @@ -31,6 +26,7 @@ type nativeLimits struct { lineCount int } +//nolint:govet // Keep the workflow scores adjacent to their timing breakdowns. type nativeMeasurement struct { ranking Ranking totalLatency Latency @@ -60,7 +56,7 @@ func ProbeRG(parent context.Context, timeout time.Duration) (string, error) { return line, nil } -func RunNativeRG(parent context.Context, workspace string, query Query, source archivedSource, gold []GoldSpan, repetitions int, limits nativeLimits) (nativeMeasurement, error) { +func runNativeRG(parent context.Context, workspace string, query Query, source archivedSource, gold []GoldSpan, repetitions int, limits nativeLimits) (nativeMeasurement, error) { tokens := uniqueSortedTokens(query.Text) commandSamples := make([]float64, 0, repetitions) rankingSamples := make([]float64, 0, repetitions) @@ -186,9 +182,7 @@ func gatherRG(parent context.Context, workspace string, tokens []string, source reason = "ripgrep exceeded the per-query timeout" } else if waitErr != nil && complete { var exitErr *exec.ExitError - if errors.As(waitErr, &exitErr) && exitErr.ExitCode() == 1 { - // ripgrep uses exit code 1 for a successful search with no matches. - } else { + if !errors.As(waitErr, &exitErr) || exitErr.ExitCode() != 1 { complete = false reason = "ripgrep exited unsuccessfully" } diff --git a/validation/internal/retrievalstudy/run.go b/validation/internal/retrievalstudy/run.go index 5ed1ed8..72d9028 100644 --- a/validation/internal/retrievalstudy/run.go +++ b/validation/internal/retrievalstudy/run.go @@ -3,7 +3,9 @@ package retrievalstudy import ( "context" "encoding/json" + "errors" "fmt" + "io/fs" "math" "os" "path/filepath" @@ -16,22 +18,24 @@ import ( const retrievalGranularity = "go_declaration_name_anchor" const candidatePoolAuditLimit = 10_000 +// Options configures one pinned, model-free retrieval screen. type Options struct { - Manifest string - RepositoryPath string + Manifest string + RepositoryPath string SourceRepositoryPath string - Output string - GoplsBinary string - TextAblation bool - Repetitions int - Timeout time.Duration - MaxSourceBytes int64 - MaxFileBytes int64 - RGOutputBytes int64 - RGLineLimit int + Output string + GoplsBinary string + TextAblation bool + Repetitions int + Timeout time.Duration + MaxSourceBytes int64 + MaxFileBytes int64 + RGOutputBytes int64 + RGLineLimit int } -func Execute(parent context.Context, options Options) error { +// Execute runs a retrieval screen and writes its provenance-bound report. +func Execute(parent context.Context, options Options) (returnErr error) { if err := validateOptions(options); err != nil { return err } @@ -50,13 +54,17 @@ func Execute(parent context.Context, options Options) error { if err != nil { return fmt.Errorf("creating temporary benchmark workspace: %w", err) } - defer os.RemoveAll(temporaryRoot) + defer func() { + if cleanupErr := os.RemoveAll(temporaryRoot); cleanupErr != nil { + returnErr = errors.Join(returnErr, fmt.Errorf("removing temporary benchmark workspace: %w", cleanupErr)) + } + }() workspace := filepath.Join(temporaryRoot, "snapshot") - if err := os.Mkdir(workspace, 0o700); err != nil { + if err = os.Mkdir(workspace, 0o700); err != nil { return fmt.Errorf("creating snapshot workspace: %w", err) } archiveStarted := time.Now() - repository, source, err := ExportCommit(parent, options.RepositoryPath, workspace, manifest, archiveLimits{ + repository, source, err := exportCommit(parent, options.RepositoryPath, workspace, manifest, archiveLimits{ maxSourceBytes: options.MaxSourceBytes, maxFileBytes: options.MaxFileBytes, timeout: options.Timeout, @@ -66,7 +74,7 @@ func Execute(parent context.Context, options Options) error { if err != nil { return err } - if err := validateGold(source, manifest); err != nil { + if err = validateGold(source, manifest); err != nil { return err } goPackages, err := inventoryGoPackages(parent, workspace, options.Timeout) @@ -107,12 +115,9 @@ func Execute(parent context.Context, options Options) error { queryResults := make([]QueryResult, 0, len(manifest.Queries)) var firstSearchResult *retrieval.Result var allGoplsLatencies []float64 - var allNativeLatencies []float64 - var allNativeGatherLatencies []float64 - var allNativeRankingLatencies []float64 var goplsCompletedQueries int for _, query := range manifest.Queries { - if err := parent.Err(); err != nil { + if err = parent.Err(); err != nil { return err } retrievalRanking, retrievalTimings, retrievalProfiles, searchResult, err := measureRetrieval(parent, query, source, repository, options) @@ -134,15 +139,12 @@ func Execute(parent context.Context, options Options) error { heapSample(&peakHeap) } - native, err := RunNativeRG(parent, workspace, query, source, query.Gold, options.Repetitions, nativeLimits{ + native, err := runNativeRG(parent, workspace, query, source, query.Gold, options.Repetitions, nativeLimits{ timeout: options.Timeout, outputBytes: options.RGOutputBytes, lineCount: options.RGLineLimit, }) if err != nil { return fmt.Errorf("native rg query %q: %w", query.ID, err) } - allNativeLatencies = append(allNativeLatencies, native.totalLatency.Samples...) - allNativeGatherLatencies = append(allNativeGatherLatencies, native.commandLatency.Samples...) - allNativeRankingLatencies = append(allNativeRankingLatencies, native.rankingLatency.Samples...) heapSample(&peakHeap) queryResult := QueryResult{ @@ -280,10 +282,10 @@ func validateGold(source archivedSource, manifest Manifest) error { } type searchObservation struct { + err error result retrieval.Result profile retrieval.SearchProfile wallTime float64 - err error } func measureRetrieval(parent context.Context, query Query, source archivedSource, repository Repository, options Options) (Ranking, RetrievalTimings, RetrievalProfiles, *retrieval.Result, error) { @@ -623,7 +625,7 @@ func summarizeRetrievalStages(samples []RetrievalProfileSample) RetrievalStageLa } } -func writeReport(filename string, report Report) error { +func writeReport(filename string, report Report) (returnErr error) { if err := os.MkdirAll(filepath.Dir(filename), 0o755); err != nil { return fmt.Errorf("creating result directory: %w", err) } @@ -632,7 +634,11 @@ func writeReport(filename string, report Report) error { return fmt.Errorf("creating result file: %w", err) } temporaryName := file.Name() - defer os.Remove(temporaryName) + defer func() { + if removeErr := os.Remove(temporaryName); removeErr != nil && !errors.Is(removeErr, fs.ErrNotExist) { + returnErr = errors.Join(returnErr, fmt.Errorf("removing temporary result report: %w", removeErr)) + } + }() if err := file.Chmod(0o644); err != nil { _ = file.Close() return fmt.Errorf("setting result file permissions: %w", err) diff --git a/validation/internal/retrievalstudy/types.go b/validation/internal/retrievalstudy/types.go index fa4926d..9b39d0b 100644 --- a/validation/internal/retrievalstudy/types.go +++ b/validation/internal/retrievalstudy/types.go @@ -4,8 +4,10 @@ package retrievalstudy import "time" +// ManifestVersion identifies the accepted retrieval-study manifest schema. const ManifestVersion = "agentic-go.retrieval-study/v1" +// Manifest describes one repository snapshot and its reviewed retrieval questions. type Manifest struct { Version string `json:"version"` RepositoryID string `json:"repository_id"` @@ -15,6 +17,7 @@ type Manifest struct { Queries []Query `json:"queries"` } +// Query is one held-out question and its independently reviewed gold spans. type Query struct { ID string `json:"id"` Text string `json:"query"` @@ -22,6 +25,7 @@ type Query struct { Gold []GoldSpan `json:"gold"` } +// GoldSpan identifies a source range that is relevant evidence for a query. type GoldSpan struct { Type string `json:"type"` Path string `json:"path"` @@ -30,12 +34,14 @@ type GoldSpan struct { Label string `json:"label,omitempty"` } +// Repository records the exact repository objects used by a study. type Repository struct { ID string `json:"id"` Commit string `json:"commit"` Tree string `json:"tree"` } +// Reproducibility records the source, manifest, and retrieval-code hashes. type Reproducibility struct { StudySourceCommit string `json:"study_source_commit"` DirtyDiffSHA256 string `json:"dirty_diff_sha256"` @@ -43,6 +49,7 @@ type Reproducibility struct { RetrievalSourceSHA256 string `json:"retrieval_source_sha256"` } +// Coverage records which committed source files were extracted and indexed. type Coverage struct { TrackedRegularFiles int `json:"tracked_regular_files"` TrackedRegularBytes int64 `json:"tracked_regular_bytes"` @@ -69,6 +76,7 @@ type Coverage struct { TextIndexedByRetrieval int `json:"text_indexed_by_retrieval"` } +// TextCandidateIndexCoverage describes bounded text capture for the private ablation. type TextCandidateIndexCoverage struct { Status string `json:"status"` Reason string `json:"reason,omitempty"` @@ -84,6 +92,7 @@ type TextCandidateIndexCoverage struct { MaximumFragments int `json:"maximum_fragments_per_query"` } +// GoPackageInventory reports the bounded `go list` inventory result. type GoPackageInventory struct { Status string `json:"status"` Command string `json:"command"` @@ -98,6 +107,7 @@ type GoPackageInventory struct { Coverage string `json:"coverage"` } +// Candidate is one source anchor returned by a retrieval workflow. type Candidate struct { Path string `json:"path"` Line int `json:"line"` @@ -109,6 +119,7 @@ type Candidate struct { MatchedOccurrences int `json:"matched_occurrences,omitempty"` } +// CutoffMetrics scores one ranked candidate list at a fixed result limit. type CutoffMetrics struct { RetrievedCandidates int `json:"retrieved_candidates"` RelevantCandidates int `json:"relevant_candidates"` @@ -119,11 +130,13 @@ type CutoffMetrics struct { MeanReciprocalRank float64 `json:"mean_reciprocal_rank"` } +// Metrics contains scores at the two benchmark cutoffs. type Metrics struct { At5 CutoffMetrics `json:"at_5"` At10 CutoffMetrics `json:"at_10"` } +// EvidenceMisses counts reviewed spans omitted at each candidate cutoff. type EvidenceMisses struct { GoldSpans int `json:"gold_spans"` At5 int `json:"missed_at_5"` @@ -132,6 +145,7 @@ type EvidenceMisses struct { ByTypeAt10 map[string]int `json:"missed_at_10_by_type"` } +// Ranking is the status, scores, and candidates for one workflow and query. type Ranking struct { Status string `json:"status"` Complete bool `json:"complete"` @@ -146,6 +160,7 @@ type Ranking struct { Tokens []string `json:"tokens,omitempty"` } +// CandidatePoolAudit measures whether reviewed spans occur before top-k ranking. type CandidatePoolAudit struct { Status string `json:"status"` Complete bool `json:"complete"` @@ -158,6 +173,7 @@ type CandidatePoolAudit struct { IncompleteReason string `json:"incomplete_reason,omitempty"` } +// TextCandidateResult records the evaluation-only mixed text and Go ranking. type TextCandidateResult struct { Ranking Ranking `json:"ranking"` CandidatePool CandidatePoolAudit `json:"candidate_pool"` @@ -168,6 +184,7 @@ type TextCandidateResult struct { MaximumTextFragments int `json:"maximum_text_fragments"` } +// Latency stores measured samples and their summary statistics in milliseconds. type Latency struct { Samples []float64 `json:"samples_ms"` P50MS float64 `json:"p50_ms"` @@ -176,11 +193,13 @@ type Latency struct { MaxMS float64 `json:"max_ms"` } +// RetrievalTimings separates cold and warm query durations by result limit. type RetrievalTimings struct { Cold map[string]Latency `json:"cold_by_limit"` Warm map[string]Latency `json:"warm_by_limit"` } +// RetrievalProfileSample stores one retrieval call's stage timings and file counts. type RetrievalProfileSample struct { SearchMS float64 `json:"search_ms"` ParseMS float64 `json:"parse_ms"` @@ -191,6 +210,7 @@ type RetrievalProfileSample struct { FilesParsed int `json:"files_parsed"` } +// RetrievalProfile aggregates stage timings for retrieval calls. type RetrievalProfile struct { Samples []RetrievalProfileSample `json:"samples"` Search Latency `json:"search"` @@ -199,11 +219,13 @@ type RetrievalProfile struct { Rank Latency `json:"rank"` } +// RetrievalProfiles groups cold and warm stage profiles by result limit. type RetrievalProfiles struct { Cold map[string]RetrievalProfile `json:"cold_by_limit"` Warm map[string]RetrievalProfile `json:"warm_by_limit"` } +// GoplsMeasurement records the optional workspace-symbol workflow and its limits. type GoplsMeasurement struct { Status string `json:"status"` Version string `json:"version,omitempty"` @@ -214,6 +236,7 @@ type GoplsMeasurement struct { Granularity string `json:"candidate_granularity"` } +// QueryResult combines gold spans, rankings, coverage, and timings for one query. type QueryResult struct { ID string `json:"id"` Text string `json:"query"` @@ -232,6 +255,7 @@ type QueryResult struct { GoplsQueryLatency Latency `json:"gopls_query_latency,omitempty"` } +// AggregateMetrics contains macro and micro scores across complete queries. type AggregateMetrics struct { Queries int `json:"queries"` MetricsQueries int `json:"metrics_queries"` @@ -243,11 +267,13 @@ type AggregateMetrics struct { MicroPrecision float64 `json:"micro_precision"` } +// ArmSummary contains aggregate scores for both result cutoffs. type ArmSummary struct { At5 AggregateMetrics `json:"at_5"` At10 AggregateMetrics `json:"at_10"` } +// RetrievalStageLatencySummary reports search, parse, aggregation, and rank timings. type RetrievalStageLatencySummary struct { SearchP50MS float64 `json:"search_p50_ms"` SearchP95MS float64 `json:"search_p95_ms"` @@ -259,11 +285,13 @@ type RetrievalStageLatencySummary struct { RankP95MS float64 `json:"rank_p95_ms"` } +// RetrievalStageSummary groups cold and warm stage timings by result limit. type RetrievalStageSummary struct { Cold map[string]RetrievalStageLatencySummary `json:"cold_by_limit"` Warm map[string]RetrievalStageLatencySummary `json:"warm_by_limit"` } +// Summary contains aggregate retrieval, native-tool, gopls, and latency results. type Summary struct { Retrieval ArmSummary `json:"retrieval"` NativeRG ArmSummary `json:"native_rg"` @@ -284,6 +312,7 @@ type Summary struct { GoplsQueryP95MS *float64 `json:"gopls_query_p95_ms,omitempty"` } +// HeapStats stores sampled Go runtime heap values for the benchmark process. type HeapStats struct { StartHeapAllocBytes uint64 `json:"start_heap_alloc_bytes"` EndHeapAllocBytes uint64 `json:"end_heap_alloc_bytes"` @@ -293,6 +322,7 @@ type HeapStats struct { NumGC uint32 `json:"num_gc"` } +// Report is the complete provenance-bound result of one retrieval screen. type Report struct { SchemaVersion string `json:"schema_version"` CreatedUTC time.Time `json:"created_utc"` @@ -316,6 +346,7 @@ type Report struct { Heap HeapStats `json:"heap"` } +// NativeWorkflow describes the fixed ripgrep and local ranking procedure. type NativeWorkflow struct { CommandTemplate string `json:"command_template"` Tokenizer string `json:"tokenizer"` @@ -324,6 +355,7 @@ type NativeWorkflow struct { PathScope string `json:"path_scope"` } +// Configuration records the command bounds and repetitions used in a report. type Configuration struct { Repetitions int `json:"repetitions"` CommandTimeoutMS int64 `json:"command_timeout_ms"` From b2e75e2e9cfb5e02345bb1c9e2dfa0403f867630 Mon Sep 17 00:00:00 2001 From: Ashwin Gopalsamy <47941624+ashwingopalsamy@users.noreply.github.com> Date: Sun, 27 Sep 2026 13:25:15 +0530 Subject: [PATCH 20/20] fix(ci): resolve retrieval study lint findings --- internal/sourceview/sourceview_test.go | 2 +- validation/internal/retrievalstudy/archive.go | 22 +- validation/internal/retrievalstudy/golist.go | 2 +- validation/internal/retrievalstudy/gopls.go | 2 +- .../internal/retrievalstudy/identity.go | 25 +- .../internal/retrievalstudy/manifest.go | 23 +- validation/internal/retrievalstudy/native.go | 6 +- validation/internal/retrievalstudy/run.go | 50 +-- .../internal/retrievalstudy/run_test.go | 6 +- validation/internal/retrievalstudy/types.go | 340 ++++++++++-------- 10 files changed, 252 insertions(+), 226 deletions(-) diff --git a/internal/sourceview/sourceview_test.go b/internal/sourceview/sourceview_test.go index 2676783..6da3808 100644 --- a/internal/sourceview/sourceview_test.go +++ b/internal/sourceview/sourceview_test.go @@ -169,7 +169,7 @@ func TestCreateRequiresNewOutputOutsideSourceRepository(t *testing.T) { source, runner := sourceViewRunner(t, repository) for name, output := range map[string]string{ "inside repository": filepath.Join(repository, "nested-view"), - "existing path": repository, + "existing path": repository, } { t.Run(name, func(t *testing.T) { if _, err := Create(context.Background(), runner, source, Request{OutputPath: output}); err == nil { diff --git a/validation/internal/retrievalstudy/archive.go b/validation/internal/retrievalstudy/archive.go index 180818d..1cf0791 100644 --- a/validation/internal/retrievalstudy/archive.go +++ b/validation/internal/retrievalstudy/archive.go @@ -26,14 +26,14 @@ import ( ) const ( - maximumTextCandidateFiles = 100_000 - maximumTextCandidateFileSize = 1 << 20 - maximumTextCandidateBytes = 64 << 20 + maximumTextCandidateFiles = 100_000 + maximumTextCandidateFileSize = 1 << 20 + maximumTextCandidateBytes = 64 << 20 maximumTextCandidateFragments = retrieval.MaximumTextLineFragments ) type sourceMeta struct { - lines int + lines int goFile bool } @@ -181,11 +181,11 @@ func archiveSelection(parent context.Context, repositoryRoot, commit string, tim } separator := bytes.IndexByte(record, '\t') if separator < 0 { - return nil, Coverage{}, nil, fmt.Errorf("Git tree entry has an invalid record") + return nil, Coverage{}, nil, fmt.Errorf("git tree entry has an invalid record") } metadata := strings.Fields(string(record[:separator])) if len(metadata) < 3 { - return nil, Coverage{}, nil, fmt.Errorf("Git tree entry has incomplete metadata") + return nil, Coverage{}, nil, fmt.Errorf("git tree entry has incomplete metadata") } mode, objectType := metadata[0], metadata[1] name := string(record[separator+1:]) @@ -204,11 +204,11 @@ func archiveSelection(parent context.Context, repositoryRoot, commit string, tim continue } if len(metadata) < 4 { - return nil, Coverage{}, nil, fmt.Errorf("Git tree entry has no blob size") + return nil, Coverage{}, nil, fmt.Errorf("git tree entry has no blob size") } size, parseErr := strconv.ParseInt(metadata[3], 10, 64) if parseErr != nil || size < 0 { - return nil, Coverage{}, nil, fmt.Errorf("Git tree entry has an invalid blob size") + return nil, Coverage{}, nil, fmt.Errorf("git tree entry has an invalid blob size") } coverage.TrackedRegularFiles++ coverage.TrackedRegularBytes += size @@ -432,21 +432,21 @@ func readArchive(reader *tar.Reader, workspace string, expected map[string]archi return source, nil } -var errArchivePathEncoding = errors.New("Git archive path is not valid UTF-8") +var errArchivePathEncoding = errors.New("git archive path is not valid UTF-8") func archivePath(headerName string) (string, bool, error) { if headerName == "repo/" || headerName == "repo" { return "", true, nil } if !strings.HasPrefix(headerName, "repo/") { - return "", false, fmt.Errorf("Git archive entry %q escaped its expected prefix", headerName) + return "", false, fmt.Errorf("git archive entry %q escaped its expected prefix", headerName) } name := strings.TrimSuffix(strings.TrimPrefix(headerName, "repo/"), "/") if name == "" { return "", true, nil } if !filepath.IsLocal(filepath.FromSlash(name)) || path.Clean(name) != name { - return "", false, fmt.Errorf("Git archive contains an unsafe path") + return "", false, fmt.Errorf("git archive contains an unsafe path") } if !utf8.ValidString(name) { return "", false, errArchivePathEncoding diff --git a/validation/internal/retrievalstudy/golist.go b/validation/internal/retrievalstudy/golist.go index 216ae2f..8460fe8 100644 --- a/validation/internal/retrievalstudy/golist.go +++ b/validation/internal/retrievalstudy/golist.go @@ -4,9 +4,9 @@ import ( "bytes" "context" "encoding/json" + "io" "os" "os/exec" - "io" "strings" "time" ) diff --git a/validation/internal/retrievalstudy/gopls.go b/validation/internal/retrievalstudy/gopls.go index a8734ec..2c193f8 100644 --- a/validation/internal/retrievalstudy/gopls.go +++ b/validation/internal/retrievalstudy/gopls.go @@ -73,7 +73,7 @@ func (session *goplsSession) search(parent context.Context, timeout time.Duratio } samples := make([]float64, 0, repetitions) var candidates []Candidate - var complete = true + complete := true var reason string for repetition := 0; repetition < repetitions; repetition++ { if err := parent.Err(); err != nil { diff --git a/validation/internal/retrievalstudy/identity.go b/validation/internal/retrievalstudy/identity.go index 5ecdb3f..b0301aa 100644 --- a/validation/internal/retrievalstudy/identity.go +++ b/validation/internal/retrievalstudy/identity.go @@ -49,19 +49,19 @@ func sourceIdentity(parent context.Context, sourceRepository string, timeout tim } sort.Strings(paths) for _, relative := range paths { - if err := parent.Err(); err != nil { - return Reproducibility{}, err + if contextErr := parent.Err(); contextErr != nil { + return Reproducibility{}, contextErr } if !filepath.IsLocal(filepath.FromSlash(relative)) || path.Clean(relative) != relative { return Reproducibility{}, fmt.Errorf("untracked study-source path is unsafe") } - if err := writeField(dirtyDigest, []byte(relative)); err != nil { - return Reproducibility{}, err + if fieldErr := writeField(dirtyDigest, []byte(relative)); fieldErr != nil { + return Reproducibility{}, fieldErr } filename := filepath.Join(root, filepath.FromSlash(relative)) - info, err := os.Lstat(filename) - if err != nil { - return Reproducibility{}, fmt.Errorf("inspecting an untracked study-source file: %w", err) + info, statErr := os.Lstat(filename) + if statErr != nil { + return Reproducibility{}, fmt.Errorf("inspecting an untracked study-source file: %w", statErr) } if info.Mode()&os.ModeSymlink != 0 { target, readlinkErr := os.Readlink(filename) @@ -87,9 +87,9 @@ func sourceIdentity(parent context.Context, sourceRepository string, timeout tim if _, writeErr := dirtyDigest.Write(binaryLength); writeErr != nil { return Reproducibility{}, writeErr } - file, err := os.Open(filename) - if err != nil { - return Reproducibility{}, fmt.Errorf("opening an untracked study-source file: %w", err) + file, openErr := os.Open(filename) + if openErr != nil { + return Reproducibility{}, fmt.Errorf("opening an untracked study-source file: %w", openErr) } _, copyErr := copyContext(parent, dirtyDigest, file) closeErr := file.Close() @@ -107,8 +107,9 @@ func sourceIdentity(parent context.Context, sourceRepository string, timeout tim retrievalDigest := sha256.Sum256(retrievalSource) dirtyHex := hex.EncodeToString(dirtyDigest.Sum(nil)) return Reproducibility{ - StudySourceCommit: strings.TrimSpace(commit), - DirtyDiffSHA256: dirtyHex, ManifestSHA256: manifestSHA256, + StudySourceCommit: strings.TrimSpace(commit), + DirtyDiffSHA256: dirtyHex, + ManifestSHA256: manifestSHA256, RetrievalSourceSHA256: hex.EncodeToString(retrievalDigest[:]), }, nil } diff --git a/validation/internal/retrievalstudy/manifest.go b/validation/internal/retrievalstudy/manifest.go index 1a23aa6..4197e76 100644 --- a/validation/internal/retrievalstudy/manifest.go +++ b/validation/internal/retrievalstudy/manifest.go @@ -2,9 +2,9 @@ package retrievalstudy import ( "bytes" - "encoding/json" "crypto/sha256" "encoding/hex" + "encoding/json" "fmt" "io" "io/fs" @@ -15,16 +15,17 @@ import ( "unicode" ) -var fullCommitPattern = regexp.MustCompile(`^(?:[0-9a-f]{40}|[0-9a-f]{64})$`) -var repositoryIDPattern = regexp.MustCompile(`^[a-z0-9][a-z0-9.:-]*/[a-z0-9_.-]+(/[a-z0-9_.-]+)*$`) - -var evidenceTypes = map[string]struct{}{ - "declaration": {}, - "caller": {}, - "test": {}, - "documentation": {}, - "configuration": {}, -} +var ( + fullCommitPattern = regexp.MustCompile(`^(?:[0-9a-f]{40}|[0-9a-f]{64})$`) + repositoryIDPattern = regexp.MustCompile(`^[a-z0-9][a-z0-9.:-]*/[a-z0-9_.-]+(/[a-z0-9_.-]+)*$`) + evidenceTypes = map[string]struct{}{ + "declaration": {}, + "caller": {}, + "test": {}, + "documentation": {}, + "configuration": {}, + } +) // LoadManifest decodes a strict, versioned JSON manifest and validates all // fields that do not depend on the archived source tree. diff --git a/validation/internal/retrievalstudy/native.go b/validation/internal/retrievalstudy/native.go index 026ba94..72dcfd4 100644 --- a/validation/internal/retrievalstudy/native.go +++ b/validation/internal/retrievalstudy/native.go @@ -36,8 +36,8 @@ type nativeMeasurement struct { type rgLine struct { path string - line int text string + line int } // ProbeRG returns the first version line without recording the executable path. @@ -99,8 +99,8 @@ func runNativeRG(parent context.Context, workspace string, query Query, source a } } return nativeMeasurement{ - ranking: ranking, - totalLatency: latency(totalSamples), + ranking: ranking, + totalLatency: latency(totalSamples), commandLatency: latency(commandSamples), rankingLatency: latency(rankingSamples), }, nil diff --git a/validation/internal/retrievalstudy/run.go b/validation/internal/retrievalstudy/run.go index 72d9028..7825bf2 100644 --- a/validation/internal/retrievalstudy/run.go +++ b/validation/internal/retrievalstudy/run.go @@ -15,8 +15,10 @@ import ( "github.com/agentic-mcps/go/internal/intelligence/retrieval" ) -const retrievalGranularity = "go_declaration_name_anchor" -const candidatePoolAuditLimit = 10_000 +const ( + retrievalGranularity = "go_declaration_name_anchor" + candidatePoolAuditLimit = 10_000 +) // Options configures one pinned, model-free retrieval screen. type Options struct { @@ -65,9 +67,9 @@ func Execute(parent context.Context, options Options) (returnErr error) { } archiveStarted := time.Now() repository, source, err := exportCommit(parent, options.RepositoryPath, workspace, manifest, archiveLimits{ - maxSourceBytes: options.MaxSourceBytes, - maxFileBytes: options.MaxFileBytes, - timeout: options.Timeout, + maxSourceBytes: options.MaxSourceBytes, + maxFileBytes: options.MaxFileBytes, + timeout: options.Timeout, indexTextCandidates: options.TextAblation, }) archiveMS := float64(time.Since(archiveStarted)) / float64(time.Millisecond) @@ -120,9 +122,9 @@ func Execute(parent context.Context, options Options) (returnErr error) { if err = parent.Err(); err != nil { return err } - retrievalRanking, retrievalTimings, retrievalProfiles, searchResult, err := measureRetrieval(parent, query, source, repository, options) - if err != nil { - return err + retrievalRanking, retrievalTimings, retrievalProfiles, searchResult, retrievalErr := measureRetrieval(parent, query, source, repository, options) + if retrievalErr != nil { + return retrievalErr } if searchResult != nil && firstSearchResult == nil { copyResult := *searchResult @@ -139,21 +141,21 @@ func Execute(parent context.Context, options Options) (returnErr error) { heapSample(&peakHeap) } - native, err := runNativeRG(parent, workspace, query, source, query.Gold, options.Repetitions, nativeLimits{ + native, nativeErr := runNativeRG(parent, workspace, query, source, query.Gold, options.Repetitions, nativeLimits{ timeout: options.Timeout, outputBytes: options.RGOutputBytes, lineCount: options.RGLineLimit, }) - if err != nil { - return fmt.Errorf("native rg query %q: %w", query.ID, err) + if nativeErr != nil { + return fmt.Errorf("native rg query %q: %w", query.ID, nativeErr) } heapSample(&peakHeap) queryResult := QueryResult{ ID: query.ID, Text: query.Text, Gold: append([]GoldSpan(nil), query.Gold...), Retrieval: retrievalRanking, RetrievalTimings: retrievalTimings, - RetrievalProfiles: retrievalProfiles, + RetrievalProfiles: retrievalProfiles, RetrievalCandidatePool: retrievalCandidatePool, - TextCandidateAblation: textCandidateAblation, - NativeRG: native.ranking, NativeRGLatency: native.totalLatency, + TextCandidateAblation: textCandidateAblation, + NativeRG: native.ranking, NativeRGLatency: native.totalLatency, NativeRGCommandLatency: native.commandLatency, NativeRGRankingLatency: native.rankingLatency, } @@ -217,15 +219,15 @@ func Execute(parent context.Context, options Options) (returnErr error) { SchemaVersion: reportVersion, CreatedUTC: created, Repository: repository, Stratum: manifest.Stratum, Reproducibility: reproducibility, - GoVersion: runtime.Version(), GOOS: runtime.GOOS, GOARCH: runtime.GOARCH, + GoVersion: runtime.Version(), GOOS: runtime.GOOS, GOARCH: runtime.GOARCH, RGVersion: rgVersion, SnapshotExportMS: archiveMS, RetrievalTimingNote: "Cache.SearchProfiled reports parse, aggregation and rank stages; its total covers the full Go declaration candidate retrieval call. Top-5 is scored from the prefix of the same top-10 result, so top-5 and top-10 timing samples are identical and the large corpus is not parsed twice for two cutoffs. The optional text candidate ablation uses the same scorer over a bounded mixed candidate pool and is reported separately. This screen measures the candidate retrieval kernel, not full go_context output or semantic resolution.", NativeRGWorkflow: NativeWorkflow{ CommandTemplate: "rg --no-ignore --hidden --glob-case-insensitive --no-heading --with-filename --line-number --color never --fixed-strings --ignore-case --text --null [supported-source globs] -e ... -- .", - Tokenizer: "retrieval Unicode letter/digit/underscore tokenizer with lower-to-upper camel-case splits; duplicate tokens removed and sorted", - Ranking: []string{"distinct query tokens present on line descending", "query-token occurrences on line descending", "repository-relative path ascending", "line ascending"}, - Globs: append([]string(nil), nativeGlobs...), - PathScope: "Git-archive files with supported source suffixes only; archive ignores checkout working-tree changes", + Tokenizer: "retrieval Unicode letter/digit/underscore tokenizer with lower-to-upper camel-case splits; duplicate tokens removed and sorted", + Ranking: []string{"distinct query tokens present on line descending", "query-token occurrences on line descending", "repository-relative path ascending", "line ascending"}, + Globs: append([]string(nil), nativeGlobs...), + PathScope: "Git-archive files with supported source suffixes only; archive ignores checkout working-tree changes", }, GoPackages: goPackages, Coverage: source.coverage, TextCandidateIndex: textCandidateIndex, Configuration: Configuration{ @@ -448,7 +450,7 @@ func measureTextAblation(parent context.Context, query Query, source archivedSou Ranking: ranking, CandidatePool: textPool, Latency: latency([]float64{wallMS}), Profile: textProfile, IndexedTextFragments: textResult.TextIndexedFragments, - SkippedTextFiles: textResult.TextSkippedFiles, + SkippedTextFiles: textResult.TextSkippedFiles, MaximumTextFragments: retrieval.MaximumTextLineFragments, }, nil } @@ -546,7 +548,7 @@ func unavailableRanking(reason string) Ranking { Status: "unavailable", Complete: false, IncompleteReason: reason, Candidates: []Candidate{}, CandidateCountComplete: false, Granularity: retrievalGranularity, - Metrics: Metrics{}, Misses: EvidenceMisses{ByTypeAt5: map[string]int{}, ByTypeAt10: map[string]int{}}, + Metrics: Metrics{}, Misses: EvidenceMisses{ByTypeAt5: map[string]int{}, ByTypeAt10: map[string]int{}}, } } @@ -577,9 +579,9 @@ func summarize(queries []QueryResult) (Summary, error) { goplsQuery = append(goplsQuery, query.GoplsQueryLatency.Samples...) } summary := Summary{ - Retrieval: ArmSummary{At5: aggregateRankings(retrievalRankings, 5), At10: aggregateRankings(retrievalRankings, 10)}, - NativeRG: ArmSummary{At5: aggregateRankings(nativeRankings, 5), At10: aggregateRankings(nativeRankings, 10)}, - Gopls: ArmSummary{At5: aggregateRankings(goplsRankings, 5), At10: aggregateRankings(goplsRankings, 10)}, + Retrieval: ArmSummary{At5: aggregateRankings(retrievalRankings, 5), At10: aggregateRankings(retrievalRankings, 10)}, + NativeRG: ArmSummary{At5: aggregateRankings(nativeRankings, 5), At10: aggregateRankings(nativeRankings, 10)}, + Gopls: ArmSummary{At5: aggregateRankings(goplsRankings, 5), At10: aggregateRankings(goplsRankings, 10)}, RetrievalColdP50MS: make(map[string]float64), RetrievalColdP95MS: make(map[string]float64), RetrievalWarmP50MS: make(map[string]float64), RetrievalWarmP95MS: make(map[string]float64), RetrievalStages: RetrievalStageSummary{Cold: make(map[string]RetrievalStageLatencySummary), Warm: make(map[string]RetrievalStageLatencySummary)}, diff --git a/validation/internal/retrievalstudy/run_test.go b/validation/internal/retrievalstudy/run_test.go index e89732a..cd094e3 100644 --- a/validation/internal/retrievalstudy/run_test.go +++ b/validation/internal/retrievalstudy/run_test.go @@ -12,11 +12,11 @@ func TestCandidatePoolAuditSeparatesCoverageFromRanking(t *testing.T) { {Type: "test", Path: "focus_test.go", StartLine: 20, EndLine: 22}, } source := archivedSource{ - coverage: Coverage{SourceArchiveComplete: true}, + coverage: Coverage{SourceArchiveComplete: true}, textCandidateIndex: TextCandidateIndexCoverage{Status: "complete"}, } result := retrieval.Result{ - Candidates: []retrieval.Candidate{{Path: "docs/contracts.md", Line: 11, Kind: "text.line"}}, + Candidates: []retrieval.Candidate{{Path: "docs/contracts.md", Line: 11, Kind: "text.line"}}, CandidateCount: 1, Complete: true, } audit := candidatePoolAudit(result, gold, source, true) @@ -31,7 +31,7 @@ func TestCandidatePoolAuditSeparatesCoverageFromRanking(t *testing.T) { func TestCandidatePoolAuditMarksCappedResultsPartial(t *testing.T) { source := archivedSource{coverage: Coverage{SourceArchiveComplete: true}} result := retrieval.Result{ - Candidates: []retrieval.Candidate{{Path: "docs/contracts.md", Line: 11}}, + Candidates: []retrieval.Candidate{{Path: "docs/contracts.md", Line: 11}}, CandidateCount: candidatePoolAuditLimit + 1, Complete: true, Truncated: true, } audit := candidatePoolAudit(result, []GoldSpan{{Type: "documentation", Path: "docs/contracts.md", StartLine: 10, EndLine: 12}}, source, false) diff --git a/validation/internal/retrievalstudy/types.go b/validation/internal/retrievalstudy/types.go index 9b39d0b..efab1e1 100644 --- a/validation/internal/retrievalstudy/types.go +++ b/validation/internal/retrievalstudy/types.go @@ -26,6 +26,8 @@ type Query struct { } // GoldSpan identifies a source range that is relevant evidence for a query. +// +//nolint:govet // Field order is part of the stable JSON report shape. type GoldSpan struct { Type string `json:"type"` Path string `json:"path"` @@ -43,80 +45,86 @@ type Repository struct { // Reproducibility records the source, manifest, and retrieval-code hashes. type Reproducibility struct { - StudySourceCommit string `json:"study_source_commit"` - DirtyDiffSHA256 string `json:"dirty_diff_sha256"` - ManifestSHA256 string `json:"manifest_sha256"` - RetrievalSourceSHA256 string `json:"retrieval_source_sha256"` + StudySourceCommit string `json:"study_source_commit"` + DirtyDiffSHA256 string `json:"dirty_diff_sha256"` + ManifestSHA256 string `json:"manifest_sha256"` + RetrievalSourceSHA256 string `json:"retrieval_source_sha256"` } // Coverage records which committed source files were extracted and indexed. +// +//nolint:govet // Field order is part of the stable JSON report shape. type Coverage struct { - TrackedRegularFiles int `json:"tracked_regular_files"` - TrackedRegularBytes int64 `json:"tracked_regular_bytes"` - ExpectedSupportedSourceFiles int `json:"expected_supported_source_files"` - ExpectedSupportedSourceBytes int64 `json:"expected_supported_source_bytes"` - SupportedSourceFiles int `json:"supported_source_files"` - SupportedSourceBytes int64 `json:"supported_source_bytes"` - GoFiles int `json:"go_files"` - GoBytes int64 `json:"go_bytes"` - GoFilesWithDeclarations int `json:"go_files_with_declarations"` - GoParseIncompleteFiles int `json:"go_parse_incomplete_files"` - SupportedTextFiles int `json:"supported_text_files"` - SupportedTextBytes int64 `json:"supported_text_bytes"` - RejectedTextFiles int `json:"rejected_text_files"` - RejectedTextBytes int64 `json:"rejected_text_bytes"` - SymlinksNotIndexed int `json:"symlinks_not_indexed"` - OtherArchiveEntriesIgnored int `json:"other_archive_entries_ignored"` - ArchiveOmittedSourceFiles int `json:"archive_omitted_source_files"` - ArchiveOmittedSourceBytes int64 `json:"archive_omitted_source_bytes"` - ArchiveTransformedSourceFiles int `json:"archive_transformed_source_files"` - ArchiveTransformedSourceBytes int64 `json:"archive_transformed_source_bytes"` - SourceArchiveComplete bool `json:"source_archive_complete"` + TrackedRegularFiles int `json:"tracked_regular_files"` + TrackedRegularBytes int64 `json:"tracked_regular_bytes"` + ExpectedSupportedSourceFiles int `json:"expected_supported_source_files"` + ExpectedSupportedSourceBytes int64 `json:"expected_supported_source_bytes"` + SupportedSourceFiles int `json:"supported_source_files"` + SupportedSourceBytes int64 `json:"supported_source_bytes"` + GoFiles int `json:"go_files"` + GoBytes int64 `json:"go_bytes"` + GoFilesWithDeclarations int `json:"go_files_with_declarations"` + GoParseIncompleteFiles int `json:"go_parse_incomplete_files"` + SupportedTextFiles int `json:"supported_text_files"` + SupportedTextBytes int64 `json:"supported_text_bytes"` + RejectedTextFiles int `json:"rejected_text_files"` + RejectedTextBytes int64 `json:"rejected_text_bytes"` + SymlinksNotIndexed int `json:"symlinks_not_indexed"` + OtherArchiveEntriesIgnored int `json:"other_archive_entries_ignored"` + ArchiveOmittedSourceFiles int `json:"archive_omitted_source_files"` + ArchiveOmittedSourceBytes int64 `json:"archive_omitted_source_bytes"` + ArchiveTransformedSourceFiles int `json:"archive_transformed_source_files"` + ArchiveTransformedSourceBytes int64 `json:"archive_transformed_source_bytes"` + SourceArchiveComplete bool `json:"source_archive_complete"` SourceArchiveIncompleteReason string `json:"source_archive_incomplete_reason,omitempty"` - TextIndexedByRetrieval int `json:"text_indexed_by_retrieval"` + TextIndexedByRetrieval int `json:"text_indexed_by_retrieval"` } // TextCandidateIndexCoverage describes bounded text capture for the private ablation. type TextCandidateIndexCoverage struct { - Status string `json:"status"` - Reason string `json:"reason,omitempty"` - SupportedFiles int `json:"supported_files"` - SupportedBytes int64 `json:"supported_bytes"` - IndexedFiles int `json:"indexed_files"` - IndexedBytes int64 `json:"indexed_bytes"` - OmittedFiles int `json:"omitted_files"` - OmittedBytes int64 `json:"omitted_bytes"` - MaximumFiles int `json:"maximum_files"` - MaximumFileBytes int64 `json:"maximum_file_bytes"` - MaximumTotalBytes int64 `json:"maximum_total_bytes"` - MaximumFragments int `json:"maximum_fragments_per_query"` + Status string `json:"status"` + Reason string `json:"reason,omitempty"` + SupportedFiles int `json:"supported_files"` + SupportedBytes int64 `json:"supported_bytes"` + IndexedFiles int `json:"indexed_files"` + IndexedBytes int64 `json:"indexed_bytes"` + OmittedFiles int `json:"omitted_files"` + OmittedBytes int64 `json:"omitted_bytes"` + MaximumFiles int `json:"maximum_files"` + MaximumFileBytes int64 `json:"maximum_file_bytes"` + MaximumTotalBytes int64 `json:"maximum_total_bytes"` + MaximumFragments int `json:"maximum_fragments_per_query"` } // GoPackageInventory reports the bounded `go list` inventory result. +// +//nolint:govet // Field order is part of the stable JSON report shape. type GoPackageInventory struct { - Status string `json:"status"` - Command string `json:"command"` - PackageCount int `json:"package_count"` - PackagesWithErrors int `json:"packages_with_errors"` - LatencyMS float64 `json:"latency_ms"` - TimeoutMS int64 `json:"timeout_ms"` - OutputBytes int `json:"captured_output_bytes"` - OutputLimitBytes int `json:"captured_output_limit_bytes"` - OutputTruncated bool `json:"output_truncated"` - Error string `json:"error,omitempty"` - Coverage string `json:"coverage"` + Status string `json:"status"` + Command string `json:"command"` + PackageCount int `json:"package_count"` + PackagesWithErrors int `json:"packages_with_errors"` + LatencyMS float64 `json:"latency_ms"` + TimeoutMS int64 `json:"timeout_ms"` + OutputBytes int `json:"captured_output_bytes"` + OutputLimitBytes int `json:"captured_output_limit_bytes"` + OutputTruncated bool `json:"output_truncated"` + Error string `json:"error,omitempty"` + Coverage string `json:"coverage"` } // Candidate is one source anchor returned by a retrieval workflow. +// +//nolint:govet // Field order is part of the stable JSON report shape. type Candidate struct { - Path string `json:"path"` - Line int `json:"line"` - Name string `json:"name,omitempty"` - Kind string `json:"kind,omitempty"` - Container string `json:"container,omitempty"` - ProviderKind int `json:"provider_kind,omitempty"` - MatchedUniqueTerms int `json:"matched_unique_terms,omitempty"` - MatchedOccurrences int `json:"matched_occurrences,omitempty"` + Path string `json:"path"` + Line int `json:"line"` + Name string `json:"name,omitempty"` + Kind string `json:"kind,omitempty"` + Container string `json:"container,omitempty"` + ProviderKind int `json:"provider_kind,omitempty"` + MatchedUniqueTerms int `json:"matched_unique_terms,omitempty"` + MatchedOccurrences int `json:"matched_occurrences,omitempty"` } // CutoffMetrics scores one ranked candidate list at a fixed result limit. @@ -137,51 +145,57 @@ type Metrics struct { } // EvidenceMisses counts reviewed spans omitted at each candidate cutoff. +// +//nolint:govet // Field order is part of the stable JSON report shape. type EvidenceMisses struct { - GoldSpans int `json:"gold_spans"` - At5 int `json:"missed_at_5"` - At10 int `json:"missed_at_10"` - ByTypeAt5 map[string]int `json:"missed_at_5_by_type"` + GoldSpans int `json:"gold_spans"` + At5 int `json:"missed_at_5"` + At10 int `json:"missed_at_10"` + ByTypeAt5 map[string]int `json:"missed_at_5_by_type"` ByTypeAt10 map[string]int `json:"missed_at_10_by_type"` } // Ranking is the status, scores, and candidates for one workflow and query. +// +//nolint:govet // Field order is part of the stable JSON report shape. type Ranking struct { - Status string `json:"status"` - Complete bool `json:"complete"` - MetricsUsable bool `json:"metrics_usable"` - IncompleteReason string `json:"incomplete_reason,omitempty"` - Metrics Metrics `json:"metrics"` - Misses EvidenceMisses `json:"misses"` - Candidates []Candidate `json:"candidates"` - CandidateCount int `json:"candidate_count"` - CandidateCountComplete bool `json:"candidate_count_complete"` - Granularity string `json:"candidate_granularity"` - Tokens []string `json:"tokens,omitempty"` + Status string `json:"status"` + Complete bool `json:"complete"` + MetricsUsable bool `json:"metrics_usable"` + IncompleteReason string `json:"incomplete_reason,omitempty"` + Metrics Metrics `json:"metrics"` + Misses EvidenceMisses `json:"misses"` + Candidates []Candidate `json:"candidates"` + CandidateCount int `json:"candidate_count"` + CandidateCountComplete bool `json:"candidate_count_complete"` + Granularity string `json:"candidate_granularity"` + Tokens []string `json:"tokens,omitempty"` } // CandidatePoolAudit measures whether reviewed spans occur before top-k ranking. +// +//nolint:govet // Field order is part of the stable JSON report shape. type CandidatePoolAudit struct { - Status string `json:"status"` - Complete bool `json:"complete"` - CandidateCount int `json:"candidate_count"` - CandidatesObserved int `json:"candidates_observed"` - CandidateLimit int `json:"candidate_limit"` - GoldSpans int `json:"gold_spans"` - HitGoldSpans int `json:"hit_gold_spans"` - Recall float64 `json:"recall"` - IncompleteReason string `json:"incomplete_reason,omitempty"` + Status string `json:"status"` + Complete bool `json:"complete"` + CandidateCount int `json:"candidate_count"` + CandidatesObserved int `json:"candidates_observed"` + CandidateLimit int `json:"candidate_limit"` + GoldSpans int `json:"gold_spans"` + HitGoldSpans int `json:"hit_gold_spans"` + Recall float64 `json:"recall"` + IncompleteReason string `json:"incomplete_reason,omitempty"` } // TextCandidateResult records the evaluation-only mixed text and Go ranking. type TextCandidateResult struct { - Ranking Ranking `json:"ranking"` - CandidatePool CandidatePoolAudit `json:"candidate_pool"` - Latency Latency `json:"latency"` - Profile RetrievalProfile `json:"profile"` - IndexedTextFragments int `json:"indexed_text_fragments"` - SkippedTextFiles int `json:"skipped_text_files"` - MaximumTextFragments int `json:"maximum_text_fragments"` + Ranking Ranking `json:"ranking"` + CandidatePool CandidatePoolAudit `json:"candidate_pool"` + Latency Latency `json:"latency"` + Profile RetrievalProfile `json:"profile"` + IndexedTextFragments int `json:"indexed_text_fragments"` + SkippedTextFiles int `json:"skipped_text_files"` + MaximumTextFragments int `json:"maximum_text_fragments"` } // Latency stores measured samples and their summary statistics in milliseconds. @@ -226,45 +240,47 @@ type RetrievalProfiles struct { } // GoplsMeasurement records the optional workspace-symbol workflow and its limits. +// +//nolint:govet // Field order is part of the stable JSON report shape. type GoplsMeasurement struct { - Status string `json:"status"` - Version string `json:"version,omitempty"` - UnavailableReason string `json:"unavailable_reason,omitempty"` - CompletenessNote string `json:"completeness_note,omitempty"` - InitializeMS *float64 `json:"initialize_ms,omitempty"` - QueryLatency Latency `json:"query_latency"` - Granularity string `json:"candidate_granularity"` + Status string `json:"status"` + Version string `json:"version,omitempty"` + UnavailableReason string `json:"unavailable_reason,omitempty"` + CompletenessNote string `json:"completeness_note,omitempty"` + InitializeMS *float64 `json:"initialize_ms,omitempty"` + QueryLatency Latency `json:"query_latency"` + Granularity string `json:"candidate_granularity"` } // QueryResult combines gold spans, rankings, coverage, and timings for one query. type QueryResult struct { - ID string `json:"id"` - Text string `json:"query"` - Gold []GoldSpan `json:"gold"` - Retrieval Ranking `json:"retrieval"` - RetrievalTimings RetrievalTimings `json:"retrieval_timings"` - RetrievalProfiles RetrievalProfiles `json:"retrieval_profiles"` - RetrievalCandidatePool *CandidatePoolAudit `json:"retrieval_candidate_pool,omitempty"` - TextCandidateAblation *TextCandidateResult `json:"text_candidate_ablation,omitempty"` - NativeRG Ranking `json:"native_rg"` - NativeRGLatency Latency `json:"native_rg_latency"` - NativeRGCommandLatency Latency `json:"native_rg_candidate_gather_latency"` - NativeRGRankingLatency Latency `json:"native_rg_ranking_latency"` - GoplsQuery string `json:"gopls_query,omitempty"` - Gopls Ranking `json:"gopls,omitempty"` - GoplsQueryLatency Latency `json:"gopls_query_latency,omitempty"` + ID string `json:"id"` + Text string `json:"query"` + Gold []GoldSpan `json:"gold"` + Retrieval Ranking `json:"retrieval"` + RetrievalTimings RetrievalTimings `json:"retrieval_timings"` + RetrievalProfiles RetrievalProfiles `json:"retrieval_profiles"` + RetrievalCandidatePool *CandidatePoolAudit `json:"retrieval_candidate_pool,omitempty"` + TextCandidateAblation *TextCandidateResult `json:"text_candidate_ablation,omitempty"` + NativeRG Ranking `json:"native_rg"` + NativeRGLatency Latency `json:"native_rg_latency"` + NativeRGCommandLatency Latency `json:"native_rg_candidate_gather_latency"` + NativeRGRankingLatency Latency `json:"native_rg_ranking_latency"` + GoplsQuery string `json:"gopls_query,omitempty"` + Gopls Ranking `json:"gopls,omitempty"` + GoplsQueryLatency Latency `json:"gopls_query_latency,omitempty"` } // AggregateMetrics contains macro and micro scores across complete queries. type AggregateMetrics struct { - Queries int `json:"queries"` - MetricsQueries int `json:"metrics_queries"` - CompleteQueries int `json:"complete_queries"` - MacroRecall float64 `json:"macro_recall"` - MacroPrecision float64 `json:"macro_precision"` + Queries int `json:"queries"` + MetricsQueries int `json:"metrics_queries"` + CompleteQueries int `json:"complete_queries"` + MacroRecall float64 `json:"macro_recall"` + MacroPrecision float64 `json:"macro_precision"` MacroMeanReciprocalRank float64 `json:"macro_mean_reciprocal_rank"` - MicroRecall float64 `json:"micro_recall"` - MicroPrecision float64 `json:"micro_precision"` + MicroRecall float64 `json:"micro_recall"` + MicroPrecision float64 `json:"micro_precision"` } // ArmSummary contains aggregate scores for both result cutoffs. @@ -292,61 +308,67 @@ type RetrievalStageSummary struct { } // Summary contains aggregate retrieval, native-tool, gopls, and latency results. +// +//nolint:govet // Field order is part of the stable JSON report shape. type Summary struct { - Retrieval ArmSummary `json:"retrieval"` - NativeRG ArmSummary `json:"native_rg"` - Gopls ArmSummary `json:"gopls"` - GoplsStatus string `json:"gopls_status"` - RetrievalStages RetrievalStageSummary `json:"retrieval_stages"` - RetrievalColdP50MS map[string]float64 `json:"retrieval_cold_p50_ms_by_limit"` - RetrievalColdP95MS map[string]float64 `json:"retrieval_cold_p95_ms_by_limit"` - RetrievalWarmP50MS map[string]float64 `json:"retrieval_warm_p50_ms_by_limit"` - RetrievalWarmP95MS map[string]float64 `json:"retrieval_warm_p95_ms_by_limit"` - NativeRGP50MS float64 `json:"native_rg_p50_ms"` - NativeRGP95MS float64 `json:"native_rg_p95_ms"` - NativeRGCommandP50MS float64 `json:"native_rg_candidate_gather_p50_ms"` - NativeRGCommandP95MS float64 `json:"native_rg_candidate_gather_p95_ms"` - NativeRGRankingP50MS float64 `json:"native_rg_ranking_p50_ms"` - NativeRGRankingP95MS float64 `json:"native_rg_ranking_p95_ms"` - GoplsQueryP50MS *float64 `json:"gopls_query_p50_ms,omitempty"` - GoplsQueryP95MS *float64 `json:"gopls_query_p95_ms,omitempty"` + Retrieval ArmSummary `json:"retrieval"` + NativeRG ArmSummary `json:"native_rg"` + Gopls ArmSummary `json:"gopls"` + GoplsStatus string `json:"gopls_status"` + RetrievalStages RetrievalStageSummary `json:"retrieval_stages"` + RetrievalColdP50MS map[string]float64 `json:"retrieval_cold_p50_ms_by_limit"` + RetrievalColdP95MS map[string]float64 `json:"retrieval_cold_p95_ms_by_limit"` + RetrievalWarmP50MS map[string]float64 `json:"retrieval_warm_p50_ms_by_limit"` + RetrievalWarmP95MS map[string]float64 `json:"retrieval_warm_p95_ms_by_limit"` + NativeRGP50MS float64 `json:"native_rg_p50_ms"` + NativeRGP95MS float64 `json:"native_rg_p95_ms"` + NativeRGCommandP50MS float64 `json:"native_rg_candidate_gather_p50_ms"` + NativeRGCommandP95MS float64 `json:"native_rg_candidate_gather_p95_ms"` + NativeRGRankingP50MS float64 `json:"native_rg_ranking_p50_ms"` + NativeRGRankingP95MS float64 `json:"native_rg_ranking_p95_ms"` + GoplsQueryP50MS *float64 `json:"gopls_query_p50_ms,omitempty"` + GoplsQueryP95MS *float64 `json:"gopls_query_p95_ms,omitempty"` } // HeapStats stores sampled Go runtime heap values for the benchmark process. type HeapStats struct { - StartHeapAllocBytes uint64 `json:"start_heap_alloc_bytes"` - EndHeapAllocBytes uint64 `json:"end_heap_alloc_bytes"` + StartHeapAllocBytes uint64 `json:"start_heap_alloc_bytes"` + EndHeapAllocBytes uint64 `json:"end_heap_alloc_bytes"` PeakSampledHeapAllocBytes uint64 `json:"peak_sampled_heap_alloc_bytes"` - EndHeapSysBytes uint64 `json:"end_heap_sys_bytes"` - TotalAllocBytes uint64 `json:"total_alloc_bytes"` - NumGC uint32 `json:"num_gc"` + EndHeapSysBytes uint64 `json:"end_heap_sys_bytes"` + TotalAllocBytes uint64 `json:"total_alloc_bytes"` + NumGC uint32 `json:"num_gc"` } // Report is the complete provenance-bound result of one retrieval screen. +// +//nolint:govet // Field order is part of the stable JSON report shape. type Report struct { - SchemaVersion string `json:"schema_version"` - CreatedUTC time.Time `json:"created_utc"` - Repository Repository `json:"repository"` - Reproducibility Reproducibility `json:"reproducibility"` - Stratum string `json:"stratum"` - GoVersion string `json:"go_version"` - GOOS string `json:"goos"` - GOARCH string `json:"goarch"` - RGVersion string `json:"rg_version"` - SnapshotExportMS float64 `json:"snapshot_export_ms"` - RetrievalTimingNote string `json:"retrieval_timing_note"` - NativeRGWorkflow NativeWorkflow `json:"native_rg_workflow"` - GoPackages GoPackageInventory `json:"go_package_inventory"` - Coverage Coverage `json:"coverage"` - TextCandidateIndex *TextCandidateIndexCoverage `json:"text_candidate_index,omitempty"` - Configuration Configuration `json:"configuration"` - Gopls GoplsMeasurement `json:"gopls"` - Queries []QueryResult `json:"queries"` - Summary Summary `json:"summary"` - Heap HeapStats `json:"heap"` + SchemaVersion string `json:"schema_version"` + CreatedUTC time.Time `json:"created_utc"` + Repository Repository `json:"repository"` + Reproducibility Reproducibility `json:"reproducibility"` + Stratum string `json:"stratum"` + GoVersion string `json:"go_version"` + GOOS string `json:"goos"` + GOARCH string `json:"goarch"` + RGVersion string `json:"rg_version"` + SnapshotExportMS float64 `json:"snapshot_export_ms"` + RetrievalTimingNote string `json:"retrieval_timing_note"` + NativeRGWorkflow NativeWorkflow `json:"native_rg_workflow"` + GoPackages GoPackageInventory `json:"go_package_inventory"` + Coverage Coverage `json:"coverage"` + TextCandidateIndex *TextCandidateIndexCoverage `json:"text_candidate_index,omitempty"` + Configuration Configuration `json:"configuration"` + Gopls GoplsMeasurement `json:"gopls"` + Queries []QueryResult `json:"queries"` + Summary Summary `json:"summary"` + Heap HeapStats `json:"heap"` } // NativeWorkflow describes the fixed ripgrep and local ranking procedure. +// +//nolint:govet // Field order is part of the stable JSON report shape. type NativeWorkflow struct { CommandTemplate string `json:"command_template"` Tokenizer string `json:"tokenizer"`