Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
187 changes: 187 additions & 0 deletions agent/harness/loop/evaluators.go
Original file line number Diff line number Diff line change
Expand Up @@ -4,10 +4,197 @@ package loop

import (
"context"
"encoding/json"
"errors"
"strings"

"github.com/microsoft/agent-framework-go/agent"
"github.com/microsoft/agent-framework-go/message"
)

const (
// AIJudgeDoneMarker is the text-fallback marker the judge is asked to emit
// (for judges that do not honor structured output) when the original request
// has been fully addressed.
AIJudgeDoneMarker = "VERDICT: DONE"
// AIJudgeMoreMarker is the text-fallback marker for "more work required". It
// deliberately does not overlap AIJudgeDoneMarker and wins when the verdict
// is ambiguous or absent, so the loop keeps running rather than stopping on
// an incomplete answer.
AIJudgeMoreMarker = "VERDICT: MORE"

// AIJudgeCriteriaPlaceholder in the judge instructions is replaced with the
// rendered criteria (or removed when none are supplied).
AIJudgeCriteriaPlaceholder = "{criteria}"
// AIJudgeGapAnalysisPlaceholder in the feedback template is replaced with the
// judge's gap analysis.
AIJudgeGapAnalysisPlaceholder = "{gap_analysis}"

aiJudgeUnknownGapAnalysis = "<unknown>"

// DefaultAIJudgeInstructions are the default system instructions for the judge.
DefaultAIJudgeInstructions = "You are an evaluator. You are given a user's original request and an agent's latest response. " +
"Decide whether the agent has fully addressed the original request. " +
"Set 'answered' to true if the request has been fully addressed, or false if more work is still required. " +
"When 'answered' is false, use 'gapAnalysis' to explain what is still missing or what work remains. " +
"If you cannot return structured output, reply with " + AIJudgeDoneMarker + " when the request has been fully " +
"addressed, or " + AIJudgeMoreMarker + " when more work is still required." + AIJudgeCriteriaPlaceholder

// DefaultAIJudgeFeedbackTemplate is the default feedback template used when
// the request is not yet answered.
DefaultAIJudgeFeedbackTemplate = "Your previous response did not fully address the original request. " +
"The following is still missing or incomplete: " + AIJudgeGapAnalysisPlaceholder + " " +
"Please continue and fully address the original request."
)

// JudgeVerdict is the structured verdict an AI judge returns.
type JudgeVerdict struct {
// Answered is true when the judge decided the original request was fully addressed.
Answered bool `json:"answered"`
// GapAnalysis explains what is still missing when Answered is false.
GapAnalysis string `json:"gapAnalysis"`
}

// AIJudgeConfig configures [NewAIJudgeEvaluator].
type AIJudgeConfig struct {
// Instructions overrides the judge system instructions. Any occurrence of
// [AIJudgeCriteriaPlaceholder] is replaced with the rendered Criteria. When
// nil, [DefaultAIJudgeInstructions] is used.
Instructions *string

// Criteria are bespoke standards the response must satisfy, appended to the
// instructions at [AIJudgeCriteriaPlaceholder].
Criteria []string

// FeedbackMessageTemplate overrides the feedback produced when the request is
// not yet answered. [AIJudgeGapAnalysisPlaceholder] is replaced with the
// judge's gap analysis. When nil, [DefaultAIJudgeFeedbackTemplate] is used.
FeedbackMessageTemplate *string
}

// AIJudgeEvaluator uses a separate judge agent to decide whether the user's
// original request has been fully addressed, continuing the loop (with the
// judge's gap analysis as feedback) while the answer is "no".
//
// Security: the judge agent is sent the original request and the agent's latest
// response on every iteration, both of which may contain sensitive or untrusted
// content. A compromised judge endpoint could exfiltrate that data or return a
// manipulated verdict that steers the agent via indirect prompt injection. Only
// use a judge you trust as much as the primary model, and prefer a stricter
// [Config.MaxIterations] since LLM-judged loops are costly and probabilistic.
type AIJudgeEvaluator struct {
judge *agent.Agent
instructions string
feedbackMessageTemplate string
}

// NewAIJudgeEvaluator creates an evaluator that queries the judge agent after
// each iteration. It panics if judge is nil.
func NewAIJudgeEvaluator(judge *agent.Agent, config AIJudgeConfig) *AIJudgeEvaluator {
if judge == nil {
panic("loop: judge agent cannot be nil")
}
instructions := DefaultAIJudgeInstructions
if config.Instructions != nil {
instructions = *config.Instructions
}
instructions = strings.ReplaceAll(instructions, AIJudgeCriteriaPlaceholder, renderJudgeCriteria(config.Criteria))
feedback := DefaultAIJudgeFeedbackTemplate
if config.FeedbackMessageTemplate != nil {
feedback = *config.FeedbackMessageTemplate
}
return &AIJudgeEvaluator{
judge: judge,
instructions: instructions,
feedbackMessageTemplate: feedback,
}
}

// Evaluate implements Evaluator.
func (e *AIJudgeEvaluator) Evaluate(ctx context.Context, loop *Context) (Evaluation, error) {
if loop == nil {
return Stop(), errors.New("loop: context cannot be nil")
}
if loop.LastResponse == nil {
return Stop(), errors.New("loop: last response cannot be nil")
}

// Build the judge's user message from the original request contents (so
// non-text content is preserved rather than flattened) framed by header text,
// followed by the agent's latest response.
contents := message.Contents{&message.TextContent{Text: "# Has the original request been fully addressed?\n\n## Original request:\n"}}
for _, m := range loop.InitialMessages {
contents = append(contents, m.Contents...)
}
contents = append(contents, &message.TextContent{Text: "\n\n## Agent's latest response:\n" + loop.LastResponse.String()})

resp, err := e.judge.Run(ctx, []*message.Message{{Role: message.RoleUser, Contents: contents}}, agent.WithInstructions(e.instructions)).Collect()

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Parity gap vs. the upstream .NET AIJudgeLoopEvaluator:

  1. No judge isolation. NewAIJudgeEvaluator takes a full *agent.Agent and Evaluate runs it through e.judge.Run(...), which goes through the entire agent pipeline (default in-memory history provider, any ContextProviders, Tools, Middlewares configured on that agent). Upstream's AIJudgeLoopEvaluator(IChatClient judgeClient, ...) intentionally takes a raw IChatClient so the judge is, per its doc remarks, "queried directly (without any agent tools, session, or middleware)". See dotnet/src/Microsoft.Agents.AI/Harness/Loop/AIJudgeLoopEvaluator.cs. If a caller passes a Go agent that has tools or session/history configured, the judge could execute tools or leak state across loop iterations — a structural guarantee the .NET type enforces at the type level that Go does not. Consider constraining the accepted judge type (e.g. a raw provider/run function) or explicitly documenting/enforcing the no-tools/no-session expectation.

  2. No native structured-output request. Upstream calls judgeClient.GetResponseAsync<JudgeVerdict>(judgeMessages, LoopJsonContext.Default.Options, ...), asking the provider for structured JSON output before falling back to text-marker parsing only when the client doesn't honor it (see TryGetResult branch in the same file). This Go Evaluate only passes agent.WithInstructions(e.instructions) and never uses agent.WithStructuredOutput/agent.WithResponseFormat (already available in agent/options.go / agent/structuredoutput.go), so it always falls back to manual {/} substring scraping (extractJudgeVerdict) rather than requesting the provider's native structured-output support like .NET does.

if err != nil {
return Stop(), err
}

answered, gapAnalysis := parseJudgeVerdict(resp.String())
if answered {
return Stop(), nil
}
return Continue(strings.ReplaceAll(e.feedbackMessageTemplate, AIJudgeGapAnalysisPlaceholder, gapAnalysis)), nil
}

// parseJudgeVerdict prefers a structured JudgeVerdict, falling back to the
// text markers. AIJudgeMoreMarker wins when ambiguous or absent so the loop
// keeps running rather than stopping on an incomplete answer.
func parseJudgeVerdict(text string) (answered bool, gapAnalysis string) {
gapAnalysis = aiJudgeUnknownGapAnalysis
if verdict, ok := extractJudgeVerdict(text); ok {
if strings.TrimSpace(verdict.GapAnalysis) != "" {
gapAnalysis = verdict.GapAnalysis
}
return verdict.Answered, gapAnalysis
}
upper := strings.ToUpper(text)
answered = !strings.Contains(upper, AIJudgeMoreMarker) && strings.Contains(upper, AIJudgeDoneMarker)
return answered, gapAnalysis
}

// extractJudgeVerdict pulls a JudgeVerdict out of the response text, tolerating
// surrounding prose or code fences. It only succeeds when the JSON object
// actually carries an "answered" key, so plain text does not decode to a
// spurious zero-value verdict.
func extractJudgeVerdict(text string) (JudgeVerdict, bool) {
start := strings.Index(text, "{")
end := strings.LastIndex(text, "}")
if start < 0 || end <= start {
return JudgeVerdict{}, false
}
raw := []byte(text[start : end+1])
var probe map[string]json.RawMessage
if err := json.Unmarshal(raw, &probe); err != nil {
return JudgeVerdict{}, false
}
if _, ok := probe["answered"]; !ok {
return JudgeVerdict{}, false
}
var verdict JudgeVerdict
if err := json.Unmarshal(raw, &verdict); err != nil {
return JudgeVerdict{}, false
}
return verdict, true
}

func renderJudgeCriteria(criteria []string) string {
var b strings.Builder
for _, c := range criteria {
if strings.TrimSpace(c) != "" {
b.WriteString("\n- ")
b.WriteString(c)
}
}
if b.Len() == 0 {
return ""
}
return "\n\nThe response must satisfy all of the following criteria:" + b.String()
}

// CompletionMarkerConfig configures a completion-marker evaluator.
type CompletionMarkerConfig struct {
// Marker is the completion marker that stops the loop when present in the
Expand Down
97 changes: 97 additions & 0 deletions agent/harness/loop/loop_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -559,3 +559,100 @@ func cloneMessages(messages []*message.Message) []*message.Message {
}
return out
}

// judgeReturning builds an agent whose every run returns the given fixed text,
// for use as an AIJudgeEvaluator judge.
func judgeReturning(text string) *agent.Agent {
capture := newCaptureAgent(func(int, []*message.Message) []*agent.ResponseUpdate {
return textUpdates(text)
})
return agent.New(capture.provider(), agent.Config{})
}

func TestAIJudgeEvaluator_StructuredVerdict(t *testing.T) {
// Answered -> stop.
stop, err := loop.NewAIJudgeEvaluator(judgeReturning(`{"answered":true}`), loop.AIJudgeConfig{}).
Evaluate(context.Background(), contextWithResponse("done"))
if err != nil {
t.Fatal(err)
}
if stop.ShouldReinvoke {
t.Fatal("answered verdict should stop the loop")
}

// Not answered with gap analysis -> continue with the gap in the feedback.
cont, err := loop.NewAIJudgeEvaluator(judgeReturning(`Here is my verdict: {"answered":false,"gapAnalysis":"missing the summary"}`), loop.AIJudgeConfig{}).
Evaluate(context.Background(), contextWithResponse("partial"))
if err != nil {
t.Fatal(err)
}
if !cont.ShouldReinvoke {
t.Fatal("unanswered verdict should continue the loop")
}
if !strings.Contains(cont.Feedback, "missing the summary") || strings.Contains(cont.Feedback, "{gap_analysis}") {
t.Fatalf("feedback = %q, want the gap analysis substituted", cont.Feedback)
}
}

func TestAIJudgeEvaluator_TextMarkerFallback(t *testing.T) {
// DONE marker (no structured JSON) -> stop.
stop, err := loop.NewAIJudgeEvaluator(judgeReturning("The task looks complete. "+loop.AIJudgeDoneMarker), loop.AIJudgeConfig{}).
Evaluate(context.Background(), contextWithResponse("x"))
if err != nil {
t.Fatal(err)
}
if stop.ShouldReinvoke {
t.Fatal("DONE marker should stop the loop")
}

// MORE marker -> continue; and ambiguous (both markers) -> MORE wins.
for _, text := range []string{
"Not there yet. " + loop.AIJudgeMoreMarker,
loop.AIJudgeDoneMarker + " but also " + loop.AIJudgeMoreMarker,
"no verdict at all",
} {
cont, err := loop.NewAIJudgeEvaluator(judgeReturning(text), loop.AIJudgeConfig{}).
Evaluate(context.Background(), contextWithResponse("x"))
if err != nil {
t.Fatal(err)
}
if !cont.ShouldReinvoke {
t.Fatalf("text %q should continue the loop (MORE wins when ambiguous/absent)", text)
}
}
}

func TestAIJudgeEvaluator_CriteriaRenderedIntoInstructions(t *testing.T) {
// Capture the system instructions the judge receives on its run.
var gotInstructions string
judge := agent.New(agent.ProviderConfig{
Run: func(_ context.Context, _ []*message.Message, opts ...agent.Option) iter.Seq2[*agent.ResponseUpdate, error] {
gotInstructions, _ = agent.GetOption(opts, agent.WithInstructions)
return func(yield func(*agent.ResponseUpdate, error) bool) {
yield(textUpdates(`{"answered":true}`)[0], nil)
}
},
}, agent.Config{})

_, err := loop.NewAIJudgeEvaluator(judge, loop.AIJudgeConfig{Criteria: []string{"cite sources", " ", "use markdown"}}).
Evaluate(context.Background(), contextWithResponse("x"))
if err != nil {
t.Fatal(err)
}
// Blank criteria are skipped; the non-blank ones are rendered as a bullet list.
if !strings.Contains(gotInstructions, "The response must satisfy all of the following criteria:") ||
!strings.Contains(gotInstructions, "\n- cite sources") ||
!strings.Contains(gotInstructions, "\n- use markdown") ||
strings.Contains(gotInstructions, "{criteria}") {
t.Fatalf("judge instructions = %q, want rendered criteria and no placeholder", gotInstructions)
}
}

func TestNewAIJudgeEvaluator_PanicsWithNilJudge(t *testing.T) {
defer func() {
if recover() == nil {
t.Fatal("expected panic")
}
}()
loop.NewAIJudgeEvaluator(nil, loop.AIJudgeConfig{})
}
Loading