Implement Tool Call Assertions in Asserter for Enhanced Validation

- Added support for `tool_called` and `tool_result` assertions in the Asserter, allowing for validation of tool execution and arguments.
- Introduced methods to check if specific tools were called and to validate their results against expected patterns.
- Updated the `README.md` to include detailed documentation on new assertion types, including usage examples and value formats.
- Enhanced the dynamic runner to set the response for tool-related assertions, improving the overall testing framework's capabilities.
This commit is contained in:
Max 2025-12-27 14:23:15 +08:00
parent 3969279f71
commit 5a63d9280b
5 changed files with 914 additions and 2 deletions

View file

@ -264,6 +264,12 @@ func (ast *Assistant) Stream(ctx *context.Context, inputMessages []context.Messa
// Execute all tool calls
toolResults, hasErrors := ast.executeToolCalls(ctx, currentResponse.ToolCalls, attempt)
// Build a map of tool call ID to arguments for quick lookup
toolCallArgsMap := make(map[string]interface{})
for _, tc := range currentResponse.ToolCalls {
toolCallArgsMap[tc.ID] = tc.Function.Arguments
}
// Convert toolResults to toolCallResponses
toolCallResponses = make([]context.ToolCallResponse, len(toolResults))
for i, result := range toolResults {
@ -272,7 +278,7 @@ func (ast *Assistant) Stream(ctx *context.Context, inputMessages []context.Messa
ToolCallID: result.ToolCallID,
Server: result.Server(),
Tool: result.Tool(),
Arguments: nil,
Arguments: toolCallArgsMap[result.ToolCallID],
Result: parsedContent,
Error: "",
}

View file

@ -575,6 +575,8 @@ Use `assert` for flexible validation. If `assert` is defined, it takes precedenc
| `json_path` | Extract JSON path and compare | `{"type": "json_path", "path": "$.field", "value": true}` |
| `regex` | Match regex pattern | `{"type": "regex", "value": "\\d+"}` |
| `type` | Check output type | `{"type": "type", "value": "object"}` |
| `tool_called` | Check if a tool was called | `{"type": "tool_called", "value": "setup"}` |
| `tool_result` | Check tool execution result | `{"type": "tool_result", "value": {"tool": "setup", "result": {"success": true}}}` |
### Assertion Fields
@ -634,6 +636,76 @@ The validator agent receives the output and criteria, then returns `{"passed": t
- `response`: The raw JSON response from the validator agent
- `criteria`: The validation criteria from the test case
### Tool Assertions
For validating that specific tools were called and their results:
#### tool_called
Check if a specific tool was called:
```jsonl
{
"id": "T001",
"input": "Set up my expense system",
"assert": {
"type": "tool_called",
"value": "setup"
}
}
```
**Value formats:**
- **String**: Tool name (supports suffix matching, e.g., `"setup"` matches `"agents_expense_tools__setup"`)
- **Array**: Any of the specified tools must be called
- **Object**: Match tool name and optionally arguments
```jsonl
// Match any of these tools
{"type": "tool_called", "value": ["setup", "init"]}
// Match tool with specific arguments
{"type": "tool_called", "value": {"name": "setup", "arguments": {"action": "init"}}}
```
#### tool_result
Check the result of a tool execution:
```jsonl
{
"id": "T001",
"input": "Set up my expense system",
"assert": {
"type": "tool_result",
"value": {
"tool": "setup",
"result": {
"success": true
}
}
}
}
```
**Result matching:**
- If `result` is omitted, only checks that the tool executed without error
- Supports partial matching (only specified fields are checked)
- Supports regex patterns with `regex:` prefix for string values
```jsonl
// Just check tool executed without error
{"type": "tool_result", "value": {"tool": "setup"}}
// Check specific result fields
{"type": "tool_result", "value": {"tool": "setup", "result": {"success": true}}}
// Use regex for message matching
{"type": "tool_result", "value": {"tool": "setup", "result": {"message": "regex:(?i)setup.*complete"}}}
```
### Script Assertions
For custom validation logic:

View file

@ -15,13 +15,22 @@ import (
)
// Asserter handles test assertions
type Asserter struct{}
type Asserter struct {
// response holds the current response for tool-related assertions
response *context.Response
}
// NewAsserter creates a new asserter
func NewAsserter() *Asserter {
return &Asserter{}
}
// WithResponse sets the response for tool-related assertions
func (a *Asserter) WithResponse(response *context.Response) *Asserter {
a.response = response
return a
}
// Validate validates the output against the test case's assertions
// Returns (passed, error message)
func (a *Asserter) Validate(tc *Case, output interface{}) (bool, string) {
@ -205,6 +214,10 @@ func (a *Asserter) evaluateAssertion(assertion *Assertion, output, input interfa
result = a.assertScript(assertion, output, input)
case "agent":
result = a.assertAgent(assertion, output, input)
case "tool_called":
result = a.assertToolCalled(assertion)
case "tool_result":
result = a.assertToolResult(assertion)
default:
result.Passed = false
result.Message = fmt.Sprintf("unknown assertion type: %s", assertion.Type)
@ -747,6 +760,268 @@ func (a *Asserter) assertScript(assertion *Assertion, output, input interface{})
return result
}
// assertToolCalled checks if a specific tool was called
// value can be:
// - string: exact tool name to match
// - []string: any of the tool names
// - map with "name" and optional "arguments" for more specific matching
func (a *Asserter) assertToolCalled(assertion *Assertion) *AssertionResult {
result := &AssertionResult{
Assertion: assertion,
Expected: assertion.Value,
}
if a.response == nil {
result.Passed = false
result.Message = "no response available for tool_called assertion"
return result
}
if len(a.response.Tools) == 0 {
result.Passed = false
result.Message = "no tools were called"
return result
}
// Get tool names that were called
calledTools := make([]string, 0, len(a.response.Tools))
for _, tool := range a.response.Tools {
calledTools = append(calledTools, tool.Tool)
}
result.Actual = calledTools
switch v := assertion.Value.(type) {
case string:
// Simple case: check if tool name matches (supports prefix matching)
for _, tool := range a.response.Tools {
if matchToolName(tool.Tool, v) {
result.Passed = true
result.Message = fmt.Sprintf("tool '%s' was called", v)
return result
}
}
result.Passed = false
result.Message = fmt.Sprintf("tool '%s' was not called, called: %v", v, calledTools)
case []interface{}:
// Check if any of the specified tools were called
for _, expected := range v {
if expectedStr, ok := expected.(string); ok {
for _, tool := range a.response.Tools {
if matchToolName(tool.Tool, expectedStr) {
result.Passed = true
result.Message = fmt.Sprintf("tool '%s' was called", expectedStr)
return result
}
}
}
}
result.Passed = false
result.Message = fmt.Sprintf("none of the expected tools were called, called: %v", calledTools)
case map[string]interface{}:
// Advanced case: match name and optionally arguments
expectedName, _ := v["name"].(string)
expectedArgs := v["arguments"]
for _, tool := range a.response.Tools {
if matchToolName(tool.Tool, expectedName) {
// If arguments specified, check them too
if expectedArgs != nil {
if matchArguments(tool.Arguments, expectedArgs) {
result.Passed = true
result.Message = fmt.Sprintf("tool '%s' was called with matching arguments", expectedName)
return result
}
} else {
result.Passed = true
result.Message = fmt.Sprintf("tool '%s' was called", expectedName)
return result
}
}
}
result.Passed = false
if expectedArgs != nil {
result.Message = fmt.Sprintf("tool '%s' was not called with expected arguments", expectedName)
} else {
result.Message = fmt.Sprintf("tool '%s' was not called, called: %v", expectedName, calledTools)
}
default:
result.Passed = false
result.Message = fmt.Sprintf("invalid tool_called value type: %T", assertion.Value)
}
return result
}
// assertToolResult checks the result of a tool call
// value should be a map with "tool" (name) and "result" (expected result pattern)
func (a *Asserter) assertToolResult(assertion *Assertion) *AssertionResult {
result := &AssertionResult{
Assertion: assertion,
Expected: assertion.Value,
}
if a.response == nil {
result.Passed = false
result.Message = "no response available for tool_result assertion"
return result
}
if len(a.response.Tools) == 0 {
result.Passed = false
result.Message = "no tools were called"
return result
}
spec, ok := assertion.Value.(map[string]interface{})
if !ok {
result.Passed = false
result.Message = "tool_result assertion requires a map with 'tool' and 'result' fields"
return result
}
toolName, _ := spec["tool"].(string)
expectedResult := spec["result"]
if toolName == "" {
result.Passed = false
result.Message = "tool_result assertion requires 'tool' field"
return result
}
// Find the tool call
for _, tool := range a.response.Tools {
if matchToolName(tool.Tool, toolName) {
result.Actual = tool.Result
// Check if there was an error
if tool.Error != "" {
result.Passed = false
result.Message = fmt.Sprintf("tool '%s' returned error: %s", toolName, tool.Error)
return result
}
// If no expected result specified, just check success (no error)
if expectedResult == nil {
result.Passed = true
result.Message = fmt.Sprintf("tool '%s' executed successfully", toolName)
return result
}
// Match result
if matchResult(tool.Result, expectedResult) {
result.Passed = true
result.Message = fmt.Sprintf("tool '%s' result matches expected", toolName)
return result
}
result.Passed = false
result.Message = fmt.Sprintf("tool '%s' result does not match expected", toolName)
return result
}
}
result.Passed = false
result.Message = fmt.Sprintf("tool '%s' was not called", toolName)
return result
}
// matchToolName checks if a tool name matches the expected pattern
// Supports exact match and suffix match (e.g., "setup" matches "agents_expense_tools__setup")
func matchToolName(actual, expected string) bool {
if actual == expected {
return true
}
// Support suffix matching (tool name without namespace prefix)
if strings.HasSuffix(actual, "__"+expected) || strings.HasSuffix(actual, "."+expected) {
return true
}
// Support contains matching for partial names
if strings.Contains(actual, expected) {
return true
}
return false
}
// matchArguments checks if tool arguments match expected pattern
func matchArguments(actual, expected interface{}) bool {
expectedMap, ok := expected.(map[string]interface{})
if !ok {
return false
}
actualMap, ok := actual.(map[string]interface{})
if !ok {
// Try parsing as JSON string
if actualStr, ok := actual.(string); ok {
var parsed map[string]interface{}
if err := jsoniter.UnmarshalFromString(actualStr, &parsed); err == nil {
actualMap = parsed
} else {
return false
}
} else {
return false
}
}
// Check that all expected keys exist and match
for key, expectedVal := range expectedMap {
actualVal, exists := actualMap[key]
if !exists {
return false
}
if !validateOutput(actualVal, expectedVal) {
return false
}
}
return true
}
// matchResult checks if tool result matches expected pattern
func matchResult(actual, expected interface{}) bool {
switch exp := expected.(type) {
case map[string]interface{}:
actualMap, ok := actual.(map[string]interface{})
if !ok {
return false
}
// Check that all expected keys exist and match
for key, expectedVal := range exp {
actualVal, exists := actualMap[key]
if !exists {
return false
}
if !matchResult(actualVal, expectedVal) {
return false
}
}
return true
case string:
// Support regex pattern matching for strings
if strings.HasPrefix(exp, "regex:") {
pattern := strings.TrimPrefix(exp, "regex:")
re, err := regexp.Compile(pattern)
if err != nil {
return false
}
actualStr := fmt.Sprintf("%v", actual)
return re.MatchString(actualStr)
}
return fmt.Sprintf("%v", actual) == exp
case bool:
actualBool, ok := actual.(bool)
return ok && actualBool == exp
default:
return validateOutput(actual, expected)
}
}
// toString converts a value to string for comparison
func (a *Asserter) toString(v interface{}) string {
if v == nil {

View file

@ -2,6 +2,8 @@ package test
import (
"testing"
"github.com/yaoapp/yao/agent/context"
)
func TestAsserter_JSONPath_ArrayEquality(t *testing.T) {
@ -303,3 +305,552 @@ func TestAsserter_Regex(t *testing.T) {
})
}
}
func TestAsserter_ToolCalled(t *testing.T) {
tests := []struct {
name string
tc *Case
response *context.Response
expected bool
errMsg string
}{
{
name: "tool called - exact match",
tc: &Case{
Assert: map[string]interface{}{
"type": "tool_called",
"value": "agents_expense_tools__setup",
},
},
response: &context.Response{
Tools: []context.ToolCallResponse{
{Tool: "agents_expense_tools__setup", Result: map[string]interface{}{"success": true}},
},
},
expected: true,
},
{
name: "tool called - suffix match",
tc: &Case{
Assert: map[string]interface{}{
"type": "tool_called",
"value": "setup",
},
},
response: &context.Response{
Tools: []context.ToolCallResponse{
{Tool: "agents_expense_tools__setup", Result: map[string]interface{}{"success": true}},
},
},
expected: true,
},
{
name: "tool not called",
tc: &Case{
Assert: map[string]interface{}{
"type": "tool_called",
"value": "setup",
},
},
response: &context.Response{
Tools: []context.ToolCallResponse{},
},
expected: false,
},
{
name: "tool called - wrong tool",
tc: &Case{
Assert: map[string]interface{}{
"type": "tool_called",
"value": "setup",
},
},
response: &context.Response{
Tools: []context.ToolCallResponse{
{Tool: "agents_expense_tools__submit", Result: map[string]interface{}{"success": true}},
},
},
expected: false,
},
{
name: "tool called - any of multiple",
tc: &Case{
Assert: map[string]interface{}{
"type": "tool_called",
"value": []interface{}{"setup", "init"},
},
},
response: &context.Response{
Tools: []context.ToolCallResponse{
{Tool: "agents_expense_tools__init", Result: map[string]interface{}{"success": true}},
},
},
expected: true,
},
{
name: "tool called - with arguments (map)",
tc: &Case{
Assert: map[string]interface{}{
"type": "tool_called",
"value": map[string]interface{}{
"name": "setup",
"arguments": map[string]interface{}{
"action": "init",
},
},
},
},
response: &context.Response{
Tools: []context.ToolCallResponse{
{
Tool: "agents_expense_tools__setup",
Arguments: map[string]interface{}{"action": "init", "config": map[string]interface{}{}},
Result: map[string]interface{}{"success": true},
},
},
},
expected: true,
},
{
name: "tool called - with arguments (JSON string)",
tc: &Case{
Assert: map[string]interface{}{
"type": "tool_called",
"value": map[string]interface{}{
"name": "setup",
"arguments": map[string]interface{}{
"action": "init",
},
},
},
},
response: &context.Response{
Tools: []context.ToolCallResponse{
{
Tool: "agents_expense_tools__setup",
Arguments: `{"action":"init","config":{"default_currency":"USD"}}`,
Result: map[string]interface{}{"success": true},
},
},
},
expected: true,
},
{
name: "tool called - wrong arguments",
tc: &Case{
Assert: map[string]interface{}{
"type": "tool_called",
"value": map[string]interface{}{
"name": "setup",
"arguments": map[string]interface{}{
"action": "update",
},
},
},
},
response: &context.Response{
Tools: []context.ToolCallResponse{
{
Tool: "agents_expense_tools__setup",
Arguments: map[string]interface{}{"action": "init"},
Result: map[string]interface{}{"success": true},
},
},
},
expected: false,
},
{
name: "no response",
tc: &Case{
Assert: map[string]interface{}{
"type": "tool_called",
"value": "setup",
},
},
response: nil,
expected: false,
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
asserter := NewAsserter().WithResponse(tt.response)
passed, errMsg := asserter.Validate(tt.tc, nil)
if passed != tt.expected {
t.Errorf("Expected passed=%v, got passed=%v, error: %s", tt.expected, passed, errMsg)
}
})
}
}
func TestAsserter_ToolResult(t *testing.T) {
tests := []struct {
name string
tc *Case
response *context.Response
expected bool
}{
{
name: "tool result - success check",
tc: &Case{
Assert: map[string]interface{}{
"type": "tool_result",
"value": map[string]interface{}{
"tool": "setup",
"result": map[string]interface{}{
"success": true,
},
},
},
},
response: &context.Response{
Tools: []context.ToolCallResponse{
{
Tool: "agents_expense_tools__setup",
Result: map[string]interface{}{"success": true, "message": "Setup complete"},
},
},
},
expected: true,
},
{
name: "tool result - message check with regex",
tc: &Case{
Assert: map[string]interface{}{
"type": "tool_result",
"value": map[string]interface{}{
"tool": "setup",
"result": map[string]interface{}{
"message": "regex:(?i)setup.*complete",
},
},
},
},
response: &context.Response{
Tools: []context.ToolCallResponse{
{
Tool: "agents_expense_tools__setup",
Result: map[string]interface{}{"success": true, "message": "Setup complete!"},
},
},
},
expected: true,
},
{
name: "tool result - no expected result (just check no error)",
tc: &Case{
Assert: map[string]interface{}{
"type": "tool_result",
"value": map[string]interface{}{
"tool": "setup",
},
},
},
response: &context.Response{
Tools: []context.ToolCallResponse{
{
Tool: "agents_expense_tools__setup",
Result: map[string]interface{}{"success": true},
},
},
},
expected: true,
},
{
name: "tool result - tool has error",
tc: &Case{
Assert: map[string]interface{}{
"type": "tool_result",
"value": map[string]interface{}{
"tool": "setup",
},
},
},
response: &context.Response{
Tools: []context.ToolCallResponse{
{
Tool: "agents_expense_tools__setup",
Error: "permission denied",
},
},
},
expected: false,
},
{
name: "tool result - result mismatch",
tc: &Case{
Assert: map[string]interface{}{
"type": "tool_result",
"value": map[string]interface{}{
"tool": "setup",
"result": map[string]interface{}{
"success": true,
},
},
},
},
response: &context.Response{
Tools: []context.ToolCallResponse{
{
Tool: "agents_expense_tools__setup",
Result: map[string]interface{}{"success": false},
},
},
},
expected: false,
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
asserter := NewAsserter().WithResponse(tt.response)
passed, errMsg := asserter.Validate(tt.tc, nil)
if passed != tt.expected {
t.Errorf("Expected passed=%v, got passed=%v, error: %s", tt.expected, passed, errMsg)
}
})
}
}
func TestAsserter_MultipleToolAssertions(t *testing.T) {
// This tests the exact scenario from setup-006: tool_called + tool_result
response := &context.Response{
Tools: []context.ToolCallResponse{
{
Tool: "agents_expense_tools__setup",
Arguments: `{"action":"init","config":{"default_currency":"USD","categories":[{"id":"meals","name":"Meals","daily_limit":100}]}}`,
Result: map[string]interface{}{
"success": true,
"action": "init",
"config": map[string]interface{}{
"default_currency": "USD",
},
"message": "Setup complete!",
},
},
},
}
asserter := NewAsserter().WithResponse(response)
tc := &Case{
Assert: []interface{}{
map[string]interface{}{
"type": "tool_called",
"value": map[string]interface{}{
"name": "setup",
"arguments": map[string]interface{}{
"action": "init",
},
},
},
map[string]interface{}{
"type": "tool_result",
"value": map[string]interface{}{
"tool": "setup",
"result": map[string]interface{}{
"success": true,
},
},
},
},
}
result := asserter.ValidateWithDetails(tc, nil)
if !result.Passed {
t.Errorf("Expected multiple tool assertions to pass, got: %s", result.Message)
}
}
func TestAsserter_SharedAsserterWithResponse(t *testing.T) {
// Test that a shared asserter correctly uses WithResponse
asserter := NewAsserter()
// First call without response - should fail
tc := &Case{
Assert: map[string]interface{}{
"type": "tool_called",
"value": "setup",
},
}
result := asserter.ValidateWithDetails(tc, nil)
if result.Passed {
t.Error("Expected tool_called to fail without response")
}
if result.Message != "no response available for tool_called assertion" {
t.Errorf("Unexpected message: %s", result.Message)
}
// Now set response
response := &context.Response{
Tools: []context.ToolCallResponse{
{
Tool: "agents_expense_tools__setup",
Result: map[string]interface{}{"success": true},
},
},
}
asserter.WithResponse(response)
// Should pass now
result = asserter.ValidateWithDetails(tc, nil)
if !result.Passed {
t.Errorf("Expected tool_called to pass with response, got: %s", result.Message)
}
}
func TestAsserter_Setup006Scenario(t *testing.T) {
// Exact reproduction of setup-006 scenario
// Turn 2: tool was called with action: init, result has success: true
response := &context.Response{
Tools: []context.ToolCallResponse{
{
Tool: "agents_expense_tools__setup",
Arguments: `{"action":"init","config":{"default_currency":"USD","categories":[{"id":"meals","name":"Meals","daily_limit":100},{"id":"travel","name":"Travel","daily_limit":500}]}}`,
Result: map[string]interface{}{
"config": map[string]interface{}{
"categories": []interface{}{
map[string]interface{}{"daily_limit": float64(100), "id": "meals", "name": "Meals"},
map[string]interface{}{"daily_limit": float64(500), "id": "travel", "name": "Travel"},
},
"default_currency": "USD",
},
"message": "Setup complete! The expense system has been initialized successfully with the configured settings. You can now start submitting expenses.",
"success": true,
"action": "init",
},
},
},
}
// This is the exact assert from setup-006's quick_complete checkpoint
assertDef := []interface{}{
map[string]interface{}{
"type": "tool_called",
"value": map[string]interface{}{
"name": "setup",
"arguments": map[string]interface{}{
"action": "init",
},
},
},
map[string]interface{}{
"type": "tool_result",
"value": map[string]interface{}{
"tool": "setup",
"result": map[string]interface{}{
"success": true,
},
},
},
}
asserter := NewAsserter().WithResponse(response)
tc := &Case{Assert: assertDef}
result := asserter.ValidateWithDetails(tc, nil)
if !result.Passed {
t.Errorf("Expected setup-006 scenario to pass, got: %s", result.Message)
}
// Also test individual assertions
t.Run("tool_called only", func(t *testing.T) {
tc2 := &Case{Assert: assertDef[0]}
result2 := asserter.ValidateWithDetails(tc2, nil)
if !result2.Passed {
t.Errorf("Expected tool_called to pass, got: %s", result2.Message)
}
})
t.Run("tool_result only", func(t *testing.T) {
tc3 := &Case{Assert: assertDef[1]}
result3 := asserter.ValidateWithDetails(tc3, nil)
if !result3.Passed {
t.Errorf("Expected tool_result to pass, got: %s", result3.Message)
}
})
}
func TestAsserter_Setup003Scenario(t *testing.T) {
// Exact reproduction of setup-003 scenario
// Turn 3: tool was called with action: update
response := &context.Response{
Tools: []context.ToolCallResponse{
{
Tool: "agents_expense_tools__setup",
Arguments: `{"action":"update","config":{"categories":[{"daily_limit":500,"id":"meals","name":"Business Meals"}]}}`,
Result: map[string]interface{}{
"action": "update",
"config": map[string]interface{}{
"categories": []interface{}{
map[string]interface{}{"daily_limit": float64(500), "id": "meals", "name": "Business Meals"},
},
},
"message": "Configuration updated successfully! Your changes have been saved.",
"success": true,
},
},
},
}
// This is the exact assert from setup-003's update_complete checkpoint
assertDef := []interface{}{
map[string]interface{}{
"type": "tool_called",
"value": map[string]interface{}{
"name": "setup",
"arguments": map[string]interface{}{
"action": "update",
},
},
},
map[string]interface{}{
"type": "tool_result",
"value": map[string]interface{}{
"tool": "setup",
"result": map[string]interface{}{
"success": true,
},
},
},
}
asserter := NewAsserter().WithResponse(response)
tc := &Case{Assert: assertDef}
result := asserter.ValidateWithDetails(tc, nil)
if !result.Passed {
t.Errorf("Expected setup-003 scenario to pass, got: %s", result.Message)
}
// Test individual assertions
t.Run("tool_called with action:update", func(t *testing.T) {
tc2 := &Case{Assert: assertDef[0]}
result2 := asserter.ValidateWithDetails(tc2, nil)
if !result2.Passed {
t.Errorf("Expected tool_called to pass, got: %s", result2.Message)
}
})
}
func TestMatchToolName(t *testing.T) {
tests := []struct {
actual string
expected string
match bool
}{
{"agents_expense_tools__setup", "agents_expense_tools__setup", true},
{"agents_expense_tools__setup", "setup", true},
{"agents.expense.tools.setup", "setup", true},
{"agents_expense_tools__setup", "init", false},
{"setup", "setup", true},
}
for _, tt := range tests {
t.Run(tt.actual+"_"+tt.expected, func(t *testing.T) {
if matchToolName(tt.actual, tt.expected) != tt.match {
t.Errorf("matchToolName(%q, %q) = %v, want %v", tt.actual, tt.expected, !tt.match, tt.match)
}
})
}
}

View file

@ -377,6 +377,9 @@ func (r *DynamicRunner) checkCheckpoints(checkpoints []*Checkpoint, response *co
// This includes both content and tool result messages
combinedOutput := buildCombinedOutput(response)
// Set response on asserter for tool-related assertions
r.asserter.WithResponse(response)
for _, cp := range checkpoints {
cpResult := result.Checkpoints[cp.ID]
if cpResult.Reached {
@ -409,6 +412,11 @@ func (r *DynamicRunner) checkCheckpoints(checkpoints []*Checkpoint, response *co
cpResult.ReachedAtTurn = len(result.Turns) + 1
cpResult.Message = assertResult.Message
reachedIDs = append(reachedIDs, cp.ID)
} else {
// Store failure message for debugging (but don't mark as failed yet - it might pass in a later turn)
if cpResult.Message == "" {
cpResult.Message = assertResult.Message
}
}
// Store agent validation details if this is an agent assertion