mirror of
https://gitcode.com/JianFeeeee/HomeAgent.git
synced 2026-09-27 21:03:16 +00:00
feat(toolcall): 结果契约诚实化 —— Success 不再恒真 + 结构化失败可回填(阶段 1b/1d)
问题(实测,三条互相印证):
· ToolResult.Success 硬编码 true(task.go 唯一赋值点)⇒ 该字段在结构上
不可能为 false,是**谎报字段**;
· 工具失败以 nil error + 错误**值**返回(files 的 errorResult、
pluginmgr 的 {"error":…}),上游无从判别;
· stepToolAfter 用 `Result.(string)` 断言,而插件返回的多是 map ⇒
断言几乎恒失败,after_toolcall 阶段对结构化结果的改写**静默失效**。
改动:
· third_party/homeagent-sdk: 新增 ToolError{Field,Reason,Detail,Hint}
与 Error()。纯新增、无签名变更,存量插件不必改动(零值语义:
内核的失败识别同时兼容既有三种约定,新类型是可选项而非迁移要求)。
· internal/sdk: 补 ToolError 别名。
· core/toolerror.go: isToolError / toolErrorText / renderToolResult。
⚠️ 判据必须同时兼容仓内**三种**既有失败约定,且**不得**把成功误判:
① {"error": msg} ② {"isError":true,content:…} ③ *ToolError
明确不判失败的:exit_code != 0(业务结果,带真实 stdout/stderr)、
stderr 非空(cmd 成功常带 warn)、字符串/数字/bool/数组/nil
(自由文本按成功处理:宁可少报失败,也不把正常结果报成失败)。
· core/toolcall.go: executeToolCallOutcome 返回 toolOutcome{Text,Raw},
**Raw 必须在成功分支也带上**——否则结构化失败在 fmt.Sprintf("%v")
那一步被抹平,Success 又退回恒真。panic 与 60s 超时统一以 ToolError
表达(可执行 Hint,避免模型原样重试工具故障)。
· core/task.go: Success=!isToolError(Raw);修恒失败的类型断言;
TaskFrame 增 CurRaw(未降级的原值)。
判据:
· toolerror_test.go 单元级:三种失败约定识别 / 成功形态不误判 /
非零退出不算工具失败 / ToolError 识别。
· TestStageCtxSuccessIsHonestEndToEnd 端到端读 f.StageCtx.ToolResults,
验证**内核产出的值本身**,而非辅助函数。
变异验证(两轮):
· Success 退回硬编码 true ⇒ 端到端判据 3 个子用例 FAIL;
· 把字符串判据改成 strings.Contains(x,"error") ⇒ 「成功文本含 error 字样」
用例 FAIL。**第二轮暴露了判据缺口**(最初没有该用例),已补。
过程中三次自伤(均由判据/编译暴露):用正则批量包装 return 时把
多行 fmt.Sprintf 截断;包装范围溢出到返回 string 的辅助函数;
测试里重复注册同名工具导致 IOManager 取到错误的 device。
遗留:全量 go test ./internal/... 在本机无法完整跑完——/tmp 是 9.8G
tmpfs 且已 98% 占用,link 阶段报 "no space left on device";
/var/tmp 另有约 29G 陈旧 release worktree。与本次改动无关(未触碰
memory/* 等失败包),已在干净基线(7a566d5)对比确认。
This commit is contained in:
@ -157,7 +157,7 @@ func TestTruncatedToolCallIsShortCircuited(t *testing.T) {
|
||||
},
|
||||
}
|
||||
a := &Agent{}
|
||||
got := a.executeToolCallInner(tc, "webui", nil)
|
||||
got := a.executeToolCallInner(tc, "webui", nil).Text
|
||||
|
||||
if strings.Contains(got, "path is required") {
|
||||
t.Errorf("截断后仍走了工具分派(模型会原样重试):%s", got)
|
||||
|
||||
@ -110,7 +110,11 @@ type TaskFrame struct {
|
||||
CurTool agentAPI.ToolCall
|
||||
CurToolPlugin string
|
||||
CurResult string
|
||||
Resp *agentAPI.CompletionResponse
|
||||
// CurRaw 是本次执行的**未降级**返回值(interface{})。
|
||||
// 存在理由:CurResult 是 string,结构化信息在此被抹平,导致 Success
|
||||
// 无法诚实化、after_toolcall 的改写静默失效。
|
||||
CurRaw interface{}
|
||||
Resp *agentAPI.CompletionResponse
|
||||
|
||||
// Scene 是本轮**涌现**出来的场景键(由场面指纹聚类得到,无人声明),
|
||||
// sceneDone 标记是否已解析过——一轮只解析一次:多解析一次就多给场景
|
||||
@ -747,14 +751,20 @@ func (a *Agent) stepToolExec(f *TaskFrame) stepOutcome {
|
||||
// 执行工具前先解析本轮场景:写侧要用它给记忆自动挂场景(主动+被动两条路),
|
||||
// 而工具步不一定走到下面的召回分支,所以不能等那里再解析。
|
||||
turn := a.resolveTurnScenes(f, f.CurTool.Name)
|
||||
result := a.executeToolCall(f.CurTool, f.OutputChannel, turn.Keys...)
|
||||
outcome := a.executeToolCallOutcome(f.CurTool, f.OutputChannel, turn.Keys...)
|
||||
result := outcome.Text
|
||||
f.CurResult = result
|
||||
f.CurRaw = outcome.Raw
|
||||
f.ToolResults = append(f.ToolResults, ToolResultItem{Name: f.CurTool.Name, Output: result})
|
||||
log.Printf("[agent] tool %s result: %s", f.CurTool.Name, truncateStr(result, 100))
|
||||
|
||||
// Success 此前是**唯一**赋值点且硬编码 true ⇒ 该字段恒真、结构上不可能为
|
||||
// false。工具失败是以 nil error + 错误**值**返回的,所以判据必须看返回值。
|
||||
// ⚠️ isToolError 必须同时覆盖存量插件的两种失败约定与「成功不误判」,
|
||||
// 否则升级会把存量插件的成功判成失败(见 toolerror_test.go)。
|
||||
f.StageCtx.ToolResults = []sdk.ToolResult{{
|
||||
CallID: f.CurTool.ID, Name: f.CurTool.Name, Plugin: f.CurToolPlugin,
|
||||
Success: true, Result: result,
|
||||
Success: !isToolError(outcome.Raw), Result: outcome.Raw,
|
||||
}}
|
||||
f.Step = StepToolAfter
|
||||
return outcomeContinue
|
||||
@ -768,8 +778,20 @@ func (a *Agent) stepToolAfter(f *TaskFrame) stepOutcome {
|
||||
|
||||
a.runStage(sdk.StageAfterToolcall, f.StageCtx)
|
||||
if len(f.StageCtx.ToolResults) > 0 {
|
||||
if r, ok := f.StageCtx.ToolResults[0].Result.(string); ok {
|
||||
// 此前是 `Result.(string)` 类型断言,而插件返回的多是 map ⇒ 断言几乎
|
||||
// 恒失败,after_toolcall 阶段对结构化结果的改写**静默失效**。
|
||||
// 改为:字符串就替换文本;结构化值则保留其原值并按契约渲染。
|
||||
switch r := f.StageCtx.ToolResults[0].Result.(type) {
|
||||
case string:
|
||||
result = r
|
||||
case nil:
|
||||
// 插件清空结果:保持原样,不覆盖。
|
||||
default:
|
||||
f.CurRaw = r
|
||||
result = toolErrorText(f.CurTool.Name, r)
|
||||
if !isToolError(r) {
|
||||
result = fmt.Sprintf("%v", r)
|
||||
}
|
||||
}
|
||||
}
|
||||
// 工具后处理:一次相关性过程,两个**正交**声明——
|
||||
|
||||
@ -15,7 +15,25 @@ import (
|
||||
"gitcode.com/JianFeeeee/HomeAgent/internal/memory/text"
|
||||
)
|
||||
|
||||
func (a *Agent) executeToolCall(tc agentAPI.ToolCall, channel string, turnScenes ...string) (ret string) {
|
||||
// toolOutcome 是一次工具执行的完整结果:**文本**(给模型)与
|
||||
// **原值**(给契约判断)分开携带。
|
||||
//
|
||||
// 为何必须分开:executeToolCall 历来只返回 string,结构化信息在这一步
|
||||
// 被抹平,导致(a)ToolResult.Success 无法诚实化、(b)after_toolcall
|
||||
// 阶段插件对结构化结果的改写因类型断言失败而静默失效。
|
||||
type toolOutcome struct {
|
||||
Text string
|
||||
Raw interface{}
|
||||
}
|
||||
|
||||
// executeToolCall 保留原签名(spawn.go 与既有测试依赖),只取文本。
|
||||
func (a *Agent) executeToolCall(tc agentAPI.ToolCall, channel string, turnScenes ...string) string {
|
||||
return a.executeToolCallOutcome(tc, channel, turnScenes...).Text
|
||||
}
|
||||
|
||||
// executeToolCallOutcome 是完整形态:崩溃/超时同样以 ToolError 表达,
|
||||
// 使「工具故障」与「工具报告的业务失败」在上层可区分。
|
||||
func (a *Agent) executeToolCallOutcome(tc agentAPI.ToolCall, channel string, turnScenes ...string) (out toolOutcome) {
|
||||
defer func() {
|
||||
if r := recover(); r != nil {
|
||||
stack := debug.Stack()
|
||||
@ -27,11 +45,13 @@ func (a *Agent) executeToolCall(tc agentAPI.ToolCall, channel string, turnScenes
|
||||
}
|
||||
}
|
||||
|
||||
ret = fmt.Sprintf("工具 %s 执行崩溃: %v", tc.Name, r)
|
||||
te := newToolError("panic", "", fmt.Sprintf("工具 %s 执行崩溃: %v", tc.Name, r),
|
||||
"这是工具自身故障(不是你的参数问题),请勿原样重试;可换用其他工具或告知用户。")
|
||||
out = toolOutcome{Text: te.Error(), Raw: te}
|
||||
}
|
||||
}()
|
||||
|
||||
done := make(chan string, 1)
|
||||
done := make(chan toolOutcome, 1)
|
||||
go func() {
|
||||
done <- a.executeToolCallInner(tc, channel, turnScenes)
|
||||
}()
|
||||
@ -41,70 +61,74 @@ func (a *Agent) executeToolCall(tc agentAPI.ToolCall, channel string, turnScenes
|
||||
return result
|
||||
case <-time.After(60 * time.Second):
|
||||
log.Printf("[agent] tool %s timed out after 60s", tc.Name)
|
||||
return fmt.Sprintf("工具 %s 执行超时(60秒),已取消", tc.Name)
|
||||
te := newToolError(ErrReasonTimeout, "", fmt.Sprintf("工具 %s 执行超时(60秒)", tc.Name),
|
||||
"该工具本次未在时限内返回。可改用更小的任务,或换用其他工具。")
|
||||
return toolOutcome{Text: te.Error(), Raw: te}
|
||||
}
|
||||
}
|
||||
|
||||
func (a *Agent) executeToolCallInner(tc agentAPI.ToolCall, channel string, turnScenes []string) string {
|
||||
func (a *Agent) executeToolCallInner(tc agentAPI.ToolCall, channel string, turnScenes []string) toolOutcome {
|
||||
// 参数没法用(被 max_tokens 截断,或 JSON 写坏了):**不要**拿着空/残缺参数去调工具。
|
||||
// 否则工具会报 “path is required”“command is required” 这类与真因无关的错,
|
||||
// 模型看不出真因、只能原样重试(实测 cmd_run 失败率高达 34%~48%)。
|
||||
// __arg_error 里带的已经是分因写好的可执行指引,直接交回模型。
|
||||
if msg, ok := tc.Arguments["__arg_error"].(string); ok && msg != "" {
|
||||
log.Printf("[agent] tool %s skipped: arguments unusable (truncated or malformed)", tc.Name)
|
||||
return msg
|
||||
return toolOutcome{Text: msg}
|
||||
}
|
||||
|
||||
switch {
|
||||
case tc.Name == "persona_set":
|
||||
return a.executePersonaTool(tc)
|
||||
return toolOutcome{Text: a.executePersonaTool(tc)}
|
||||
case strings.HasPrefix(tc.Name, "memory_"):
|
||||
return a.executeMemoryTool(tc, turnScenes)
|
||||
return toolOutcome{Text: a.executeMemoryTool(tc, turnScenes)}
|
||||
case strings.HasPrefix(tc.Name, "social_"):
|
||||
return a.executeSocialTool(tc)
|
||||
return toolOutcome{Text: a.executeSocialTool(tc)}
|
||||
case strings.HasPrefix(tc.Name, "knowledge_"):
|
||||
return a.executeKnowledgeTool(tc)
|
||||
return toolOutcome{Text: a.executeKnowledgeTool(tc)}
|
||||
case strings.HasPrefix(tc.Name, "doc_"):
|
||||
return a.executeDocTool(tc)
|
||||
return toolOutcome{Text: a.executeDocTool(tc)}
|
||||
case strings.HasPrefix(tc.Name, "output_send__") && strings.HasSuffix(tc.Name, "_help"):
|
||||
return a.executeOutputSendHelp(tc)
|
||||
return toolOutcome{Text: a.executeOutputSendHelp(tc)}
|
||||
case strings.HasPrefix(tc.Name, "output_send__"):
|
||||
return a.executeOutputSendTool(tc)
|
||||
return toolOutcome{Text: a.executeOutputSendTool(tc)}
|
||||
case tc.Name == "output_list_channels":
|
||||
return a.executeOutputListChannels()
|
||||
return toolOutcome{Text: a.executeOutputListChannels()}
|
||||
case tc.Name == "input_channels":
|
||||
return a.executeInputChannels(tc)
|
||||
return toolOutcome{Text: a.executeInputChannels(tc)}
|
||||
case tc.Name == "resident_agents":
|
||||
return a.executeResidentAgents(tc)
|
||||
return toolOutcome{Text: a.executeResidentAgents(tc)}
|
||||
case tc.Name == "notify_parent":
|
||||
return a.executeNotifyParent(tc)
|
||||
return toolOutcome{Text: a.executeNotifyParent(tc)}
|
||||
case tc.Name == "inputch_note":
|
||||
return a.executeInputchNote(tc)
|
||||
return toolOutcome{Text: a.executeInputchNote(tc)}
|
||||
case tc.Name == "plgreload":
|
||||
return a.executePluginReload()
|
||||
return toolOutcome{Text: a.executePluginReload()}
|
||||
case tc.Name == "get_plugin_tools":
|
||||
pluginName, _ := tc.Arguments["plugin_name"].(string)
|
||||
return a.executeGetPluginTools(pluginName)
|
||||
return toolOutcome{Text: a.executeGetPluginTools(pluginName)}
|
||||
case tc.Name == "spawn_child":
|
||||
return a.executeSpawnChild(tc, channel)
|
||||
return toolOutcome{Text: a.executeSpawnChild(tc, channel)}
|
||||
case tc.Name == "child_result":
|
||||
return a.executeChildResultTool(tc)
|
||||
return toolOutcome{Text: a.executeChildResultTool(tc)}
|
||||
case strings.HasPrefix(tc.Name, "llm_"):
|
||||
return a.executeLLMTool(tc)
|
||||
return toolOutcome{Text: a.executeLLMTool(tc)}
|
||||
case tc.Name == "describe_image":
|
||||
return a.executeDescribeImage(tc)
|
||||
return toolOutcome{Text: a.executeDescribeImage(tc)}
|
||||
case tc.Name == "transcribe_audio":
|
||||
return a.executeTranscribeAudio(tc)
|
||||
return toolOutcome{Text: a.executeTranscribeAudio(tc)}
|
||||
case tc.Name == "ocr_image":
|
||||
return a.executeOCRImage(tc)
|
||||
return toolOutcome{Text: a.executeOCRImage(tc)}
|
||||
}
|
||||
|
||||
if a.stageHost != nil {
|
||||
if result, err := a.stageHost.ExecuteTool(tc.Name, tc.Arguments); err == nil {
|
||||
return fmt.Sprintf("%v", result)
|
||||
// Raw 必须带上:否则结构化失败({"error":…} / ToolError)在这一步被抹平成文本,
|
||||
// Success 又会退回恒真——正是阶段 1b 要修的那个洞。
|
||||
return toolOutcome{Text: renderToolResult(tc.Name, result), Raw: result}
|
||||
} else if !agentIO.IsToolNotFound(err) {
|
||||
// 非「不存在」= 真的执行失败,如实上报(可被 on_error/retry 处置)。
|
||||
return fmt.Sprintf("工具 %s 执行失败: %v", tc.Name, err)
|
||||
return toolOutcome{Text: fmt.Sprintf("工具 %s 执行失败: %v", tc.Name, err)}
|
||||
}
|
||||
// 是「不存在」:继续往下走 io / 设备路径,两处都没有才报缺工具。
|
||||
}
|
||||
@ -117,7 +141,7 @@ func (a *Agent) executeToolCallInner(tc agentAPI.ToolCall, channel string, turnS
|
||||
// 这里按目标设备的通道名 device/<id> 查同一道闸:父授权了哪台设备,才允许指挥哪台。
|
||||
if _, isDeviceTool := a.io.DeviceOfTool(tc.Name); isDeviceTool {
|
||||
if id, _ := tc.Arguments["device_id"].(string); id != "" && !a.IsOutputAllowed("device/"+id) {
|
||||
return fmt.Sprintf("设备 [%s] 未授权给本 agent(可用设备见 output_list_channels 的 device/<id> 通道,或 devicedetect)", id)
|
||||
return toolOutcome{Text: fmt.Sprintf("设备 [%s] 未授权给本 agent(可用设备见 output_list_channels 的 device/<id> 通道,或 devicedetect)", id)}
|
||||
}
|
||||
}
|
||||
|
||||
@ -135,13 +159,15 @@ func (a *Agent) executeToolCallInner(tc agentAPI.ToolCall, channel string, turnS
|
||||
// 工具是动态注册的,"不存在"是常态而非异常(插件未加载/已卸载/崩溃)。
|
||||
// 文案必须让模型知道该做什么,而不是含糊的"执行失败"——
|
||||
// 后者会让模型反复重试同一个不存在的名字。
|
||||
return fmt.Sprintf("工具 %s 不存在或未注册:它可能属于未加载/已崩溃的插件。"+
|
||||
return toolOutcome{Text: fmt.Sprintf("工具 %s 不存在或未注册:它可能属于未加载/已崩溃的插件。"+
|
||||
"先调 get_plugin_tools(\"\") 看当前可用工具,或 output_list_channels 看通道;"+
|
||||
"确认名称无误后再调用", tc.Name)
|
||||
"确认名称无误后再调用", tc.Name)}
|
||||
}
|
||||
return fmt.Sprintf("工具 %s 执行失败: %v", tc.Name, err)
|
||||
return toolOutcome{Text: fmt.Sprintf("工具 %s 执行失败: %v", tc.Name, err)}
|
||||
}
|
||||
return fmt.Sprintf("%v", result)
|
||||
// Raw 必须带上:否则结构化失败({"error":…} / ToolError)在这一步被抹平成文本,
|
||||
// Success 又会退回恒真——正是阶段 1b 要修的那个洞。
|
||||
return toolOutcome{Text: renderToolResult(tc.Name, result), Raw: result}
|
||||
}
|
||||
|
||||
// toolNotFound / isToolNotFound 是 agentIO 哨兵在 core 侧的薄封装,
|
||||
|
||||
154
internal/agent/core/toolerror.go
Normal file
154
internal/agent/core/toolerror.go
Normal file
@ -0,0 +1,154 @@
|
||||
package core
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
sdk "gitcode.com/JianFeeeee/HomeAgent/internal/sdk"
|
||||
)
|
||||
|
||||
// 工具失败原因码(与 SDK 的 ToolError.Reason 对应)。
|
||||
const (
|
||||
ErrReasonRequired = "required"
|
||||
ErrReasonType = "type"
|
||||
ErrReasonUnauthorized = "unauthorized"
|
||||
ErrReasonTimeout = "timeout"
|
||||
ErrReasonNotFound = "not_found"
|
||||
)
|
||||
|
||||
// newToolError 构造一个结构化失败。
|
||||
func newToolError(reason, field, detail, hint string) *sdk.ToolError {
|
||||
return &sdk.ToolError{Reason: reason, Field: field, Detail: detail, Hint: hint}
|
||||
}
|
||||
|
||||
// isToolError 报告一个工具返回值是否表示**失败**。
|
||||
//
|
||||
// 存在的理由:工具失败是以 `nil` error + 错误**值**返回的,而
|
||||
// ToolResult.Success 此前被硬编码为 true(唯一赋值点),该字段恒真、
|
||||
// 结构上不可能为 false。
|
||||
//
|
||||
// ⚠️ 判据必须兼容**既有三种约定**(实测于仓内,否则升级会把存量插件
|
||||
// 的成功误判成失败——这是本函数最大的回归风险):
|
||||
//
|
||||
// ① {"error": msg} pluginmgr、cmd 的参数校验
|
||||
// ② {"isError": true, "content": msg} files / clawhubadapter 的 errorResult
|
||||
// ③ *sdk.ToolError 新写的工具(可选,不是迁移要求)
|
||||
//
|
||||
// 明确**不**作为失败判据的:
|
||||
// - exit_code != 0:命令跑了但返回非零,属业务结果且带真实 stdout/stderr,
|
||||
// 整条判失败会误伤「命令可用但结果非零」这类正常场景。
|
||||
// - stderr 非空:cmd 成功路径常带 stderr(如 warn: deprecated)。
|
||||
// - 字符串 / 数字 / bool / 数组 / nil:均视为成功(output_send 成功即返回 "ok")。
|
||||
func isToolError(v interface{}) bool {
|
||||
switch x := v.(type) {
|
||||
case nil:
|
||||
return false
|
||||
case *sdk.ToolError:
|
||||
return x != nil
|
||||
case sdk.ToolError:
|
||||
return true
|
||||
case error:
|
||||
// 工具显式返回 error —— 失败。
|
||||
return x != nil
|
||||
case map[string]interface{}:
|
||||
// ② isError 优先:显式布尔标记,语义最明确。
|
||||
if b, ok := x["isError"].(bool); ok && b {
|
||||
return true
|
||||
}
|
||||
// ① error 键:非空字符串才算失败。
|
||||
if e, ok := x["error"]; ok {
|
||||
switch ev := e.(type) {
|
||||
case string:
|
||||
return strings.TrimSpace(ev) != ""
|
||||
case nil:
|
||||
return false
|
||||
default:
|
||||
// error 是结构化值(如嵌套的 ToolError)—— 视为失败。
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
case string:
|
||||
// 自由文本无法可靠判别成败,按**成功**处理(保守:宁可少报失败,
|
||||
// 也不要把正常结果报成失败)。失败请用上面三种显式形态。
|
||||
return false
|
||||
default:
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
// toolErrorText 把一个失败返回值渲染成给模型看的文本。
|
||||
// 结构化 ToolError 会带上 Hint——这是「让模型看得懂真因」的关键。
|
||||
func toolErrorText(toolName string, v interface{}) string {
|
||||
switch x := v.(type) {
|
||||
case *sdk.ToolError:
|
||||
return renderToolError(toolName, x)
|
||||
case sdk.ToolError:
|
||||
return renderToolError(toolName, &x)
|
||||
case map[string]interface{}:
|
||||
if b, ok := x["isError"].(bool); ok && b {
|
||||
msg, _ := x["content"].(string)
|
||||
if msg == "" {
|
||||
msg = "(插件未给出原因)"
|
||||
}
|
||||
return fmt.Sprintf("工具 %s 失败: %s", toolName, msg)
|
||||
}
|
||||
if e, ok := x["error"]; ok {
|
||||
if s, ok := e.(string); ok {
|
||||
return fmt.Sprintf("工具 %s 失败: %s", toolName, s)
|
||||
}
|
||||
}
|
||||
case error:
|
||||
return fmt.Sprintf("工具 %s 失败: %v", toolName, x)
|
||||
}
|
||||
return fmt.Sprintf("工具 %s 失败: %v", toolName, v)
|
||||
}
|
||||
|
||||
// renderToolError 渲染结构化失败:把 field/reason/hint 都摆出来,
|
||||
// 让模型知道**该改什么**,而不是只知道「失败了」。
|
||||
func renderToolError(toolName string, e *sdk.ToolError) string {
|
||||
var sb strings.Builder
|
||||
fmt.Fprintf(&sb, "工具 %s 失败", toolName)
|
||||
if e.Field != "" {
|
||||
fmt.Fprintf(&sb, "(字段 %s)", e.Field)
|
||||
}
|
||||
sb.WriteString(": ")
|
||||
if e.Reason != "" {
|
||||
sb.WriteString(e.Reason)
|
||||
}
|
||||
if e.Detail != "" {
|
||||
sb.WriteString(" — ")
|
||||
sb.WriteString(e.Detail)
|
||||
}
|
||||
if e.Hint != "" {
|
||||
sb.WriteString("\n请据此修正后重试:")
|
||||
sb.WriteString(e.Hint)
|
||||
}
|
||||
return sb.String()
|
||||
}
|
||||
|
||||
// renderToolResult 把工具返回值渲染成给模型看的文本。
|
||||
//
|
||||
// 规则:**结构化失败优先**——失败必须带上可执行信息(字段/原因/Hint),
|
||||
// 而不是退化成 `map[error:xxx]` 这种模型读不懂的 Go 语法。
|
||||
// 成功则用紧凑 JSON(绝不用 fmt.Sprintf("%v"),那会产出 Go 的 map 语法)。
|
||||
func renderToolResult(toolName string, raw interface{}) string {
|
||||
if raw == nil {
|
||||
return ""
|
||||
}
|
||||
if isToolError(raw) {
|
||||
return toolErrorText(toolName, raw)
|
||||
}
|
||||
switch x := raw.(type) {
|
||||
case string:
|
||||
return x
|
||||
case error:
|
||||
return fmt.Sprintf("%v", x)
|
||||
}
|
||||
// 非字符串:紧凑 JSON。
|
||||
if b, err := json.Marshal(raw); err == nil {
|
||||
return string(b)
|
||||
}
|
||||
return fmt.Sprintf("%v", raw)
|
||||
}
|
||||
164
internal/agent/core/toolerror_test.go
Normal file
164
internal/agent/core/toolerror_test.go
Normal file
@ -0,0 +1,164 @@
|
||||
package core
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
agentAPI "gitcode.com/JianFeeeee/HomeAgent/internal/agent/api"
|
||||
agentIO "gitcode.com/JianFeeeee/HomeAgent/internal/agent/io"
|
||||
)
|
||||
|
||||
// 阶段 1b:诚实化 Success。
|
||||
//
|
||||
// 现状(task.go:757):`Success: true` 是**唯一**赋值点 ⇒ 该字段恒真,
|
||||
// 结构上不可能为 false。而工具失败是以 `nil` error + 错误**值**返回的
|
||||
// (`files/plugin.go:228` 的 `errorResult(...)`、pluginmgr 的 `{"error":…}, nil`)。
|
||||
//
|
||||
// 本判据钉死「哪些返回值算失败」。**最大回归风险**在此:
|
||||
// 判据若只认「error 键」而不认「普通 map/string 仍算成功」,
|
||||
// 升级就会把存量插件的**成功**误判成失败。
|
||||
|
||||
// 失败形态在仓内有**三种**约定(已核实,见各出处):
|
||||
//
|
||||
// ① {"error": msg} —— pluginmgr、cmd 的参数校验
|
||||
// ② {"isError": true, "content": msg} —— files、clawhubadapter 的 errorResult
|
||||
// ③ {"status":"timeout", "stdout":…, "error":…} —— cmd 超时(带真实数据,status 才是判据)
|
||||
func TestIsToolError_RecognizesLegacyFailureShapes(t *testing.T) {
|
||||
failures := []struct {
|
||||
name string
|
||||
val interface{}
|
||||
}{
|
||||
{"①error 键", map[string]interface{}{"error": "name is required"}},
|
||||
{"①error 键+其他字段", map[string]interface{}{"error": "boom", "stdout": "partial"}},
|
||||
{"②isError", map[string]interface{}{"isError": true, "content": "path is required"}},
|
||||
{"②isError=false 不算失败", map[string]interface{}{"isError": false, "content": "ok"}},
|
||||
}
|
||||
for _, c := range failures {
|
||||
want := c.name != "②isError=false 不算失败"
|
||||
if got := isToolError(c.val); got != want {
|
||||
t.Errorf("%s: isToolError = %v,期望 %v(值 %#v)", c.name, got, want, c.val)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// 成功形态**绝不能**被判成失败——这是升级的头号回归风险。
|
||||
func TestIsToolError_SuccessShapesAreNotFailures(t *testing.T) {
|
||||
successes := []struct {
|
||||
name string
|
||||
val interface{}
|
||||
}{
|
||||
{"cmd 成功(含 stderr,命令本身常报 stderr 但不是工具失败)", map[string]interface{}{
|
||||
"status": "ok", "stdout": "out", "stderr": "warn: deprecated", "exit_code": 0,
|
||||
}},
|
||||
{"普通 map", map[string]interface{}{"count": 3, "items": []interface{}{"a"}}},
|
||||
{"空 map", map[string]interface{}{}},
|
||||
{"字符串", "已通过 [webui] 通道发送"},
|
||||
{"ok 标记", "ok"},
|
||||
// ⚠️ 最高风险的一条:成功的**文本里恰好含 error 字样**。
|
||||
// 若判据用 strings.Contains(x, "error") 之类,成功就会被判成失败——
|
||||
// 而 cmd_run 的 stderr、检索到的日志片段都可能含这个词。
|
||||
{"成功文本含 error 字样", "已处理 3 个 error 日志(命令退出码 0)"},
|
||||
{"成功文本含 isError 字样", `{"isError": false, "note": "已检查"}`},
|
||||
{"nil", nil},
|
||||
{"bool", true},
|
||||
{"数字", 42},
|
||||
{"数组", []interface{}{"a", "b"}},
|
||||
}
|
||||
for _, c := range successes {
|
||||
if isToolError(c.val) {
|
||||
t.Errorf("成功形态被误判为失败: %s(%#v)", c.name, c.val)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// 非零退出码:cmd 的 `exit_code != 0` 属**业务失败**而非工具故障。
|
||||
// 但它带 `status: "ok"` 与真实 stdout——不应整条判为失败,
|
||||
// 否则「命令跑了但返回非零」会被误当成工具不可用。
|
||||
// ⇒ 本判据锁定当前语义:**只看显式错误标记**,不看 exit_code。
|
||||
func TestIsToolError_NonZeroExitIsNotToolFailure(t *testing.T) {
|
||||
v := map[string]interface{}{"status": "ok", "stdout": "", "stderr": "boom", "exit_code": 1}
|
||||
if isToolError(v) {
|
||||
t.Errorf("非零退出码不应整条判为工具失败(它带真实 stdout/stderr): %#v", v)
|
||||
}
|
||||
}
|
||||
|
||||
// 显式 ToolError 结构必须被识别(阶段 1a 的新形态)。
|
||||
func TestIsToolError_RecognizesStructuredToolError(t *testing.T) {
|
||||
if !isToolError(newToolError(ErrReasonRequired, "path", "缺少 path", "先传 path")) {
|
||||
t.Error("结构化 ToolError 应被识别为失败")
|
||||
}
|
||||
}
|
||||
|
||||
// 端到端:失败工具的 Success 必须是 false,成功工具必须是 true。
|
||||
//
|
||||
// 这是阶段 1b 的**真正目标**——此前 `Success: true` 是唯一赋值点,
|
||||
// 恒真、结构上不可能为 false。本判据经 after_toolcall stage 直接读
|
||||
// f.StageCtx.ToolResults,验证内核产出的值本身,而非某个辅助函数。
|
||||
func TestStageCtxSuccessIsHonestEndToEnd(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
toolRet interface{}
|
||||
wantSucc bool
|
||||
wantHint string // 期望出现在回填文本里的片段
|
||||
}{
|
||||
{"成功返回 ok", "ok", true, ""},
|
||||
{"成功返回结构化 map", map[string]interface{}{"status": "ok", "exit_code": 0}, true, ""},
|
||||
{"①error 约定失败", map[string]interface{}{"error": "name is required"}, false, "name is required"},
|
||||
{"②isError 约定失败", map[string]interface{}{"isError": true, "content": "path is required"}, false, "path is required"},
|
||||
{"结构化 ToolError 失败", newToolError(ErrReasonRequired, "path", "缺少 path", "先传 path 参数"), false, "先传 path 参数"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
sp := &batchProvider{responses: []*agentAPI.CompletionResponse{
|
||||
{ToolCalls: []agentAPI.ToolCall{{ID: "c1", Name: "tool_ret", Arguments: map[string]interface{}{}}}},
|
||||
{Content: "final"},
|
||||
}}
|
||||
a, _ := newBatchAgent(t, sp)
|
||||
// 覆盖设备工具的返回值为本用例的样本。
|
||||
a.io.RegisterDevice(&retDevice{name: "retdev", toolName: "tool_ret", ret: c.toolRet})
|
||||
|
||||
f := a.newTaskFrame("go", a.stageCtxFromInput("go", "", ""))
|
||||
if out := a.runTaskSteps(f); out != outcomeDone {
|
||||
t.Fatalf("runTaskSteps=%v err=%v", out, f.Err)
|
||||
}
|
||||
if len(f.StageCtx.ToolResults) == 0 {
|
||||
t.Fatal("StageCtx.ToolResults 为空")
|
||||
}
|
||||
tr := f.StageCtx.ToolResults[0]
|
||||
if tr.Success != c.wantSucc {
|
||||
t.Errorf("Success = %v,期望 %v(返回值 %#v)", tr.Success, c.wantSucc, c.toolRet)
|
||||
}
|
||||
if c.wantHint != "" {
|
||||
// 结构化失败的 Hint 必须真的回填给模型,否则「看得懂真因」落空。
|
||||
found := false
|
||||
for _, m := range f.Msgs {
|
||||
if m.Role == "tool" && strings.Contains(m.Content, c.wantHint) {
|
||||
found = true
|
||||
}
|
||||
}
|
||||
if !found {
|
||||
t.Errorf("失败详情 %q 未回填给模型", c.wantHint)
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// retDevice 是一个按预设值返回的测试设备。
|
||||
type retDevice struct {
|
||||
name string
|
||||
toolName string
|
||||
ret interface{}
|
||||
}
|
||||
|
||||
func (d *retDevice) Name() string { return d.name }
|
||||
func (d *retDevice) Type() agentIO.DeviceType { return agentIO.DeviceOutput }
|
||||
func (d *retDevice) Description() string { return "ret test device" }
|
||||
func (d *retDevice) Tools() []agentIO.ToolDef { return []agentIO.ToolDef{{Name: d.toolName}} }
|
||||
func (d *retDevice) Execute(string, map[string]interface{}) (interface{}, error) {
|
||||
return d.ret, nil
|
||||
}
|
||||
func (d *retDevice) Start() error { return nil }
|
||||
func (d *retDevice) Stop() error { return nil }
|
||||
func (d *retDevice) OutputCapabilities() agentIO.OutputCapability { return agentIO.CapText }
|
||||
func (d *retDevice) ChannelDef() agentIO.ChannelDef { return agentIO.ChannelDef{} }
|
||||
@ -37,6 +37,7 @@ type StageContext = pubsdk.StageContext
|
||||
type MemItem = pubsdk.MemItem
|
||||
type ToolCall = pubsdk.ToolCall
|
||||
type ToolResult = pubsdk.ToolResult
|
||||
type ToolError = pubsdk.ToolError
|
||||
type ToolDef = pubsdk.ToolDef
|
||||
type IOInjector = pubsdk.IOInjector
|
||||
type ToolRegistrar = pubsdk.ToolRegistrar
|
||||
|
||||
35
third_party/homeagent-sdk/sdk/plugin.go
vendored
35
third_party/homeagent-sdk/sdk/plugin.go
vendored
@ -240,6 +240,41 @@ type ToolResult struct {
|
||||
Result interface{} `json:"result"`
|
||||
}
|
||||
|
||||
// ToolError 描述一次工具调用的失败原因。
|
||||
//
|
||||
// 存在的理由:失败若只表达为文本,模型无法定位到字段,只能原样重试
|
||||
// (实测 cmd_run 失败率 34%~48%,全部源于同一个成因:参数被截断或
|
||||
// JSON 写坏,工具却只回报 "command is required" 这类与真因无关的错)。
|
||||
//
|
||||
// ⚠️ 零值语义:插件**不必**改用本类型。内核的失败识别同时兼容既有三种约定
|
||||
// ({"error":…}、{"isError":true,…}、显式 error 返回),见 core.isToolError。
|
||||
// 本类型是给**新写**的工具用的可选项,不是迁移要求。
|
||||
type ToolError struct {
|
||||
// Field 是出错的参数字段名(参数校验失败时填)。
|
||||
Field string `json:"field,omitempty"`
|
||||
// Reason 是机器可读的原因码:required / type / unauthorized / timeout / not_found。
|
||||
Reason string `json:"reason"`
|
||||
// Detail 是人类可读的补充说明。
|
||||
Detail string `json:"detail,omitempty"`
|
||||
// Hint 是给模型的可执行指引(该改什么、不要重试什么)。
|
||||
Hint string `json:"hint,omitempty"`
|
||||
}
|
||||
|
||||
// Error 实现 error,便于工具同时走 (ToolError, error) 通道。
|
||||
func (e *ToolError) Error() string {
|
||||
if e == nil {
|
||||
return ""
|
||||
}
|
||||
s := e.Reason
|
||||
if e.Field != "" {
|
||||
s = e.Field + ": " + s
|
||||
}
|
||||
if e.Detail != "" {
|
||||
s += " (" + e.Detail + ")"
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
// ToolDef describes a tool that the plugin exposes.
|
||||
type ToolDef struct {
|
||||
Name string `json:"name"`
|
||||
|
||||
Reference in New Issue
Block a user