From bbfcb87103ab67163df2134954ace6676af039b7 Mon Sep 17 00:00:00 2001 From: Ed1s0nZ Date: Wed, 23 Sep 2026 18:01:11 +0800 Subject: [PATCH] fix: avoid duplicate summarization token limits --- config.example.yaml | 2 +- internal/config/config.go | 4 +- internal/multiagent/eino_summarize.go | 13 +++-- .../eino_summarize_model_guard_test.go | 5 +- .../multiagent/eino_summarize_payload_test.go | 56 +++++++++++++++++-- 5 files changed, 65 insertions(+), 15 deletions(-) diff --git a/config.example.yaml b/config.example.yaml index 1b334381..05fe6e00 100644 --- a/config.example.yaml +++ b/config.example.yaml @@ -315,7 +315,7 @@ multi_agent: reduction_clear_exclude: [] # 不参与「清理阶段」的工具名额外列表(会与 task/transfer/exit 等内置排除项合并);需要时用 YAML 列表填写 reduction_sub_agents: true # true:子代理也挂 reduction;false:仅编排主代理使用 reduction summarization_trigger_ratio: 0.8 # summarization 触发比例(max_total_tokens * ratio),建议 0.75~0.85 - summarization_output_reserve_tokens: 8192 # 摘要模型输出预留 token;摘要输入预算 = 触发阈值 - 该值 + summarization_output_reserve_tokens: 40960 # 摘要模型输出预留 token;摘要输入预算 = 触发阈值 - 该值 summarization_emit_internal_events: true # true:发出 summarization 内部事件(便于诊断) summarization_user_intent_ledger_max_runes: 96000 # 压缩后注入模型上下文的「原始用户输入与约束账本」总字符上限;DB 原始消息不裁剪 summarization_user_intent_ledger_entry_max_runes: 16000 # 账本中单条用户消息的字符上限;超出仅裁剪模型可见账本,不影响 DB 原文 diff --git a/internal/config/config.go b/internal/config/config.go index 3b229176..bcf13c00 100644 --- a/internal/config/config.go +++ b/internal/config/config.go @@ -62,7 +62,7 @@ const ( DefaultLatestUserMessageMaxRunes = 48000 DefaultLatestUserMessageHeadRunes = 24000 DefaultLatestUserMessageTailRunes = 24000 - DefaultSummarizationOutputReserveTokens = 8192 + DefaultSummarizationOutputReserveTokens = 40960 ) // ProjectConfig 项目黑板(跨对话共享事实)配置。 @@ -277,7 +277,7 @@ type MultiAgentEinoMiddlewareConfig struct { ReductionSubAgents bool `yaml:"reduction_sub_agents,omitempty" json:"reduction_sub_agents,omitempty"` // also attach to sub-agents // SummarizationTriggerRatio controls summarization trigger threshold as max_total_tokens * ratio (default 0.8). SummarizationTriggerRatio float64 `yaml:"summarization_trigger_ratio,omitempty" json:"summarization_trigger_ratio,omitempty"` - // SummarizationOutputReserveTokens reserves completion headroom for the summarization model call (default 8192). + // SummarizationOutputReserveTokens reserves completion headroom for the summarization model call (default 40960). SummarizationOutputReserveTokens int `yaml:"summarization_output_reserve_tokens,omitempty" json:"summarization_output_reserve_tokens,omitempty"` // SummarizationEmitInternalEvents controls middleware internal event emission (default true). SummarizationEmitInternalEvents *bool `yaml:"summarization_emit_internal_events,omitempty" json:"summarization_emit_internal_events,omitempty"` diff --git a/internal/multiagent/eino_summarize.go b/internal/multiagent/eino_summarize.go index 50678478..63eede98 100644 --- a/internal/multiagent/eino_summarize.go +++ b/internal/multiagent/eino_summarize.go @@ -301,9 +301,13 @@ func newEinoSummarizationModelOptions(outputReserve int, modelName, kind string, if strings.TrimSpace(kind) != "" && kind != "classic" { label = "eino " + kind + " summarization generate request" } - return []model.Option{ - model.WithMaxTokens(outputReserve), - einoopenai.WithMaxCompletionTokens(outputReserve), + opts := make([]model.Option, 0, 4) + if oa != nil && isEinoAgenticClaudeProvider(oa.Provider) { + opts = append(opts, model.WithMaxTokens(outputReserve)) + } else { + opts = append(opts, einoopenai.WithMaxCompletionTokens(outputReserve)) + } + opts = append(opts, einoopenai.WithExtraHeader(map[string]string{ copenai.SummarizationRequestHeader: "1", }), @@ -317,7 +321,8 @@ func newEinoSummarizationModelOptions(outputReserve int, modelName, kind string, } return stripReasoningFromSummarizationPayload(rawBody, oa) }), - } + ) + return opts } // summarizationInputBudgetOpts controls spill/truncation behavior when a round alone exceeds budget. diff --git a/internal/multiagent/eino_summarize_model_guard_test.go b/internal/multiagent/eino_summarize_model_guard_test.go index 70d701d4..8fe2da2f 100644 --- a/internal/multiagent/eino_summarize_model_guard_test.go +++ b/internal/multiagent/eino_summarize_model_guard_test.go @@ -260,12 +260,13 @@ func TestClaudeSummaryLargeBudgetStreamsThroughNativeSDK(t *testing.T) { })) defer server.Close() factory := newEinoAgenticChatModelFactory(server.Client(), nil, nil) - native, err := factory(ctx, config.OpenAIConfig{Provider: "claude", APIKey: "test-key", BaseURL: server.URL, Model: "claude-sonnet-4-20250514"}, einoModelModeNormal) + oa := config.OpenAIConfig{Provider: "claude", APIKey: "test-key", BaseURL: server.URL, Model: "claude-sonnet-4-20250514"} + native, err := factory(ctx, oa, einoModelModeNormal) if err != nil { t.Fatal(err) } input := EinoMessagesToAgentic([]*schema.Message{schema.UserMessage("summarize history")}) - opts := newEinoSummarizationModelOptions(64000, "claude-sonnet-4-20250514", "agentic", nil, nil) + opts := newEinoSummarizationModelOptions(64000, "claude-sonnet-4-20250514", "agentic", &oa, nil) if _, err = native.Generate(ctx, input, opts...); err == nil || !strings.Contains(err.Error(), "streaming is required") { t.Fatalf("expected original SDK rejection, got %v", err) } diff --git a/internal/multiagent/eino_summarize_payload_test.go b/internal/multiagent/eino_summarize_payload_test.go index 4b6890d1..186b7793 100644 --- a/internal/multiagent/eino_summarize_payload_test.go +++ b/internal/multiagent/eino_summarize_payload_test.go @@ -1,12 +1,18 @@ package multiagent import ( + "context" + "encoding/json" + "net/http" + "net/http/httptest" "strings" "testing" "cyberstrike-ai/internal/config" + einoopenai "github.com/cloudwego/eino-ext/components/model/openai" "github.com/cloudwego/eino/components/model" + "github.com/cloudwego/eino/schema" ) func TestStripReasoningFromSummarizationPayload(t *testing.T) { @@ -93,14 +99,52 @@ func TestStripReasoningFromSummarizationPayloadHonorsOpenAICompatProfileForNonDe } } -func TestEinoSummarizationModelOptionsSetCommonMaxTokens(t *testing.T) { - const outputReserve = 4096 +func TestEinoSummarizationModelOptionsSetOnlyMaxCompletionTokens(t *testing.T) { + const outputReserve = 40960 opts := newEinoSummarizationModelOptions(outputReserve, "minimax-m3", "agentic", nil, nil) common := model.GetCommonOptions(nil, opts...) - if common == nil || common.MaxTokens == nil { - t.Fatal("expected summarization options to set common max_tokens") + if common != nil && common.MaxTokens != nil { + t.Fatalf("common max_tokens = %d, want unset", *common.MaxTokens) } - if *common.MaxTokens != outputReserve { - t.Fatalf("max_tokens = %d, want %d", *common.MaxTokens, outputReserve) + + bodyCh := make(chan map[string]any, 1) + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + var body map[string]any + if err := json.NewDecoder(r.Body).Decode(&body); err != nil { + t.Errorf("decode request body: %v", err) + w.WriteHeader(http.StatusBadRequest) + return + } + bodyCh <- body + w.Header().Set("Content-Type", "text/event-stream") + w.Write([]byte("data: {\"id\":\"test\",\"choices\":[{\"index\":0,\"delta\":{\"role\":\"assistant\",\"content\":\"summary\"},\"finish_reason\":null}]}\n\n")) + w.Write([]byte("data: {\"id\":\"test\",\"choices\":[{\"index\":0,\"delta\":{},\"finish_reason\":\"stop\"}]}\n\n")) + w.Write([]byte("data: [DONE]\n\n")) + })) + defer server.Close() + + chatModel, err := einoopenai.NewChatModel(context.Background(), &einoopenai.ChatModelConfig{ + APIKey: "test-key", + BaseURL: server.URL, + Model: "gpt-4o", + HTTPClient: server.Client(), + }) + if err != nil { + t.Fatal(err) + } + out, err := newNonEmptySummaryChatModel(chatModel).Generate(context.Background(), []*schema.Message{schema.UserMessage("summarize")}, opts...) + if err != nil { + t.Fatal(err) + } + if strings.TrimSpace(out.Content) != "summary" { + t.Fatalf("summary content = %q", out.Content) + } + + body := <-bodyCh + if _, ok := body["max_tokens"]; ok { + t.Fatalf("request contained max_tokens: %#v", body) + } + if got, ok := body["max_completion_tokens"].(float64); !ok || int(got) != outputReserve { + t.Fatalf("max_completion_tokens = %#v, want %d", body["max_completion_tokens"], outputReserve) } }