mirror of
https://github.com/multica-ai/multica.git
synced 2026-07-29 06:28:23 +02:00
fix(agent/hermes): surface provider errors instead of reporting empty output
When the model the user picked isn't usable by the configured provider — e.g. ChatGPT-account auth rejecting `gpt-5.1-codex-mini` as "not supported when using Codex with a ChatGPT account" — hermes still acknowledges `session/prompt` with `stopReason=end_turn` and an empty text payload. The daemon's task machinery then reports a useless "hermes returned empty output" and the real HTTP 400 detail (the one that actually tells the user what to change) stays buried in the daemon's stderr log. Tee stderr through a bounded provider-error sniffer that scans for `⚠️` / `❌` / `[ERROR]` headers plus the `Error:` / `detail:` tails hermes emits for API failures. When the task finishes `completed` with empty output, promote it to `failed` and use the sniffed message as the task error. Nothing changes for healthy runs — the sniffer is an additional writer behind an `io.MultiWriter`, it doesn't filter or replace the normal stderr log forwarder. Reproduced against a local hermes configured for `provider: openai-codex` + `chatgpt.com/backend-api/codex`: before this change the task result was `{status: completed, comment: "hermes returned empty output"}`; after, it's `{status: failed, error: "hermes provider error: HTTP 400: ... The 'gpt-5.1-codex-mini' model is not supported when using Codex with a ChatGPT account."}`. Tests cover the real fixture shape, partial-line buffering across Writer calls, log-line filtering (info-level lines never surface as errors), and the bounded line buffer.
This commit is contained in:
@@ -5,7 +5,9 @@ import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"os/exec"
|
||||
"regexp"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
@@ -64,7 +66,15 @@ func (b *hermesBackend) Execute(ctx context.Context, prompt string, opts ExecOpt
|
||||
cancel()
|
||||
return nil, fmt.Errorf("hermes stdin pipe: %w", err)
|
||||
}
|
||||
cmd.Stderr = newLogWriter(b.cfg.Logger, "[hermes:stderr] ")
|
||||
// Forward stderr to the daemon log *and* sniff provider-level
|
||||
// errors out of it so we can surface them in the task result.
|
||||
// Hermes' session/prompt still reports stopReason=end_turn when
|
||||
// the underlying HTTP call to the LLM returns 4xx/5xx, so
|
||||
// without this we'd report a misleading "empty output" and hide
|
||||
// the real cause (wrong model for the current provider, bad
|
||||
// credentials, rate limit, …) in the daemon log.
|
||||
providerErr := newHermesProviderErrorSniffer()
|
||||
cmd.Stderr = io.MultiWriter(newLogWriter(b.cfg.Logger, "[hermes:stderr] "), providerErr)
|
||||
|
||||
if err := cmd.Start(); err != nil {
|
||||
cancel()
|
||||
@@ -266,6 +276,20 @@ func (b *hermesBackend) Execute(ctx context.Context, prompt string, opts ExecOpt
|
||||
finalOutput := output.String()
|
||||
outputMu.Unlock()
|
||||
|
||||
// If hermes produced no visible output but we sniffed a
|
||||
// provider-level error on stderr (typically HTTP 4xx from
|
||||
// the configured LLM endpoint), promote the status to
|
||||
// failed and surface the real reason. Without this the
|
||||
// daemon reports a cryptic "hermes returned empty output"
|
||||
// and the actionable error (e.g. "model X not supported
|
||||
// with your ChatGPT account") stays buried in daemon logs.
|
||||
if finalStatus == "completed" && finalOutput == "" {
|
||||
if msg := providerErr.message(); msg != "" {
|
||||
finalStatus = "failed"
|
||||
finalError = msg
|
||||
}
|
||||
}
|
||||
|
||||
// Build usage map.
|
||||
c.usageMu.Lock()
|
||||
u := c.usage
|
||||
@@ -667,3 +691,98 @@ func hermesToolNameFromTitle(title string, kind string) string {
|
||||
return kind
|
||||
}
|
||||
}
|
||||
|
||||
// ── Provider-error sniffing ──
|
||||
//
|
||||
// hermes' session/prompt RPC reports stopReason=end_turn even when
|
||||
// the underlying HTTP call to the configured LLM endpoint returned
|
||||
// an error — the actionable detail only appears on stderr (e.g.
|
||||
// `⚠️ API call failed (attempt 1/3): BadRequestError [HTTP 400]` and
|
||||
// `Error: HTTP 400: Error code: 400 - {'detail': "The '...' model
|
||||
// is not supported when using Codex with a ChatGPT account."}`).
|
||||
// We scan for those patterns so the daemon can surface a real
|
||||
// failure instead of a generic "empty output".
|
||||
type hermesProviderErrorSniffer struct {
|
||||
mu sync.Mutex
|
||||
remains []byte // buffer for a partial trailing line across writes
|
||||
lines []string // captured error lines, bounded
|
||||
seen map[string]bool
|
||||
}
|
||||
|
||||
// hermesErrorHeaderRe matches the first line of an API-error block.
|
||||
// Hermes prefixes these with ⚠️ / ❌ and includes an HTTP status
|
||||
// code or a non-retryable-error tag.
|
||||
var hermesErrorHeaderRe = regexp.MustCompile(`(?:⚠️|❌|\[ERROR\]).*(?:BadRequestError|AuthenticationError|RateLimitError|HTTP [0-9]{3}|Non-retryable|API call failed)`)
|
||||
|
||||
// hermesErrorDetailRe pulls the most useful single-line messages
|
||||
// out of the subsequent lines of the error block (the one whose
|
||||
// "Error:" or "Details:" tag actually spells out what happened).
|
||||
var hermesErrorDetailRe = regexp.MustCompile(`(?:Error:|detail:|Details:)\s*(.+)`)
|
||||
|
||||
const hermesMaxErrorLines = 8
|
||||
|
||||
func newHermesProviderErrorSniffer() *hermesProviderErrorSniffer {
|
||||
return &hermesProviderErrorSniffer{seen: map[string]bool{}}
|
||||
}
|
||||
|
||||
// Write implements io.Writer so the sniffer can sit behind an
|
||||
// io.MultiWriter next to the normal stderr log forwarder.
|
||||
func (s *hermesProviderErrorSniffer) Write(p []byte) (int, error) {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
|
||||
data := append(s.remains, p...)
|
||||
// Keep the final partial line (no trailing newline) for the
|
||||
// next write so multi-line error blocks aren't split.
|
||||
nl := strings.LastIndexByte(string(data), '\n')
|
||||
var complete string
|
||||
if nl < 0 {
|
||||
s.remains = append(s.remains[:0], data...)
|
||||
return len(p), nil
|
||||
}
|
||||
complete = string(data[:nl])
|
||||
s.remains = append(s.remains[:0], data[nl+1:]...)
|
||||
|
||||
for _, line := range strings.Split(complete, "\n") {
|
||||
line = strings.TrimSpace(line)
|
||||
if line == "" {
|
||||
continue
|
||||
}
|
||||
if !(hermesErrorHeaderRe.MatchString(line) || hermesErrorDetailRe.MatchString(line)) {
|
||||
continue
|
||||
}
|
||||
if s.seen[line] {
|
||||
continue
|
||||
}
|
||||
s.seen[line] = true
|
||||
s.lines = append(s.lines, line)
|
||||
if len(s.lines) > hermesMaxErrorLines {
|
||||
s.lines = s.lines[len(s.lines)-hermesMaxErrorLines:]
|
||||
}
|
||||
}
|
||||
return len(p), nil
|
||||
}
|
||||
|
||||
// message returns a single-line summary suitable for the task
|
||||
// error field. Prefers the most specific "Error:" / "detail:"
|
||||
// fragment; falls back to the first captured header line; empty
|
||||
// when nothing useful was seen.
|
||||
func (s *hermesProviderErrorSniffer) message() string {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
|
||||
for _, line := range s.lines {
|
||||
if m := hermesErrorDetailRe.FindStringSubmatch(line); m != nil {
|
||||
detail := strings.TrimSpace(m[1])
|
||||
if detail != "" {
|
||||
return "hermes provider error: " + detail
|
||||
}
|
||||
}
|
||||
}
|
||||
for _, line := range s.lines {
|
||||
if hermesErrorHeaderRe.MatchString(line) {
|
||||
return "hermes provider error: " + line
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
@@ -2,6 +2,7 @@ package agent
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
@@ -375,3 +376,71 @@ func TestHermesClientIgnoresInvalidJSON(t *testing.T) {
|
||||
c.handleLine("")
|
||||
c.handleLine("{}")
|
||||
}
|
||||
|
||||
func TestHermesProviderErrorSniffer(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// Real sample of the stderr hermes emits when the configured
|
||||
// LLM endpoint rejects the requested model. We verify the
|
||||
// sniffer extracts the `Error: ...` line so the task error
|
||||
// tells the user *why* it failed.
|
||||
s := newHermesProviderErrorSniffer()
|
||||
lines := []string{
|
||||
"2026-04-20 23:41:47 [INFO] acp_adapter.server: Prompt on session abc",
|
||||
`⚠️ API call failed (attempt 1/3): BadRequestError [HTTP 400]`,
|
||||
` 🔌 Provider: openai-codex Model: gpt-5.1-codex-mini`,
|
||||
` 📝 Error: HTTP 400: Error code: 400 - {'detail': "The 'gpt-5.1-codex-mini' model is not supported when using Codex with a ChatGPT account."}`,
|
||||
`⏱️ Elapsed: 1.17s`,
|
||||
}
|
||||
for _, line := range lines {
|
||||
if _, err := s.Write([]byte(line + "\n")); err != nil {
|
||||
t.Fatalf("Write: %v", err)
|
||||
}
|
||||
}
|
||||
msg := s.message()
|
||||
if msg == "" {
|
||||
t.Fatal("expected a non-empty error message")
|
||||
}
|
||||
if !strings.Contains(msg, "model is not supported") {
|
||||
t.Errorf("expected detail about model support, got %q", msg)
|
||||
}
|
||||
}
|
||||
|
||||
func TestHermesProviderErrorSnifferIgnoresInfoLines(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
s := newHermesProviderErrorSniffer()
|
||||
s.Write([]byte("2026-04-20 23:41:45 [INFO] acp_adapter.entry: Loaded env\n"))
|
||||
s.Write([]byte("2026-04-20 23:41:47 [INFO] agent.auxiliary_client: Vision auto-detect...\n"))
|
||||
if msg := s.message(); msg != "" {
|
||||
t.Errorf("info lines should produce no error, got %q", msg)
|
||||
}
|
||||
}
|
||||
|
||||
func TestHermesProviderErrorSnifferHandlesPartialLines(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
// Writer may be called mid-line; the sniffer must buffer until
|
||||
// it sees a newline so the regex doesn't miss the header.
|
||||
s := newHermesProviderErrorSniffer()
|
||||
s.Write([]byte(`⚠️ API call failed (attempt 1/3):`))
|
||||
s.Write([]byte(` BadRequestError [HTTP 400]` + "\n"))
|
||||
s.Write([]byte(` 📝 Error: something went wrong` + "\n"))
|
||||
msg := s.message()
|
||||
if !strings.Contains(msg, "something went wrong") {
|
||||
t.Errorf("expected buffered line to be captured, got %q", msg)
|
||||
}
|
||||
}
|
||||
|
||||
func TestHermesProviderErrorSnifferBoundedBuffer(t *testing.T) {
|
||||
t.Parallel()
|
||||
|
||||
s := newHermesProviderErrorSniffer()
|
||||
for i := 0; i < 20; i++ {
|
||||
// Each line differs so dedup doesn't merge them.
|
||||
s.Write([]byte(`⚠️ API call failed (HTTP 400) attempt ` + string(rune('a'+i%26)) + `: Non-retryable error` + "\n"))
|
||||
}
|
||||
if len(s.lines) > hermesMaxErrorLines {
|
||||
t.Errorf("sniffer kept %d lines, limit is %d", len(s.lines), hermesMaxErrorLines)
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user