From 0dffc02a80f00b87b6722c0d2baa8ebf3e4484f7 Mon Sep 17 00:00:00 2001 From: Ganesh Kumar Date: Fri, 2 Oct 2026 21:48:48 +0530 Subject: [PATCH 01/11] Centralize LLM Error Handling and Fallback Messages --- docs/mcp_error_handling.md | 65 +++++++++++++++++++++ internal/mcpagent/agent.go | 77 +++++-------------------- internal/mcpagent/analytics.go | 15 ++--- internal/mcpagent/provider.go | 100 +++++++++++++++++++++++++++++++++ 4 files changed, 185 insertions(+), 72 deletions(-) create mode 100644 docs/mcp_error_handling.md diff --git a/docs/mcp_error_handling.md b/docs/mcp_error_handling.md new file mode 100644 index 00000000..5675eda5 --- /dev/null +++ b/docs/mcp_error_handling.md @@ -0,0 +1,65 @@ +# MCP LLM Error Handling + +When interacting with multiple AI Providers (Google, OpenAI, Anthropic) via `langchaingo`, LLM calls can fail in messy, unpredictable ways. Providers may return HTTP 503s for overload, 429s for rate limits, or 401s for revoked keys. + +If we return these raw errors directly to the MCP client (the chat UI), the turn crashes and the user sees ugly stack traces (e.g., `googleapi: Error 503`). + +To solve this, we use a **centralized UI fallback mechanism** that traps raw SDK errors and gracefully degrades the UI without disrupting the core ReAct loop. + +## The Architecture + +The error handling logic lives in `internal/mcpagent/provider.go` and consists of two functions: + +### 1. `CategorizeLLMError(err error) LLMErrorCategory` +This function intercepts an error, inspects it for `langchaingo` normalizations, unwraps leaked provider-specific errors (like `*googleapi.Error`), and falls back to string matching. + +It maps the messy error into one of four clean categories: +* `ErrCategoryAuth` (401, 403, 404, invalid keys, revoked models) +* `ErrCategoryOverload` (429, 503, rate limits, high demand) +* `ErrCategoryTimeout` (504, context deadline exceeded) +* `ErrCategoryUnknown` (Prompt parsing errors, network disconnects) + +### 2. `GetLLMFallbackMessage(cat LLMErrorCategory) string` +This function takes a category and returns a beautifully formatted, non-emoji Markdown string to show the user. +* If it returns a non-empty string, **the turn is considered saved**. We show the user this message. +* If it returns `""` (for `ErrCategoryUnknown`), the error is fatal and should be bubbled up. + +## How to use it + +We deliberately **do not** wrap the original `a.provider.Complete(...)` function. We keep the core LLM signatures untouched. Instead, you just "pick up" the error immediately after the LLM call fails. + +### Example (from `agent.go`) + +```go +response, usage, err := a.provider.Complete(ctx, history, tools, completeOpts...) +if err != nil { + // 1. Pick up the error and categorize it + errCategory := CategorizeLLMError(err) + + // 2. Ask for a clean UI message + msg := GetLLMFallbackMessage(errCategory) + + // 3. If there is no clean fallback, bubble up the fatal error + if msg == "" { + return "", history, nil, nil, fmt.Errorf("llm completion failed: %w", err) + } + + // 4. Otherwise, inject the fallback message as an AI response and return a nil error! + // This gracefully completes the turn and renders the message in the UI. + history = append(history, HistoryEntry{ + "role": "assistant", + "content": msg, + "text": msg, + }) + return msg, history, nil, nil, nil +} +``` + +## Adding new Fallback Scenarios + +If you need to handle a new failure mode (e.g. `ContextTooLarge`): +1. Add `ErrCategoryContextSize` to the const block in `provider.go`. +2. Add the detection logic to `CategorizeLLMError`. +3. Add the exact Markdown message you want the user to see to the `switch` block in `GetLLMFallbackMessage`. + +This guarantees every MCP LLM call across the app degrades consistently. diff --git a/internal/mcpagent/agent.go b/internal/mcpagent/agent.go index c55b7461..9a07b079 100644 --- a/internal/mcpagent/agent.go +++ b/internal/mcpagent/agent.go @@ -3,7 +3,6 @@ package mcpagent import ( "context" "encoding/json" - "errors" "fmt" "strings" "time" @@ -13,7 +12,6 @@ import ( "github.com/livereview/internal/vlrender" "github.com/rs/zerolog/log" "github.com/tmc/langchaingo/llms" - "google.golang.org/api/googleapi" ) const ( @@ -295,27 +293,27 @@ func (a *Agent) runStepLoop( if jsonMode { completeOpts = append(completeOpts, llms.WithJSONMode()) } + response, usage, err := a.provider.Complete(ctx, history, tools, completeOpts...) aiElapsed := time.Since(aiStart) if err != nil { - log.Error().Err(err).Int("step", step).Msg("LLM completion failed") + errCategory := CategorizeLLMError(err) + log.Error().Err(err).Int("step", step).Str("category", string(errCategory)).Msg("LLM completion failed") clog.AIError(callNumber, step, aiElapsed, err) - if isProviderAuthOrModelError(err) { - msg := fmt.Sprintf("> ⚠️ **Action Required: AI Provider Issue**\n> \n> The existing **%s**'s key seems expired/revoked or the model is no longer available.\n\nPlease configure a valid provider to continue: [btn:Configure AI Provider](/settings#ai)", a.provider.Describe()) - clog.FinalResponse(msg + " (stopped after provider auth/model error)") - - history = append(history, HistoryEntry{ - "role": "assistant", - "content": msg, - "text": msg, - "suggested_questions": DefaultAIErrorSuggestedQuestions, - }) - - return msg, history, nil, nil, nil + msg := GetLLMFallbackMessage(errCategory) + if msg == "" { + return "", history, nil, nil, fmt.Errorf("llm completion step %d: %w", step, err) } - return "", history, nil, nil, fmt.Errorf("llm completion step %d: %w", step, err) + clog.FinalResponse(msg + " (stopped after " + string(errCategory) + ")") + history = append(history, HistoryEntry{ + "role": "assistant", + "content": msg, + "text": msg, + "suggested_questions": DefaultAIErrorSuggestedQuestions, + }) + return msg, history, nil, nil, nil } log.Debug().Int("step", step).Int("response_len", len(response)).Msg("LLM call succeeded") clog.AIResponse(callNumber, step, aiElapsed, usage.InputTokens, usage.OutputTokens, response) @@ -599,56 +597,9 @@ func (a *Agent) interpretUserMessage(clog *logging.ChatTurnLogger, userText stri return msg, nil } -// isProviderAuthOrModelError reports whether an LLM completion error is caused -// by a revoked/expired AI provider key or a deprecated/unavailable model. -func isProviderAuthOrModelError(err error) bool { - if err == nil { - return false - } - - // 1. Check langchaingo standardized errors - var llmsErr *llms.Error - if errors.As(err, &llmsErr) { - if llmsErr.Code == llms.ErrCodeAuthentication || llmsErr.Code == llms.ErrCodeResourceNotFound { - return true - } - } - - // 2. Check Google API typed errors - var gErr *googleapi.Error - if errors.As(err, &gErr) { - if gErr.Code == 401 || gErr.Code == 403 || gErr.Code == 404 { - return true - } - } - // 3. Check generic interface methods for HTTP status codes (used by OpenAI client wrapper, etc.) - var scErr interface{ StatusCode() int } - if errors.As(err, &scErr) { - code := scErr.StatusCode() - if code == 401 || code == 403 || code == 404 { - return true - } - } - var hscErr interface{ HTTPStatusCode() int } - if errors.As(err, &hscErr) { - code := hscErr.HTTPStatusCode() - if code == 401 || code == 403 || code == 404 { - return true - } - } - // 4. Fallback for non-HTTP API responses indicating model unavailability - msg := strings.ToLower(err.Error()) - return strings.Contains(msg, "deprecated") || - strings.Contains(msg, "model not found") || - strings.Contains(msg, "status code: 401") || - strings.Contains(msg, "status code: 403") || - strings.Contains(msg, "status code: 404") || - strings.Contains(msg, "unauthorized") || - strings.Contains(msg, "forbidden") -} // isAuthError reports whether a tool result signals an expired/invalid // session token or missing auth - the exact messages LiveReview's own auth diff --git a/internal/mcpagent/analytics.go b/internal/mcpagent/analytics.go index ee5bb439..b4d216a3 100644 --- a/internal/mcpagent/analytics.go +++ b/internal/mcpagent/analytics.go @@ -42,7 +42,7 @@ const ( // analyticsTurnTimeout bounds the whole fan-out regardless of per-query // timeouts, so a slow model cannot hold a request open indefinitely. - analyticsTurnTimeout = 90 * time.Second + analyticsTurnTimeout = 300 * time.Second ) // AnalyticsEngine executes guard-rewritten SQL. Declared here rather than @@ -1651,16 +1651,13 @@ func (a *Agent) runMultiInterpret( raw, err := a.completeOnce(ctx, clog, 2, "interpret", "multi", 1, system, userMsg) if err != nil { log.Error().Err(err).Msg("multi-interpret LLM call failed") - if isProviderAuthOrModelError(err) { - msg := fmt.Sprintf("> ⚠️ **Action Required: AI Provider Issue**\n> \n> The existing **%s**'s key seems expired/revoked or the model is no longer available.\n\nPlease configure a valid provider to continue: [btn:Configure AI Provider](/settings#ai)", a.provider.Describe()) - clog.FinalResponse(msg + " (stopped after provider auth/model error)") + cat := CategorizeLLMError(err) + msg := GetLLMFallbackMessage(cat) + if msg != "" { + clog.FinalResponse(msg + " (stopped after " + string(cat) + ")") return msg, nil, debug, nil } - if errors.Is(err, context.DeadlineExceeded) || errors.Is(err, context.Canceled) { - // Intentionally return nil error here so the agent caller treats this turn as a success - // and persists the friendly timeout message to the database, rather than throwing it away. - return "The analysis took longer than expected. Please try asking a narrower or more specific question.", nil, debug, nil //nolint:nilerr - } + return "I had trouble understanding that question. Please try rephrasing.", nil, debug, nil //nolint:nilerr } debug.LLMRawResponse = raw diff --git a/internal/mcpagent/provider.go b/internal/mcpagent/provider.go index 47263577..8323c420 100644 --- a/internal/mcpagent/provider.go +++ b/internal/mcpagent/provider.go @@ -3,11 +3,14 @@ package mcpagent import ( "context" "encoding/json" + "errors" "fmt" + "strings" "github.com/livereview/internal/aiconnectors" "github.com/rs/zerolog/log" "github.com/tmc/langchaingo/llms" + "google.golang.org/api/googleapi" ) type Provider struct { @@ -186,3 +189,100 @@ func (p *Provider) historyToMessages(history []HistoryEntry) []llms.MessageConte } return messages } + +type LLMErrorCategory string + +const ( + ErrCategoryAuth LLMErrorCategory = "auth_failed" + ErrCategoryOverload LLMErrorCategory = "overloaded" + ErrCategoryTimeout LLMErrorCategory = "timeout" + ErrCategoryUnknown LLMErrorCategory = "unknown" +) + +// CategorizeLLMError maps raw SDK errors from Langchain or underlying clients +// into our standard failure categories for consistent UI fallback rendering. +func CategorizeLLMError(err error) LLMErrorCategory { + if err == nil { + return ErrCategoryUnknown + } + + // 1. Langchain-Go normalized errors + var llmsErr *llms.Error + if errors.As(err, &llmsErr) { + if llmsErr.Code == llms.ErrCodeAuthentication || llmsErr.Code == llms.ErrCodeResourceNotFound { + return ErrCategoryAuth + } + if llmsErr.Code == llms.ErrCodeRateLimit { + return ErrCategoryOverload + } + } + + // 2. Raw Google API errors (often leaked by langchaingo) + var gErr *googleapi.Error + if errors.As(err, &gErr) { + if gErr.Code == 401 || gErr.Code == 403 || gErr.Code == 404 { + return ErrCategoryAuth + } + if gErr.Code == 429 || gErr.Code == 500 || gErr.Code == 502 || gErr.Code == 503 || gErr.Code == 504 { + return ErrCategoryOverload + } + } + + // 3. Generic HTTP status codes + var scErr interface{ StatusCode() int } + if errors.As(err, &scErr) { + code := scErr.StatusCode() + if code == 401 || code == 403 || code == 404 { + return ErrCategoryAuth + } + if code == 429 || code == 500 || code == 502 || code == 503 || code == 504 { + return ErrCategoryOverload + } + } + + var hscErr interface{ HTTPStatusCode() int } + if errors.As(err, &hscErr) { + code := hscErr.HTTPStatusCode() + if code == 401 || code == 403 || code == 404 { + return ErrCategoryAuth + } + if code == 429 || code == 500 || code == 502 || code == 503 || code == 504 { + return ErrCategoryOverload + } + } + + // 4. Fallback string matching + msg := strings.ToLower(err.Error()) + if strings.Contains(msg, "context deadline") || strings.Contains(msg, "timeout") { + return ErrCategoryTimeout + } + if strings.Contains(msg, "rate limit") || strings.Contains(msg, "too many requests") || + strings.Contains(msg, "status code: 429") || strings.Contains(msg, "status code: 50") || + strings.Contains(msg, "service unavailable") || strings.Contains(msg, "bad gateway") || + strings.Contains(msg, "high demand") { + return ErrCategoryOverload + } + if strings.Contains(msg, "deprecated") || strings.Contains(msg, "model not found") || + strings.Contains(msg, "status code: 401") || strings.Contains(msg, "status code: 403") || + strings.Contains(msg, "status code: 404") || strings.Contains(msg, "unauthorized") || + strings.Contains(msg, "forbidden") { + return ErrCategoryAuth + } + + return ErrCategoryUnknown +} + +// GetLLMFallbackMessage returns a safe, beautiful UI response for the client +// based on the categorized LLM error. Returns empty string for unknown errors. +func GetLLMFallbackMessage(cat LLMErrorCategory) string { + switch cat { + case ErrCategoryAuth: + return "> **Action Required: AI Provider Issue**\n> \n> The AI Provider's API key is invalid or the model is missing. Please configure a valid provider in settings to continue.\n> \n> [btn:Configure AI Provider](/settings#ai)" + case ErrCategoryOverload: + return "> **AI Provider is Busy**\n> \n> The AI provider is currently experiencing high traffic. If you continue to see this issue, we suggest changing to a different model or provider." + case ErrCategoryTimeout: + return "> **Analysis Took Too Long**\n> \n> The AI took too long to generate your response and timed out. Try asking a narrower or more specific question." + default: + return "" + } +} From 014928d93c4ba11dab4101ef52e5ab416f25908e Mon Sep 17 00:00:00 2001 From: Ganesh Kumar Date: Sat, 3 Oct 2026 14:21:09 +0530 Subject: [PATCH 02/11] Add Centralized LLM Error Categorization and Action Cards --- internal/aiconnectors/errors.go | 87 +++++++++++++ internal/aiconnectors/errors_test.go | 101 +++++++++++++++ internal/api/chat_conversations_handler.go | 44 ++++--- internal/api/webchat_handler.go | 64 ++++++++-- internal/docindex/docs/.synced-commits.env | 4 +- internal/mcpagent/agent.go | 85 +++++++++++-- internal/mcpagent/analytics.go | 36 +++++- internal/mcpagent/provider.go | 141 +++++++-------------- internal/mcpagent/provider_test.go | 53 ++++++++ internal/mcpagent/types.go | 9 ++ ui/src/api/chatConversations.ts | 6 + ui/src/api/chatbot.ts | 7 + ui/src/pages/Chatbot/ChatConversation.tsx | 64 +++++----- 13 files changed, 524 insertions(+), 177 deletions(-) create mode 100644 internal/aiconnectors/errors.go create mode 100644 internal/aiconnectors/errors_test.go create mode 100644 internal/mcpagent/provider_test.go diff --git a/internal/aiconnectors/errors.go b/internal/aiconnectors/errors.go new file mode 100644 index 00000000..7c15883e --- /dev/null +++ b/internal/aiconnectors/errors.go @@ -0,0 +1,87 @@ +package aiconnectors + +import ( + "errors" + "strings" + + "github.com/tmc/langchaingo/llms" +) + +type LLMErrorCategory string + +const ( + ErrCategoryAuth LLMErrorCategory = "auth_failed" + ErrCategoryOverload LLMErrorCategory = "overloaded" + ErrCategoryTimeout LLMErrorCategory = "timeout" + ErrCategoryDeprecated LLMErrorCategory = "model_deprecated" + ErrCategoryUnknown LLMErrorCategory = "unknown" +) + +// CategorizeLLMError maps raw SDK errors from Langchain or underlying clients +// into our standard failure categories for consistent UI fallback rendering. +func CategorizeLLMError(err error) LLMErrorCategory { + if err == nil { + return ErrCategoryUnknown + } + + // 1. Langchain-Go normalized errors + var llmsErr *llms.Error + if errors.As(err, &llmsErr) { + if llmsErr.Code == llms.ErrCodeAuthentication || llmsErr.Code == llms.ErrCodeResourceNotFound { + return ErrCategoryAuth + } + if llmsErr.Code == llms.ErrCodeRateLimit { + return ErrCategoryOverload + } + } + + // 2. Generic HTTP status codes + var scErr interface{ StatusCode() int } + if errors.As(err, &scErr) { + code := scErr.StatusCode() + if code == 401 || code == 403 || code == 404 { + return ErrCategoryAuth + } + if code == 429 || code == 500 || code == 502 || code == 503 || code == 504 { + return ErrCategoryOverload + } + } + + var hscErr interface{ HTTPStatusCode() int } + if errors.As(err, &hscErr) { + code := hscErr.HTTPStatusCode() + if code == 401 || code == 403 || code == 404 { + return ErrCategoryAuth + } + if code == 429 || code == 500 || code == 502 || code == 503 || code == 504 { + return ErrCategoryOverload + } + } + + // 3. Fallback string matching + msg := strings.ToLower(err.Error()) + if strings.Contains(msg, "context deadline") || strings.Contains(msg, "timeout") { + return ErrCategoryTimeout + } + if strings.Contains(msg, "rate limit") || strings.Contains(msg, "too many requests") || + strings.Contains(msg, "status code: 429") || strings.Contains(msg, "status code: 50") || + strings.Contains(msg, "error 429") || strings.Contains(msg, "error 50") || + strings.Contains(msg, "service unavailable") || strings.Contains(msg, "bad gateway") || + strings.Contains(msg, "high demand") { + return ErrCategoryOverload + } + if strings.Contains(msg, "deprecated") || strings.Contains(msg, "model not found") || + strings.Contains(msg, "model_not_found") || strings.Contains(msg, "no such model") { + return ErrCategoryDeprecated + } + if strings.Contains(msg, "status code: 401") || strings.Contains(msg, "status code: 403") || + strings.Contains(msg, "status code: 404") || strings.Contains(msg, "unauthorized") || + strings.Contains(msg, "error 401") || strings.Contains(msg, "error 403") || + strings.Contains(msg, "error 404") || strings.Contains(msg, "forbidden") { + return ErrCategoryAuth + } + + return ErrCategoryUnknown +} + + diff --git a/internal/aiconnectors/errors_test.go b/internal/aiconnectors/errors_test.go new file mode 100644 index 00000000..849b2dc2 --- /dev/null +++ b/internal/aiconnectors/errors_test.go @@ -0,0 +1,101 @@ +package aiconnectors + +import ( + "errors" + "net/http" + "testing" + + "github.com/tmc/langchaingo/llms" +) + +// mockHTTPError mocks an error that provides a StatusCode() method. +type mockHTTPError struct { + code int + msg string +} + +func (e *mockHTTPError) Error() string { + return e.msg +} + +func (e *mockHTTPError) StatusCode() int { + return e.code +} + +func TestCategorizeLLMError(t *testing.T) { + tests := []struct { + name string + err error + want LLMErrorCategory + }{ + { + name: "Nil error", + err: nil, + want: ErrCategoryUnknown, + }, + { + name: "Langchain Auth Error", + err: &llms.Error{Code: llms.ErrCodeAuthentication}, + want: ErrCategoryAuth, + }, + { + name: "Langchain Rate Limit Error", + err: &llms.Error{Code: llms.ErrCodeRateLimit}, + want: ErrCategoryOverload, + }, + { + name: "HTTP 401 via StatusCode()", + err: &mockHTTPError{code: http.StatusUnauthorized, msg: "unauthorized"}, + want: ErrCategoryAuth, + }, + { + name: "HTTP 503 via StatusCode()", + err: &mockHTTPError{code: http.StatusServiceUnavailable, msg: "service unavailable"}, + want: ErrCategoryOverload, + }, + { + name: "String Match: context deadline", + err: errors.New("operation failed: context deadline exceeded"), + want: ErrCategoryTimeout, + }, + { + name: "String Match: high demand", + err: errors.New("googleapi: Error 503: This model is currently experiencing high demand."), + want: ErrCategoryOverload, + }, + { + name: "String Match: error 429", + err: errors.New("provider returned error 429 too many requests"), + want: ErrCategoryOverload, + }, + { + name: "String Match: model not found", + err: errors.New("model not found or deprecated"), + want: ErrCategoryAuth, + }, + { + name: "String Match: error 404", + err: errors.New("error 404: resource not found"), + want: ErrCategoryAuth, + }, + { + name: "String Match: error 502", + err: errors.New("googleapi: error 502: bad gateway"), + want: ErrCategoryOverload, + }, + { + name: "Unknown Error", + err: errors.New("something went completely wrong"), + want: ErrCategoryUnknown, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + got := CategorizeLLMError(tt.err) + if got != tt.want { + t.Errorf("CategorizeLLMError() = %v, want %v", got, tt.want) + } + }) + } +} diff --git a/internal/api/chat_conversations_handler.go b/internal/api/chat_conversations_handler.go index 70fc99d7..6814aee9 100644 --- a/internal/api/chat_conversations_handler.go +++ b/internal/api/chat_conversations_handler.go @@ -27,13 +27,14 @@ type ChatConversationSummary struct { // ChatMessageOut is one persisted message, with any charts and file exports // it produced. type ChatMessageOut struct { - ID int64 `json:"id"` - Role string `json:"role"` - Content string `json:"content"` - Charts []WebChatChart `json:"charts,omitempty"` - Files []WebChatFile `json:"files,omitempty"` + ID int64 `json:"id"` + Role string `json:"role"` + Content string `json:"content"` + Charts []WebChatChart `json:"charts,omitempty"` + Files []WebChatFile `json:"files,omitempty"` SuggestedQuestions []mcpagent.SuggestedQuestionCategory `json:"suggested_questions,omitempty"` - DebugArtifacts json.RawMessage `json:"debug_artifacts,omitempty"` + DebugArtifacts json.RawMessage `json:"debug_artifacts,omitempty"` + ActionCard *mcpagent.ActionCard `json:"action_card,omitempty"` } // ChatConversationDetail is a full conversation with its message history, for @@ -201,19 +202,29 @@ func (s *Server) GetConversation(c echo.Context) error { }) } var suggestedQuestions []mcpagent.SuggestedQuestionCategory + var actionCard *mcpagent.ActionCard for _, entry := range m.RawHistoryEntries { - if sq, ok := entry["suggested_questions"].([]mcpagent.SuggestedQuestionCategory); ok && len(sq) > 0 { - suggestedQuestions = sq - break - } else if rawSq, ok := entry["suggested_questions"]; ok && rawSq != nil { - if b, err := json.Marshal(rawSq); err == nil { - if err := json.Unmarshal(b, &suggestedQuestions); err != nil { - log.Warn().Err(err).Msg("failed to unmarshal raw suggested_questions") - } else if len(suggestedQuestions) > 0 { - break - } + // After JSONB round-trip from Postgres, all values come back as + // generic types ([]interface{}, map[string]interface{}, etc.), + // not as typed Go structs. Re-marshal each entry and read the + // fields we care about from a plain map so type assertions are safe. + entryBytes, err := json.Marshal(entry) + if err != nil { + continue + } + var plain map[string]json.RawMessage + if err := json.Unmarshal(entryBytes, &plain); err != nil { + continue + } + if ac, ok := plain["action_card"]; ok && actionCard == nil { + var card mcpagent.ActionCard + if err := json.Unmarshal(ac, &card); err == nil { + actionCard = &card } } + if sq, ok := plain["suggested_questions"]; ok && len(suggestedQuestions) == 0 { + _ = json.Unmarshal(sq, &suggestedQuestions) + } } out.Messages = append(out.Messages, ChatMessageOut{ ID: m.ID, @@ -223,6 +234,7 @@ func (s *Server) GetConversation(c echo.Context) error { Files: files, SuggestedQuestions: suggestedQuestions, DebugArtifacts: m.DebugArtifacts, + ActionCard: actionCard, }) } return c.JSON(http.StatusOK, out) diff --git a/internal/api/webchat_handler.go b/internal/api/webchat_handler.go index 6c873d2d..109c0dbf 100644 --- a/internal/api/webchat_handler.go +++ b/internal/api/webchat_handler.go @@ -52,13 +52,16 @@ type WebChatChart struct { } type WebChatResponse struct { - Response string `json:"response"` - Charts []WebChatChart `json:"charts,omitempty"` - Files []WebChatFile `json:"files,omitempty"` + Response string `json:"response"` + Charts []WebChatChart `json:"charts,omitempty"` + Files []WebChatFile `json:"files,omitempty"` SuggestedQuestions []mcpagent.SuggestedQuestionCategory `json:"suggested_questions,omitempty"` - DebugArtifacts json.RawMessage `json:"debug_artifacts,omitempty"` - SessionID string `json:"sessionId,omitempty"` - ConversationID int64 `json:"conversationId"` + DebugArtifacts json.RawMessage `json:"debug_artifacts,omitempty"` + SessionID string `json:"sessionId,omitempty"` + ConversationID int64 `json:"conversationId"` + // ActionCard provides structured data for the frontend to render an action card + // (e.g. "Configuration Required") without hardcoding string matching in the UI. + ActionCard *mcpagent.ActionCard `json:"action_card,omitempty"` } // analyticsRoleFor maps permissions onto the SQL catalog's roles. Super admins @@ -197,9 +200,14 @@ func (s *Server) HandleWebChat(c echo.Context) error { mcpSession, err := mcpagent.ConnectMCP(ctx, mcpURL, mcpHeaders) if err != nil { log.Error().Err(err).Str("url", mcpURL).Msg("WebChat: failed to connect to MCP server") - errText := fmt.Sprintf("Internal AI Service Error: Unable to connect to MCP tools (%s).", err.Error()) - persistReply(errText, nil, nil, nil, "", nil) - return c.JSON(http.StatusServiceUnavailable, map[string]string{"error": errText}) + userFriendlyErr := "> **AI Tools Unavailable**\n> \n> I am having trouble connecting to my internal data tools right now. Please try your request again in a few moments." + persistReply(userFriendlyErr, nil, nil, nil, "", nil) + resp := WebChatResponse{ + Response: userFriendlyErr, + SessionID: sessionID, + ConversationID: convID, + } + return c.JSON(http.StatusOK, resp) } if pc.CurrentOrg != nil && pc.CurrentOrg.Name != "" { @@ -221,9 +229,14 @@ func (s *Server) HandleWebChat(c echo.Context) error { responseText, updatedHistory, artifacts, debugArt, err := agent.RunTurnWithArtifacts(ctx, history, req.Message, sessionID, "livi") if err != nil { log.Error().Err(err).Msg("WebChat: agent loop failed") - errText := fmt.Sprintf("Agent loop failed: %s", err.Error()) - persistReply(errText, nil, nil, nil, "", nil) - return c.JSON(http.StatusInternalServerError, map[string]string{"error": errText}) + userFriendlyErr := "> **AI Processing Error**\n> \n> I encountered an unexpected error while processing your request. Please try asking again." + persistReply(userFriendlyErr, nil, nil, nil, "", nil) + resp := WebChatResponse{ + Response: userFriendlyErr, + SessionID: sessionID, + ConversationID: convID, + } + return c.JSON(http.StatusOK, resp) } // Whatever RunTurnWithArtifacts appended on top of the history we fed it // (possibly including a freshly swapped-in system prompt entry - see @@ -240,6 +253,21 @@ func (s *Server) HandleWebChat(c echo.Context) error { } for _, entry := range turnEntries { + if rawCard, ok := entry["action_card"]; ok && rawCard != nil && resp.ActionCard == nil { + log.Info().Interface("rawCard", rawCard).Msg("WebChat: found action_card in turnEntries") + if b, err := json.Marshal(rawCard); err == nil { + var ac mcpagent.ActionCard + if err := json.Unmarshal(b, &ac); err == nil { + acCopy := ac + resp.ActionCard = &acCopy + log.Info().Interface("actionCard", acCopy).Msg("WebChat: successfully unmarshaled action_card") + } else { + log.Error().Err(err).Msg("WebChat: failed to unmarshal action_card") + } + } else { + log.Error().Err(err).Msg("WebChat: failed to marshal rawCard") + } + } if sq, ok := entry["suggested_questions"].([]mcpagent.SuggestedQuestionCategory); ok && len(sq) > 0 { resp.SuggestedQuestions = sq break @@ -253,6 +281,10 @@ func (s *Server) HandleWebChat(c echo.Context) error { } } } + // If we have an action_card but no suggested questions were attached, add defaults. + if resp.ActionCard != nil && len(resp.SuggestedQuestions) == 0 { + resp.SuggestedQuestions = mcpagent.DefaultAIErrorSuggestedQuestions + } if vlrender.HasVegaLiteSpec(responseText) { charts, cleanText, err := extractChartsFromVega(responseText) @@ -324,6 +356,7 @@ func (s *Server) HandleWebChat(c echo.Context) error { resp.Files = append(resp.Files, chatFileFromArtifact(art, fileIDs[i])) } + log.Info().Interface("finalResp", resp).Msg("WebChat: returning response to client") return c.JSON(http.StatusOK, resp) } @@ -356,13 +389,18 @@ func titleFromFirstMessage(message string) string { // (agent.go:148-166), so replaying persisted history without it is safe, and // simpler than tracking which prompt variant was live on a given turn. func persistAssistantMessage(ctx context.Context, chatStore *storagechat.Store, convID int64, userText, assistantText string, turnEntries []mcpagent.HistoryEntry, charts []WebChatChart, artifacts []mcpagent.Artifact, rawLLMOutput string, debugArtifacts json.RawMessage) ([]int64, error) { - assistantEntries := []mcpagent.HistoryEntry{{"role": "assistant", "content": assistantText}} + assistantEntries := turnEntries + hasUser := false for i, e := range turnEntries { if role, _ := e["role"].(string); role == "user" { assistantEntries = turnEntries[i+1:] + hasUser = true break } } + if !hasUser && len(assistantEntries) == 0 { + assistantEntries = []mcpagent.HistoryEntry{{"role": "assistant", "content": assistantText}} + } chartInputs := make([]storagechat.ChartInput, 0, len(charts)) for _, ch := range charts { diff --git a/internal/docindex/docs/.synced-commits.env b/internal/docindex/docs/.synced-commits.env index 94a33528..9dbe34ba 100644 --- a/internal/docindex/docs/.synced-commits.env +++ b/internal/docindex/docs/.synced-commits.env @@ -1,4 +1,4 @@ -GIT_LRC_COMMIT=33e6d7f97a14b906492694ce94961c3744e84700 +GIT_LRC_COMMIT=be23131559d27f870c9a2a1de5a4162756d83f0d GIT_LRC_WIKI_COMMIT=a1bb17c392a46c1f3bf6cfad6396813ccf80a1a9 -HEXMOSHOMEPAGE_DOCS_COMMIT=cd5d33575e10ebd6c7d128d49e8807cb04f2a2d9 +HEXMOSHOMEPAGE_DOCS_COMMIT=7af310a4a53ad928ba813b0488a5a4f96f4ef546 LIVEREVIEW_WIKI_COMMIT=58ee4a1060bf109d889dd2cc969a776fb337d2f7 diff --git a/internal/mcpagent/agent.go b/internal/mcpagent/agent.go index 9a07b079..bf43a677 100644 --- a/internal/mcpagent/agent.go +++ b/internal/mcpagent/agent.go @@ -7,6 +7,7 @@ import ( "strings" "time" + "github.com/livereview/internal/aiconnectors" "github.com/livereview/internal/docindex" "github.com/livereview/internal/logging" "github.com/livereview/internal/vlrender" @@ -151,6 +152,8 @@ func (a *Agent) RunTurnWithArtifacts(ctx context.Context, history []HistoryEntry } log.Warn().Str("failure_reason", schemaIndexFailureReason()).Msg("schema index not ready; refusing the turn before any LLM call") clog.FinalResponse(msg) + history = append(history, HistoryEntry{"role": "user", "content": userText}) + history = append(history, HistoryEntry{"role": "assistant", "content": msg, "text": msg}) return msg, history, nil, nil, nil } @@ -161,7 +164,36 @@ func (a *Agent) RunTurnWithArtifacts(ctx context.Context, history []HistoryEntry res, err := a.classify(ctx, history, userText, clog) shape := res.Shape if err != nil { - log.Warn().Err(err).Msg("call #0 classify failed; degrading to product_guidance for this turn") + errCategory := aiconnectors.CategorizeLLMError(err) + msg := "" + var card *ActionCard + if tpl, ok := LLMErrorTemplates[errCategory]; ok { + msg = tpl.Message + card = &tpl.ActionCard + } + if msg != "" { + log.Warn().Err(err).Str("category", string(errCategory)).Msg("call #0 classify failed with provider error; short-circuiting turn") + clog.FinalResponse(msg + " (stopped after " + string(errCategory) + ")") + history = append(history, HistoryEntry{"role": "user", "content": userText}) + if card != nil { + history = append(history, HistoryEntry{ + "role": "assistant", + "content": msg, + "text": msg, + "suggested_questions": DefaultAIErrorSuggestedQuestions, + "action_card": card, + }) + } else { + history = append(history, HistoryEntry{ + "role": "assistant", + "content": msg, + "text": msg, + "suggested_questions": DefaultAIErrorSuggestedQuestions, + }) + } + return msg, history, nil, nil, nil + } + log.Warn().Err(err).Msg("call #0 classify failed with unparseable or unknown error; degrading to product_guidance for this turn") shape = shapeProductGuidance } @@ -175,8 +207,24 @@ func (a *Agent) RunTurnWithArtifacts(ctx context.Context, history []HistoryEntry return text, history, artifacts, debugArt, err } entry := HistoryEntry{"role": "assistant", "content": text, "text": text} - if text == NoDataAnalyticsResponseText { + switch { + case text == NoDataAnalyticsResponseText: entry["suggested_questions"] = DefaultNoDataSuggestedQuestions + case strings.Contains(text, "Action Required: AI Provider Issue"): + entry["suggested_questions"] = DefaultAIErrorSuggestedQuestions + if tpl, ok := LLMErrorTemplates[aiconnectors.ErrCategoryAuth]; ok { + entry["action_card"] = tpl.ActionCard + } + case strings.Contains(text, "AI Provider is Busy"): + entry["suggested_questions"] = DefaultAIErrorSuggestedQuestions + if tpl, ok := LLMErrorTemplates[aiconnectors.ErrCategoryOverload]; ok { + entry["action_card"] = tpl.ActionCard + } + case strings.Contains(text, "Model No Longer Available"): + entry["suggested_questions"] = DefaultAIErrorSuggestedQuestions + if tpl, ok := LLMErrorTemplates[aiconnectors.ErrCategoryDeprecated]; ok { + entry["action_card"] = tpl.ActionCard + } } history = append(history, entry) return text, history, artifacts, debugArt, nil @@ -297,22 +345,37 @@ func (a *Agent) runStepLoop( response, usage, err := a.provider.Complete(ctx, history, tools, completeOpts...) aiElapsed := time.Since(aiStart) if err != nil { - errCategory := CategorizeLLMError(err) + errCategory := aiconnectors.CategorizeLLMError(err) log.Error().Err(err).Int("step", step).Str("category", string(errCategory)).Msg("LLM completion failed") clog.AIError(callNumber, step, aiElapsed, err) - msg := GetLLMFallbackMessage(errCategory) + msg := "" + var card *ActionCard + if tpl, ok := LLMErrorTemplates[errCategory]; ok { + msg = tpl.Message + card = &tpl.ActionCard + } if msg == "" { return "", history, nil, nil, fmt.Errorf("llm completion step %d: %w", step, err) } clog.FinalResponse(msg + " (stopped after " + string(errCategory) + ")") - history = append(history, HistoryEntry{ - "role": "assistant", - "content": msg, - "text": msg, - "suggested_questions": DefaultAIErrorSuggestedQuestions, - }) + if card != nil { + history = append(history, HistoryEntry{ + "role": "assistant", + "content": msg, + "text": msg, + "suggested_questions": DefaultAIErrorSuggestedQuestions, + "action_card": card, + }) + } else { + history = append(history, HistoryEntry{ + "role": "assistant", + "content": msg, + "text": msg, + "suggested_questions": DefaultAIErrorSuggestedQuestions, + }) + } return msg, history, nil, nil, nil } log.Debug().Int("step", step).Int("response_len", len(response)).Msg("LLM call succeeded") @@ -591,7 +654,7 @@ func (a *Agent) interpretUserMessage(clog *logging.ChatTurnLogger, userText stri return "", err } msg := fmt.Sprintf( - "User query: %s\nOrg context: org_id = %d (%s)\n\n--- dbctx schema context ---\n%s\n\n--- available chart types ---\n%s", + "User query: %s\nOrg context: org_id = %d (%s)\n\n--- dbctx schema context ---\n%s\n\n--- available chart types ---\n%s\n\nNow, generate the JSON object containing the interpretations as requested. Do NOT output any markdown, reasoning, or text outside of the JSON object.", userText, orgID, orgName, tableText, a.chartTypes, ) return msg, nil diff --git a/internal/mcpagent/analytics.go b/internal/mcpagent/analytics.go index b4d216a3..22cf0e5c 100644 --- a/internal/mcpagent/analytics.go +++ b/internal/mcpagent/analytics.go @@ -10,6 +10,7 @@ import ( "strings" "time" + "github.com/livereview/internal/aiconnectors" "github.com/livereview/internal/chatstats" "github.com/livereview/internal/livisql" "github.com/livereview/internal/logging" @@ -1569,12 +1570,27 @@ type interpretEnvelope struct { func parseInterpretations(text string) ([]Interpretation, bool) { body := strings.TrimSpace(vlrender.ExtractJSONBlock(text)) if body == "" { + log.Warn().Msg("parseInterpretations: ExtractJSONBlock returned empty string") return nil, false } var env interpretEnvelope - if err := json.Unmarshal([]byte(body), &env); err == nil && len(env.Interpretations) > 0 { - return normalizeInterpretations(env.Interpretations) + if err := json.Unmarshal([]byte(body), &env); err == nil { + log.Debug().Int("interpretations_count", len(env.Interpretations)).Msg("parseInterpretations: envelope parsed") + if len(env.Interpretations) > 0 { + // Log first entry for debugging. + first := env.Interpretations[0] + log.Debug(). + Str("name", first.Name). + Str("chart_type", first.ChartType). + Int("sql_len", len(strings.TrimSpace(first.SQL))). + Msg("parseInterpretations: first interpretation") + return normalizeInterpretations(env.Interpretations) + } + log.Warn().Str("body_prefix", truncateContent(body, 300)).Msg("parseInterpretations: envelope parsed but interpretations array is empty") + return nil, false + } else { + log.Warn().Err(err).Str("body_prefix", truncateContent(body, 300)).Msg("parseInterpretations: envelope unmarshal failed") } // Bare array fallback. @@ -1590,8 +1606,9 @@ func parseInterpretations(text string) ([]Interpretation, bool) { func normalizeInterpretations(in []Interpretation) ([]Interpretation, bool) { out := make([]Interpretation, 0, len(in)) - for _, interp := range in { + for i, interp := range in { if strings.TrimSpace(interp.SQL) == "" { + log.Warn().Int("index", i).Str("name", interp.Name).Str("chart_type", interp.ChartType).Msg("normalizeInterpretations: skipping interpretation with empty sql") continue } if strings.TrimSpace(interp.ChartType) == "" { @@ -1604,6 +1621,7 @@ func normalizeInterpretations(in []Interpretation) ([]Interpretation, bool) { if strings.TrimSpace(interp.Title) == "" { interp.Title = interp.ChartType + " chart" } + log.Debug().Int("index", i).Str("name", interp.Name).Str("chart_type", interp.ChartType).Int("sql_len", len(interp.SQL)).Msg("normalizeInterpretations: accepted") out = append(out, interp) if len(out) >= maxInterpretations { break @@ -1651,21 +1669,25 @@ func (a *Agent) runMultiInterpret( raw, err := a.completeOnce(ctx, clog, 2, "interpret", "multi", 1, system, userMsg) if err != nil { log.Error().Err(err).Msg("multi-interpret LLM call failed") - cat := CategorizeLLMError(err) - msg := GetLLMFallbackMessage(cat) + cat := aiconnectors.CategorizeLLMError(err) + msg := "" + if tpl, ok := LLMErrorTemplates[cat]; ok { + msg = tpl.Message + } if msg != "" { clog.FinalResponse(msg + " (stopped after " + string(cat) + ")") return msg, nil, debug, nil } - return "I had trouble understanding that question. Please try rephrasing.", nil, debug, nil //nolint:nilerr + log.Warn().Err(err).Str("category", string(cat)).Msg("multi-interpret LLM call failed; no known fallback") + return "I wasn't able to generate an answer this time. Please try rephrasing your question.", nil, debug, nil //nolint:nilerr } debug.LLMRawResponse = raw interps, ok := parseInterpretations(raw) if !ok { log.Warn().Str("raw_preview", truncateContent(raw, 200)).Msg("could not parse interpretations from LLM response") - return "I could not produce a valid analysis plan. Please try rephrasing.", nil, debug, nil + return "I wasn't able to generate an answer this time. Please try rephrasing your question.", nil, debug, nil } debug.Interpretations = interps diff --git a/internal/mcpagent/provider.go b/internal/mcpagent/provider.go index 8323c420..66a88b82 100644 --- a/internal/mcpagent/provider.go +++ b/internal/mcpagent/provider.go @@ -3,14 +3,11 @@ package mcpagent import ( "context" "encoding/json" - "errors" "fmt" - "strings" "github.com/livereview/internal/aiconnectors" "github.com/rs/zerolog/log" "github.com/tmc/langchaingo/llms" - "google.golang.org/api/googleapi" ) type Provider struct { @@ -190,99 +187,51 @@ func (p *Provider) historyToMessages(history []HistoryEntry) []llms.MessageConte return messages } -type LLMErrorCategory string - -const ( - ErrCategoryAuth LLMErrorCategory = "auth_failed" - ErrCategoryOverload LLMErrorCategory = "overloaded" - ErrCategoryTimeout LLMErrorCategory = "timeout" - ErrCategoryUnknown LLMErrorCategory = "unknown" -) - -// CategorizeLLMError maps raw SDK errors from Langchain or underlying clients -// into our standard failure categories for consistent UI fallback rendering. -func CategorizeLLMError(err error) LLMErrorCategory { - if err == nil { - return ErrCategoryUnknown - } - - // 1. Langchain-Go normalized errors - var llmsErr *llms.Error - if errors.As(err, &llmsErr) { - if llmsErr.Code == llms.ErrCodeAuthentication || llmsErr.Code == llms.ErrCodeResourceNotFound { - return ErrCategoryAuth - } - if llmsErr.Code == llms.ErrCodeRateLimit { - return ErrCategoryOverload - } - } - - // 2. Raw Google API errors (often leaked by langchaingo) - var gErr *googleapi.Error - if errors.As(err, &gErr) { - if gErr.Code == 401 || gErr.Code == 403 || gErr.Code == 404 { - return ErrCategoryAuth - } - if gErr.Code == 429 || gErr.Code == 500 || gErr.Code == 502 || gErr.Code == 503 || gErr.Code == 504 { - return ErrCategoryOverload - } - } - - // 3. Generic HTTP status codes - var scErr interface{ StatusCode() int } - if errors.As(err, &scErr) { - code := scErr.StatusCode() - if code == 401 || code == 403 || code == 404 { - return ErrCategoryAuth - } - if code == 429 || code == 500 || code == 502 || code == 503 || code == 504 { - return ErrCategoryOverload - } - } - - var hscErr interface{ HTTPStatusCode() int } - if errors.As(err, &hscErr) { - code := hscErr.HTTPStatusCode() - if code == 401 || code == 403 || code == 404 { - return ErrCategoryAuth - } - if code == 429 || code == 500 || code == 502 || code == 503 || code == 504 { - return ErrCategoryOverload - } - } - - // 4. Fallback string matching - msg := strings.ToLower(err.Error()) - if strings.Contains(msg, "context deadline") || strings.Contains(msg, "timeout") { - return ErrCategoryTimeout - } - if strings.Contains(msg, "rate limit") || strings.Contains(msg, "too many requests") || - strings.Contains(msg, "status code: 429") || strings.Contains(msg, "status code: 50") || - strings.Contains(msg, "service unavailable") || strings.Contains(msg, "bad gateway") || - strings.Contains(msg, "high demand") { - return ErrCategoryOverload - } - if strings.Contains(msg, "deprecated") || strings.Contains(msg, "model not found") || - strings.Contains(msg, "status code: 401") || strings.Contains(msg, "status code: 403") || - strings.Contains(msg, "status code: 404") || strings.Contains(msg, "unauthorized") || - strings.Contains(msg, "forbidden") { - return ErrCategoryAuth - } - - return ErrCategoryUnknown +// LLMErrorTemplate defines the UI representation of an LLM error, +// including both the chat message text and the Action Card to render. +type LLMErrorTemplate struct { + Message string + ActionCard ActionCard } -// GetLLMFallbackMessage returns a safe, beautiful UI response for the client -// based on the categorized LLM error. Returns empty string for unknown errors. -func GetLLMFallbackMessage(cat LLMErrorCategory) string { - switch cat { - case ErrCategoryAuth: - return "> **Action Required: AI Provider Issue**\n> \n> The AI Provider's API key is invalid or the model is missing. Please configure a valid provider in settings to continue.\n> \n> [btn:Configure AI Provider](/settings#ai)" - case ErrCategoryOverload: - return "> **AI Provider is Busy**\n> \n> The AI provider is currently experiencing high traffic. If you continue to see this issue, we suggest changing to a different model or provider." - case ErrCategoryTimeout: - return "> **Analysis Took Too Long**\n> \n> The AI took too long to generate your response and timed out. Try asking a narrower or more specific question." - default: - return "" - } +// LLMErrorTemplates stores the fallback configurations for various LLM errors. +// This allows the frontend to automatically render the correct box without +// hardcoded display logic. +var LLMErrorTemplates = map[aiconnectors.LLMErrorCategory]LLMErrorTemplate{ + aiconnectors.ErrCategoryAuth: { + Message: "**Action Required: AI Provider Issue**\n\nThe AI Provider's API key is invalid or the model is missing. Please configure a valid provider in settings to continue.", + ActionCard: ActionCard{ + Title: "Configuration Required", + Description: "Please edit the current AI provider configuration to continue.", + ButtonText: "Configure AI Provider", + ActionURL: "/ai", + }, + }, + aiconnectors.ErrCategoryOverload: { + Message: "**AI Provider is Busy**\n\nThe selected model is currently experiencing high traffic. If you continue to see this issue, please edit your configuration to change to a different model or provider.", + ActionCard: ActionCard{ + Title: "High Traffic Detected", + Description: "Please select a different model in settings.", + ButtonText: "Configure AI Provider", + ActionURL: "/ai", + }, + }, + aiconnectors.ErrCategoryDeprecated: { + Message: "**Model No Longer Available**\n\nThe selected AI model has been deprecated or removed by the provider. Please update your AI provider configuration to use a different model.", + ActionCard: ActionCard{ + Title: "Model No Longer Available", + Description: "The selected model has been deprecated. Please update your AI provider configuration.", + ButtonText: "Configure AI Provider", + ActionURL: "/ai", + }, + }, + aiconnectors.ErrCategoryTimeout: { + Message: "**Analysis Took Too Long**\n\nThe AI took too long to generate your response and timed out. Try asking a narrower or more specific question.", + ActionCard: ActionCard{ + Title: "Timeout Error", + Description: "The request took too long to complete.", + ButtonText: "Try Again", + ActionURL: "/chat", + }, + }, } diff --git a/internal/mcpagent/provider_test.go b/internal/mcpagent/provider_test.go new file mode 100644 index 00000000..d9aac34b --- /dev/null +++ b/internal/mcpagent/provider_test.go @@ -0,0 +1,53 @@ +package mcpagent + +import ( + "testing" + + "github.com/livereview/internal/aiconnectors" +) + +func TestLLMErrorTemplates(t *testing.T) { + tests := []struct { + name string + category aiconnectors.LLMErrorCategory + want string + }{ + { + name: "Auth Error", + category: aiconnectors.ErrCategoryAuth, + want: "**Action Required: AI Provider Issue**\n\nThe AI Provider's API key is invalid or the model is missing. Please configure a valid provider in settings to continue.", + }, + { + name: "Overload Error", + category: aiconnectors.ErrCategoryOverload, + want: "**AI Provider is Busy**\n\nThe selected model is currently experiencing high traffic. If you continue to see this issue, please edit your configuration to change to a different model or provider.", + }, + { + name: "Timeout Error", + category: aiconnectors.ErrCategoryTimeout, + want: "**Analysis Took Too Long**\n\nThe AI took too long to generate your response and timed out. Try asking a narrower or more specific question.", + }, + { + name: "Unknown Error", + category: aiconnectors.ErrCategoryUnknown, + want: "", + }, + { + name: "Unmapped Category", + category: aiconnectors.LLMErrorCategory("something_else"), + want: "", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + got := "" + if tpl, ok := LLMErrorTemplates[tt.category]; ok { + got = tpl.Message + } + if got != tt.want { + t.Errorf("LLMErrorTemplates[%v].Message = %q, want %q", tt.category, got, tt.want) + } + }) + } +} diff --git a/internal/mcpagent/types.go b/internal/mcpagent/types.go index e07130a3..e5f39b06 100644 --- a/internal/mcpagent/types.go +++ b/internal/mcpagent/types.go @@ -134,3 +134,12 @@ type DebugResultEntry struct { RetryCount int `json:"retry_count,omitempty"` Retries []RetryInfo `json:"retries,omitempty"` } + +// ActionCard represents an actionable message for the UI to render as a card +// instead of plain text, like when a configuration change is required. +type ActionCard struct { + Title string `json:"title"` + Description string `json:"description"` + ButtonText string `json:"button_text"` + ActionURL string `json:"action_url"` +} diff --git a/ui/src/api/chatConversations.ts b/ui/src/api/chatConversations.ts index 1cadb9f2..7f2f8c56 100644 --- a/ui/src/api/chatConversations.ts +++ b/ui/src/api/chatConversations.ts @@ -27,6 +27,12 @@ export interface ConversationMessage { files?: ChatFile[]; suggested_questions?: SuggestedQuestionCategory[]; debug_artifacts?: unknown; + action_card?: { + title: string; + description: string; + button_text: string; + action_url: string; + }; } export interface ConversationDetail { diff --git a/ui/src/api/chatbot.ts b/ui/src/api/chatbot.ts index f081e071..1deb4193 100644 --- a/ui/src/api/chatbot.ts +++ b/ui/src/api/chatbot.ts @@ -142,6 +142,13 @@ export interface ChatResponse { debug_artifacts?: unknown; sessionId?: string; conversationId: number; + /** 'ai_auth' | 'ai_busy' — set only when the backend hit a provider error. */ + action_card?: { + title: string; + description: string; + button_text: string; + action_url: string; + }; } // The backend now owns conversation history: it loads prior turns by diff --git a/ui/src/pages/Chatbot/ChatConversation.tsx b/ui/src/pages/Chatbot/ChatConversation.tsx index 13e29e61..f07761b2 100644 --- a/ui/src/pages/Chatbot/ChatConversation.tsx +++ b/ui/src/pages/Chatbot/ChatConversation.tsx @@ -117,6 +117,12 @@ interface ChatEntry { files?: ChatFile[]; suggestedQuestions?: SuggestedQuestionCategory[]; debugArtifacts?: DebugArtifacts | null; + actionCard?: { + title: string; + description: string; + button_text: string; + action_url: string; + }; } function formatRowCount(rows?: number): string { @@ -891,6 +897,7 @@ export const ChatConversation: React.FC<{ surface: ChatSurface }> = ({ surface } files: m.files && m.files.length > 0 ? m.files : undefined, suggestedQuestions: m.suggested_questions, debugArtifacts: m.debug_artifacts as DebugArtifacts | undefined, + actionCard: m.action_card, })), ); }, [conversationDetail]); @@ -933,6 +940,7 @@ export const ChatConversation: React.FC<{ surface: ChatSurface }> = ({ surface } } const result = await sendChatMessage(text, activeConversationId); + console.log("LIVE API RESPONSE:", result); const assistantEntry: ChatEntry = { id: generateId(), @@ -942,6 +950,7 @@ export const ChatConversation: React.FC<{ surface: ChatSurface }> = ({ surface } files: result.files && result.files.length > 0 ? result.files : undefined, suggestedQuestions: result.suggested_questions, debugArtifacts: result.debug_artifacts as DebugArtifacts | undefined, + actionCard: result.action_card, }; setMessages((prev) => [...prev, assistantEntry]); queryClient.invalidateQueries({ queryKey: CONVERSATIONS_QUERY_KEY }); @@ -1082,8 +1091,8 @@ export const ChatConversation: React.FC<{ surface: ChatSurface }> = ({ surface } const debugModalMsg = debugModalMsgId ? messages.find((m) => m.id === debugModalMsgId) : undefined; return ( -
-
+
+
@@ -1142,7 +1151,7 @@ export const ChatConversation: React.FC<{ surface: ChatSurface }> = ({ surface }
-
+
{isSuperAdmin && !dismissedProdUrlWarning && ( setDismissedProdUrlWarning(true)} /> @@ -1472,34 +1481,25 @@ export const ChatConversation: React.FC<{ surface: ChatSurface }> = ({ surface } ))}
)} - {msg.text && (() => { - const isAiError = msg.text.includes('Action Required: AI Provider Issue'); - const displayText = isAiError && msg.text.includes('Please configure a valid provider to continue:') - ? msg.text.split('Please configure a valid provider to continue:')[0].trim() - : msg.text; - - return ( - <> -
0) || (msg.files && msg.files.length > 0) ? 'mt-6' : ''} text-base leading-snug whitespace-pre-wrap break-words text-slate-200 [&>*:first-child]:mt-0`}> - {formatText(displayText)} -
- {isAiError && ( -
-
-

Configuration Required

-

Please edit the current AI provider model to continue.

-
- -
- )} - - ); - })()} + {msg.text && ( +
0) || (msg.files && msg.files.length > 0) ? 'mt-6' : ''} text-base leading-snug whitespace-pre-wrap break-words text-slate-200 [&>*:first-child]:mt-0`}> + {formatText(msg.text)} +
+ )} + {msg.actionCard && ( +
+
+

{msg.actionCard.title}

+

{msg.actionCard.description}

+
+ +
+ )} {msg.suggestedQuestions && msg.suggestedQuestions.length > 0 && (
{msg.suggestedQuestions.map((cat: SuggestedQuestionCategory, catIdx: number) => ( @@ -1546,7 +1546,7 @@ export const ChatConversation: React.FC<{ surface: ChatSurface }> = ({ surface }
-
+
Date: Sat, 3 Oct 2026 15:30:05 +0530 Subject: [PATCH 03/11] Clean Up WebChat Logging and Refactor Chat UI Layout --- internal/api/webchat_handler.go | 2 +- ui/src/pages/Chatbot/ChatConversation.tsx | 15 ++++++++------- ui/src/pages/Chatbot/ChatLayout.tsx | 12 +++++++++++- 3 files changed, 20 insertions(+), 9 deletions(-) diff --git a/internal/api/webchat_handler.go b/internal/api/webchat_handler.go index 109c0dbf..8978cb52 100644 --- a/internal/api/webchat_handler.go +++ b/internal/api/webchat_handler.go @@ -356,7 +356,7 @@ func (s *Server) HandleWebChat(c echo.Context) error { resp.Files = append(resp.Files, chatFileFromArtifact(art, fileIDs[i])) } - log.Info().Interface("finalResp", resp).Msg("WebChat: returning response to client") + // log.Info().Interface("finalResp", resp).Msg("WebChat: returning response to client") return c.JSON(http.StatusOK, resp) } diff --git a/ui/src/pages/Chatbot/ChatConversation.tsx b/ui/src/pages/Chatbot/ChatConversation.tsx index f07761b2..71931629 100644 --- a/ui/src/pages/Chatbot/ChatConversation.tsx +++ b/ui/src/pages/Chatbot/ChatConversation.tsx @@ -68,9 +68,7 @@ const ContextDetails: React.FC<{ context: ChartContext }> = ({ context }) => ( // Unified logo wrapper to ensure the 16px visual right gap and 8px visual left gap // stay mathematically synchronized across the header, chat messages, and loading states. const LiviLogo: React.FC<{ className?: string }> = ({ className = '' }) => ( -
- Bot -
+ Bot ); // Debug artifacts (SQL, CSV, Vega spec, schema context, system prompt, raw @@ -1091,8 +1089,11 @@ export const ChatConversation: React.FC<{ surface: ChatSurface }> = ({ surface } const debugModalMsg = debugModalMsgId ? messages.find((m) => m.id === debugModalMsgId) : undefined; return ( -
-
+
+ {/* Header: pl-4 pr-6 instead of px-4 so max-w-4xl mx-auto centers identically + to the messages section which has scrollbar-gutter:stable (8px) on the right. + Math: header right=24px, messages right=16px(px-4)+8px(gutter)=24px → same. */} +
@@ -1151,7 +1152,7 @@ export const ChatConversation: React.FC<{ surface: ChatSurface }> = ({ surface }
-
+
{isSuperAdmin && !dismissedProdUrlWarning && ( setDismissedProdUrlWarning(true)} /> @@ -1546,7 +1547,7 @@ export const ChatConversation: React.FC<{ surface: ChatSurface }> = ({ surface }
-
+
{ const reservedWidth = useChatSidebarReservedWidth(); const resizing = useChatSidebarResizing(); + useEffect(() => { + // Disable body scrolling on the chat layout to prevent a double scrollbar + // caused by subpixel layout overflows or banners pushing the layout down. + const originalOverflow = document.body.style.overflow; + document.body.style.overflow = 'hidden'; + return () => { + document.body.style.overflow = originalOverflow; + }; + }, []); + return (
From fd0926c771409fd54dc4ce198c481a320b3645b7 Mon Sep 17 00:00:00 2001 From: Ganesh Kumar Date: Sat, 3 Oct 2026 15:41:39 +0530 Subject: [PATCH 04/11] Add Raw LLM Error Tracking to Debug Artifacts --- internal/api/webchat_handler.go | 5 ++++ internal/mcpagent/agent.go | 4 ++- internal/mcpagent/analytics.go | 1 + internal/mcpagent/provider.go | 8 +++--- internal/mcpagent/types.go | 1 + ui/src/pages/Chatbot/ChatConversation.tsx | 33 ++++++++++++++++------- 6 files changed, 37 insertions(+), 15 deletions(-) diff --git a/internal/api/webchat_handler.go b/internal/api/webchat_handler.go index 8978cb52..f9ebcef8 100644 --- a/internal/api/webchat_handler.go +++ b/internal/api/webchat_handler.go @@ -268,6 +268,11 @@ func (s *Server) HandleWebChat(c echo.Context) error { log.Error().Err(err).Msg("WebChat: failed to marshal rawCard") } } + if rawDebug, ok := entry["debug_artifacts"]; ok && rawDebug != nil && resp.DebugArtifacts == nil { + if b, err := json.Marshal(rawDebug); err == nil { + resp.DebugArtifacts = b + } + } if sq, ok := entry["suggested_questions"].([]mcpagent.SuggestedQuestionCategory); ok && len(sq) > 0 { resp.SuggestedQuestions = sq break diff --git a/internal/mcpagent/agent.go b/internal/mcpagent/agent.go index bf43a677..ab24b862 100644 --- a/internal/mcpagent/agent.go +++ b/internal/mcpagent/agent.go @@ -182,6 +182,7 @@ func (a *Agent) RunTurnWithArtifacts(ctx context.Context, history []HistoryEntry "text": msg, "suggested_questions": DefaultAIErrorSuggestedQuestions, "action_card": card, + "debug_artifacts": &DebugArtifacts{RawLLMError: err.Error()}, }) } else { history = append(history, HistoryEntry{ @@ -191,7 +192,7 @@ func (a *Agent) RunTurnWithArtifacts(ctx context.Context, history []HistoryEntry "suggested_questions": DefaultAIErrorSuggestedQuestions, }) } - return msg, history, nil, nil, nil + return msg, history, nil, &DebugArtifacts{RawLLMError: err.Error()}, nil } log.Warn().Err(err).Msg("call #0 classify failed with unparseable or unknown error; degrading to product_guidance for this turn") shape = shapeProductGuidance @@ -367,6 +368,7 @@ func (a *Agent) runStepLoop( "text": msg, "suggested_questions": DefaultAIErrorSuggestedQuestions, "action_card": card, + "debug_artifacts": &DebugArtifacts{RawLLMError: err.Error()}, }) } else { history = append(history, HistoryEntry{ diff --git a/internal/mcpagent/analytics.go b/internal/mcpagent/analytics.go index 22cf0e5c..001f5e06 100644 --- a/internal/mcpagent/analytics.go +++ b/internal/mcpagent/analytics.go @@ -1668,6 +1668,7 @@ func (a *Agent) runMultiInterpret( debug.FullRequest = system + "\n---\n" + userMsg raw, err := a.completeOnce(ctx, clog, 2, "interpret", "multi", 1, system, userMsg) if err != nil { + debug.RawLLMError = err.Error() log.Error().Err(err).Msg("multi-interpret LLM call failed") cat := aiconnectors.CategorizeLLMError(err) msg := "" diff --git a/internal/mcpagent/provider.go b/internal/mcpagent/provider.go index 66a88b82..850c0758 100644 --- a/internal/mcpagent/provider.go +++ b/internal/mcpagent/provider.go @@ -199,7 +199,7 @@ type LLMErrorTemplate struct { // hardcoded display logic. var LLMErrorTemplates = map[aiconnectors.LLMErrorCategory]LLMErrorTemplate{ aiconnectors.ErrCategoryAuth: { - Message: "**Action Required: AI Provider Issue**\n\nThe AI Provider's API key is invalid or the model is missing. Please configure a valid provider in settings to continue.", + Message: "> **Action Required: AI Provider Issue**\n> \n> The AI Provider's API key is invalid or the model is missing. Please configure a valid provider in settings to continue.", ActionCard: ActionCard{ Title: "Configuration Required", Description: "Please edit the current AI provider configuration to continue.", @@ -208,7 +208,7 @@ var LLMErrorTemplates = map[aiconnectors.LLMErrorCategory]LLMErrorTemplate{ }, }, aiconnectors.ErrCategoryOverload: { - Message: "**AI Provider is Busy**\n\nThe selected model is currently experiencing high traffic. If you continue to see this issue, please edit your configuration to change to a different model or provider.", + Message: "> **AI Provider is Busy**\n> \n> The selected model is currently experiencing high traffic. If you continue to see this issue, please edit your configuration to change to a different model or provider.", ActionCard: ActionCard{ Title: "High Traffic Detected", Description: "Please select a different model in settings.", @@ -217,7 +217,7 @@ var LLMErrorTemplates = map[aiconnectors.LLMErrorCategory]LLMErrorTemplate{ }, }, aiconnectors.ErrCategoryDeprecated: { - Message: "**Model No Longer Available**\n\nThe selected AI model has been deprecated or removed by the provider. Please update your AI provider configuration to use a different model.", + Message: "> **Model No Longer Available**\n> \n> The selected AI model has been deprecated or removed by the provider. Please update your AI provider configuration to use a different model.", ActionCard: ActionCard{ Title: "Model No Longer Available", Description: "The selected model has been deprecated. Please update your AI provider configuration.", @@ -226,7 +226,7 @@ var LLMErrorTemplates = map[aiconnectors.LLMErrorCategory]LLMErrorTemplate{ }, }, aiconnectors.ErrCategoryTimeout: { - Message: "**Analysis Took Too Long**\n\nThe AI took too long to generate your response and timed out. Try asking a narrower or more specific question.", + Message: "> **Analysis Took Too Long**\n> \n> The AI took too long to generate your response and timed out. Try asking a narrower or more specific question.", ActionCard: ActionCard{ Title: "Timeout Error", Description: "The request took too long to complete.", diff --git a/internal/mcpagent/types.go b/internal/mcpagent/types.go index e5f39b06..77e2391f 100644 --- a/internal/mcpagent/types.go +++ b/internal/mcpagent/types.go @@ -117,6 +117,7 @@ type DebugArtifacts struct { FullRequest string `json:"full_request"` // system_prompt + schema_context sent to LLM Interpretations []Interpretation `json:"interpretations"` Results []DebugResultEntry `json:"results"` + RawLLMError string `json:"raw_llm_error,omitempty"` } // DebugResultEntry is one interpretation's outcome for the debug view. diff --git a/ui/src/pages/Chatbot/ChatConversation.tsx b/ui/src/pages/Chatbot/ChatConversation.tsx index 71931629..b2e669e1 100644 --- a/ui/src/pages/Chatbot/ChatConversation.tsx +++ b/ui/src/pages/Chatbot/ChatConversation.tsx @@ -105,6 +105,7 @@ interface DebugArtifacts { repaired_sql?: string; }>; }>; + raw_llm_error?: string; } interface ChatEntry { @@ -1488,17 +1489,29 @@ export const ChatConversation: React.FC<{ surface: ChatSurface }> = ({ surface }
)} {msg.actionCard && ( -
-
-

{msg.actionCard.title}

-

{msg.actionCard.description}

+
+
+
+

{msg.actionCard.title}

+

{msg.actionCard.description}

+
+
- + {msg.debugArtifacts?.raw_llm_error && ( +
+ + Debug logs + +
+                                {msg.debugArtifacts.raw_llm_error}
+                              
+
+ )}
)} {msg.suggestedQuestions && msg.suggestedQuestions.length > 0 && ( From fc9d523c3a56cf5014452d5f2a8821cfd7e5aed0 Mon Sep 17 00:00:00 2001 From: Ganesh Kumar Date: Sat, 3 Oct 2026 15:51:00 +0530 Subject: [PATCH 05/11] Update Error Identification and Layout in Chat Conversation --- ui/src/pages/Chatbot/ChatConversation.tsx | 21 ++++++++++++++------- 1 file changed, 14 insertions(+), 7 deletions(-) diff --git a/ui/src/pages/Chatbot/ChatConversation.tsx b/ui/src/pages/Chatbot/ChatConversation.tsx index b2e669e1..d2beba86 100644 --- a/ui/src/pages/Chatbot/ChatConversation.tsx +++ b/ui/src/pages/Chatbot/ChatConversation.tsx @@ -222,22 +222,29 @@ function formatText(rawText: string): React.ReactNode[] { } i = currI - 1; - const isError = blockquoteLines.length > 0 && blockquoteLines[0].includes('Action Required:'); + const errorPhrases = [ + 'Action Required:', + 'AI Provider is Busy', + 'Model No Longer Available', + 'Analysis Took Too Long', + 'AI Tools Unavailable' + ]; + const isError = blockquoteLines.length > 0 && errorPhrases.some(phrase => blockquoteLines[0].includes(phrase)); if (isError) { parts.push( -
-
+
+
{formatLine(blockquoteLines[0])}
-
+
{blockquoteLines.slice(1).map((bLine, bIdx) => (
*:first-child]:mt-0'}> {formatLine(bLine)}
))}
-
+
); } else { parts.push( @@ -1222,7 +1229,7 @@ export const ChatConversation: React.FC<{ surface: ChatSurface }> = ({ surface }
) : (
- +
{msg.charts && msg.charts.length > 0 && (
@@ -1550,7 +1557,7 @@ export const ChatConversation: React.FC<{ surface: ChatSurface }> = ({ surface } {isLoading && (
- +
From a6add1de778b2fef0c4c9617a66d56e08ccd01af Mon Sep 17 00:00:00 2001 From: Ganesh Kumar Date: Sat, 3 Oct 2026 17:44:06 +0530 Subject: [PATCH 06/11] Refactor LLM Classification Error Handling --- internal/mcpagent/agent.go | 67 +++++++++----------------------------- 1 file changed, 15 insertions(+), 52 deletions(-) diff --git a/internal/mcpagent/agent.go b/internal/mcpagent/agent.go index ab24b862..ca01c198 100644 --- a/internal/mcpagent/agent.go +++ b/internal/mcpagent/agent.go @@ -164,37 +164,7 @@ func (a *Agent) RunTurnWithArtifacts(ctx context.Context, history []HistoryEntry res, err := a.classify(ctx, history, userText, clog) shape := res.Shape if err != nil { - errCategory := aiconnectors.CategorizeLLMError(err) - msg := "" - var card *ActionCard - if tpl, ok := LLMErrorTemplates[errCategory]; ok { - msg = tpl.Message - card = &tpl.ActionCard - } - if msg != "" { - log.Warn().Err(err).Str("category", string(errCategory)).Msg("call #0 classify failed with provider error; short-circuiting turn") - clog.FinalResponse(msg + " (stopped after " + string(errCategory) + ")") - history = append(history, HistoryEntry{"role": "user", "content": userText}) - if card != nil { - history = append(history, HistoryEntry{ - "role": "assistant", - "content": msg, - "text": msg, - "suggested_questions": DefaultAIErrorSuggestedQuestions, - "action_card": card, - "debug_artifacts": &DebugArtifacts{RawLLMError: err.Error()}, - }) - } else { - history = append(history, HistoryEntry{ - "role": "assistant", - "content": msg, - "text": msg, - "suggested_questions": DefaultAIErrorSuggestedQuestions, - }) - } - return msg, history, nil, &DebugArtifacts{RawLLMError: err.Error()}, nil - } - log.Warn().Err(err).Msg("call #0 classify failed with unparseable or unknown error; degrading to product_guidance for this turn") + log.Warn().Err(err).Str("category", string(aiconnectors.CategorizeLLMError(err))).Msg("call #0 classify failed; degrading to product_guidance for this turn") shape = shapeProductGuidance } @@ -227,6 +197,7 @@ func (a *Agent) RunTurnWithArtifacts(ctx context.Context, history []HistoryEntry entry["action_card"] = tpl.ActionCard } } + history = append(history, entry) return text, history, artifacts, debugArt, nil } @@ -287,7 +258,8 @@ func (a *Agent) RunTurnWithArtifacts(ctx context.Context, history []HistoryEntry } clog.BranchSelected(string(shape), len(systemPrompt), len(tools)) - return a.runStepLoop(ctx, history, userText, systemPrompt, tools, callNumber, jsonMode, planRetried, "", clog) + text, history, artifacts, debugArt, err := a.runStepLoop(ctx, history, userText, systemPrompt, tools, callNumber, jsonMode, planRetried, "", clog) + return text, history, artifacts, debugArt, err } // Non-analytics path (plain tool-only agent). @@ -342,7 +314,7 @@ func (a *Agent) runStepLoop( if jsonMode { completeOpts = append(completeOpts, llms.WithJSONMode()) } - + response, usage, err := a.provider.Complete(ctx, history, tools, completeOpts...) aiElapsed := time.Since(aiStart) if err != nil { @@ -361,23 +333,18 @@ func (a *Agent) runStepLoop( } clog.FinalResponse(msg + " (stopped after " + string(errCategory) + ")") + entry := HistoryEntry{ + "role": "assistant", + "content": msg, + "text": msg, + "suggested_questions": DefaultAIErrorSuggestedQuestions, + } if card != nil { - history = append(history, HistoryEntry{ - "role": "assistant", - "content": msg, - "text": msg, - "suggested_questions": DefaultAIErrorSuggestedQuestions, - "action_card": card, - "debug_artifacts": &DebugArtifacts{RawLLMError: err.Error()}, - }) - } else { - history = append(history, HistoryEntry{ - "role": "assistant", - "content": msg, - "text": msg, - "suggested_questions": DefaultAIErrorSuggestedQuestions, - }) + entry["action_card"] = card + entry["debug_artifacts"] = &DebugArtifacts{RawLLMError: err.Error()} } + + history = append(history, entry) return msg, history, nil, nil, nil } log.Debug().Int("step", step).Int("response_len", len(response)).Msg("LLM call succeeded") @@ -662,10 +629,6 @@ func (a *Agent) interpretUserMessage(clog *logging.ChatTurnLogger, userText stri return msg, nil } - - - - // isAuthError reports whether a tool result signals an expired/invalid // session token or missing auth - the exact messages LiveReview's own auth // middleware returns (internal/api/auth/middleware.go). Retrying with a From 8e974d6bdccdf9502969ae65ced74435cf43291c Mon Sep 17 00:00:00 2001 From: Ganesh Kumar Date: Sat, 3 Oct 2026 18:35:06 +0530 Subject: [PATCH 07/11] Update LLM Error Handling and UI Styles --- internal/aiconnectors/errors.go | 12 ++++--- internal/mcpagent/agent.go | 40 ++++++++++++++--------- internal/mcpagent/analytics.go | 3 +- ui/src/pages/Chatbot/ChatConversation.tsx | 6 ++-- 4 files changed, 37 insertions(+), 24 deletions(-) diff --git a/internal/aiconnectors/errors.go b/internal/aiconnectors/errors.go index 7c15883e..c31475fe 100644 --- a/internal/aiconnectors/errors.go +++ b/internal/aiconnectors/errors.go @@ -24,6 +24,8 @@ func CategorizeLLMError(err error) LLMErrorCategory { return ErrCategoryUnknown } + + // 1. Langchain-Go normalized errors var llmsErr *llms.Error if errors.As(err, &llmsErr) { @@ -60,6 +62,12 @@ func CategorizeLLMError(err error) LLMErrorCategory { // 3. Fallback string matching msg := strings.ToLower(err.Error()) + if strings.Contains(msg, "deprecated") || strings.Contains(msg, "model not found") || + strings.Contains(msg, "model_not_found") || strings.Contains(msg, "no such model") || + strings.Contains(msg, "no longer available") { + return ErrCategoryDeprecated + } + if strings.Contains(msg, "context deadline") || strings.Contains(msg, "timeout") { return ErrCategoryTimeout } @@ -70,10 +78,6 @@ func CategorizeLLMError(err error) LLMErrorCategory { strings.Contains(msg, "high demand") { return ErrCategoryOverload } - if strings.Contains(msg, "deprecated") || strings.Contains(msg, "model not found") || - strings.Contains(msg, "model_not_found") || strings.Contains(msg, "no such model") { - return ErrCategoryDeprecated - } if strings.Contains(msg, "status code: 401") || strings.Contains(msg, "status code: 403") || strings.Contains(msg, "status code: 404") || strings.Contains(msg, "unauthorized") || strings.Contains(msg, "error 401") || strings.Contains(msg, "error 403") || diff --git a/internal/mcpagent/agent.go b/internal/mcpagent/agent.go index ca01c198..71e32cf3 100644 --- a/internal/mcpagent/agent.go +++ b/internal/mcpagent/agent.go @@ -178,25 +178,31 @@ func (a *Agent) RunTurnWithArtifacts(ctx context.Context, history []HistoryEntry return text, history, artifacts, debugArt, err } entry := HistoryEntry{"role": "assistant", "content": text, "text": text} - switch { - case text == NoDataAnalyticsResponseText: + if text == NoDataAnalyticsResponseText { entry["suggested_questions"] = DefaultNoDataSuggestedQuestions - case strings.Contains(text, "Action Required: AI Provider Issue"): - entry["suggested_questions"] = DefaultAIErrorSuggestedQuestions - if tpl, ok := LLMErrorTemplates[aiconnectors.ErrCategoryAuth]; ok { - entry["action_card"] = tpl.ActionCard - } - case strings.Contains(text, "AI Provider is Busy"): - entry["suggested_questions"] = DefaultAIErrorSuggestedQuestions - if tpl, ok := LLMErrorTemplates[aiconnectors.ErrCategoryOverload]; ok { - entry["action_card"] = tpl.ActionCard + } else { + for _, tpl := range LLMErrorTemplates { + if strings.HasPrefix(text, tpl.Message) { + entry["suggested_questions"] = DefaultAIErrorSuggestedQuestions + card := tpl.ActionCard // copy + entry["action_card"] = &card + entry["debug_artifacts"] = debugArt + break } - case strings.Contains(text, "Model No Longer Available"): + } + + // Also check for format errors, which are hallucinations, not provider errors + if strings.Contains(text, "Model Output Error") { entry["suggested_questions"] = DefaultAIErrorSuggestedQuestions - if tpl, ok := LLMErrorTemplates[aiconnectors.ErrCategoryDeprecated]; ok { - entry["action_card"] = tpl.ActionCard + entry["action_card"] = &ActionCard{ + Title: "Formatting Error", + Description: "The model failed to format the response correctly.", + ButtonText: "Try Again", + ActionURL: "/chat", } + entry["debug_artifacts"] = debugArt } + } // close else block history = append(history, entry) return text, history, artifacts, debugArt, nil @@ -339,13 +345,15 @@ func (a *Agent) runStepLoop( "text": msg, "suggested_questions": DefaultAIErrorSuggestedQuestions, } + var debugArt *DebugArtifacts if card != nil { entry["action_card"] = card - entry["debug_artifacts"] = &DebugArtifacts{RawLLMError: err.Error()} + debugArt = &DebugArtifacts{RawLLMError: err.Error()} + entry["debug_artifacts"] = debugArt } history = append(history, entry) - return msg, history, nil, nil, nil + return msg, history, nil, debugArt, nil } log.Debug().Int("step", step).Int("response_len", len(response)).Msg("LLM call succeeded") clog.AIResponse(callNumber, step, aiElapsed, usage.InputTokens, usage.OutputTokens, response) diff --git a/internal/mcpagent/analytics.go b/internal/mcpagent/analytics.go index 001f5e06..d4778697 100644 --- a/internal/mcpagent/analytics.go +++ b/internal/mcpagent/analytics.go @@ -1688,7 +1688,8 @@ func (a *Agent) runMultiInterpret( interps, ok := parseInterpretations(raw) if !ok { log.Warn().Str("raw_preview", truncateContent(raw, 200)).Msg("could not parse interpretations from LLM response") - return "I wasn't able to generate an answer this time. Please try rephrasing your question.", nil, debug, nil + formatMsg := "> **Model Output Error**\n> \n> The AI returned data in an unexpected format. Please try rephrasing your question." + return formatMsg, nil, debug, nil } debug.Interpretations = interps diff --git a/ui/src/pages/Chatbot/ChatConversation.tsx b/ui/src/pages/Chatbot/ChatConversation.tsx index d2beba86..fc12bfac 100644 --- a/ui/src/pages/Chatbot/ChatConversation.tsx +++ b/ui/src/pages/Chatbot/ChatConversation.tsx @@ -1418,7 +1418,7 @@ export const ChatConversation: React.FC<{ surface: ChatSurface }> = ({ surface } )} {(chart.query || chart.time_range || chart.granularity || chart.context) && (
- + Data details
@@ -1469,7 +1469,7 @@ export const ChatConversation: React.FC<{ surface: ChatSurface }> = ({ surface } )} {(file.query || file.time_range || file.granularity || file.context) && (
- + Data details
@@ -1511,7 +1511,7 @@ export const ChatConversation: React.FC<{ surface: ChatSurface }> = ({ surface }
{msg.debugArtifacts?.raw_llm_error && (
- + Debug logs

From 6bc80af88d3bdc7d31eeabb3bd671cc7690a0792 Mon Sep 17 00:00:00 2001
From: Ganesh Kumar 
Date: Sat, 3 Oct 2026 18:54:29 +0530
Subject: [PATCH 08/11] Refactor Error Handling and Timeouts in AI and Chat
 Services

---
 internal/aiconnectors/errors.go           | 7 +++++--
 internal/api/webchat_handler.go           | 2 +-
 internal/mcpagent/analytics.go            | 2 +-
 internal/mcpagent/provider.go             | 9 +++++++++
 ui/src/pages/Chatbot/ChatConversation.tsx | 2 +-
 5 files changed, 17 insertions(+), 5 deletions(-)

diff --git a/internal/aiconnectors/errors.go b/internal/aiconnectors/errors.go
index c31475fe..7223b3eb 100644
--- a/internal/aiconnectors/errors.go
+++ b/internal/aiconnectors/errors.go
@@ -72,8 +72,11 @@ func CategorizeLLMError(err error) LLMErrorCategory {
 		return ErrCategoryTimeout
 	}
 	if strings.Contains(msg, "rate limit") || strings.Contains(msg, "too many requests") ||
-		strings.Contains(msg, "status code: 429") || strings.Contains(msg, "status code: 50") ||
-		strings.Contains(msg, "error 429") || strings.Contains(msg, "error 50") ||
+		strings.Contains(msg, "status code: 429") || strings.Contains(msg, "status code: 500") ||
+		strings.Contains(msg, "status code: 502") || strings.Contains(msg, "status code: 503") ||
+		strings.Contains(msg, "status code: 504") || strings.Contains(msg, "error 429") ||
+		strings.Contains(msg, "error 500") || strings.Contains(msg, "error 502") ||
+		strings.Contains(msg, "error 503") || strings.Contains(msg, "error 504") ||
 		strings.Contains(msg, "service unavailable") || strings.Contains(msg, "bad gateway") ||
 		strings.Contains(msg, "high demand") {
 		return ErrCategoryOverload
diff --git a/internal/api/webchat_handler.go b/internal/api/webchat_handler.go
index f9ebcef8..00985006 100644
--- a/internal/api/webchat_handler.go
+++ b/internal/api/webchat_handler.go
@@ -207,7 +207,7 @@ func (s *Server) HandleWebChat(c echo.Context) error {
 			SessionID:      sessionID,
 			ConversationID: convID,
 		}
-		return c.JSON(http.StatusOK, resp)
+		return c.JSON(http.StatusBadGateway, resp)
 	}
 
 	if pc.CurrentOrg != nil && pc.CurrentOrg.Name != "" {
diff --git a/internal/mcpagent/analytics.go b/internal/mcpagent/analytics.go
index d4778697..c564486c 100644
--- a/internal/mcpagent/analytics.go
+++ b/internal/mcpagent/analytics.go
@@ -43,7 +43,7 @@ const (
 
 	// analyticsTurnTimeout bounds the whole fan-out regardless of per-query
 	// timeouts, so a slow model cannot hold a request open indefinitely.
-	analyticsTurnTimeout = 300 * time.Second
+	analyticsTurnTimeout = 90 * time.Second
 )
 
 // AnalyticsEngine executes guard-rewritten SQL. Declared here rather than
diff --git a/internal/mcpagent/provider.go b/internal/mcpagent/provider.go
index 850c0758..4fca7544 100644
--- a/internal/mcpagent/provider.go
+++ b/internal/mcpagent/provider.go
@@ -234,4 +234,13 @@ var LLMErrorTemplates = map[aiconnectors.LLMErrorCategory]LLMErrorTemplate{
 			ActionURL:   "/chat",
 		},
 	},
+	aiconnectors.ErrCategoryUnknown: {
+		Message: "> **Unexpected Error**\n> \n> An unexpected error occurred while communicating with the AI provider. Please check the debug logs for more details.",
+		ActionCard: ActionCard{
+			Title:       "Unexpected Error",
+			Description: "An unknown error occurred during processing.",
+			ButtonText:  "Try Again",
+			ActionURL:   "/chat",
+		},
+	},
 }
diff --git a/ui/src/pages/Chatbot/ChatConversation.tsx b/ui/src/pages/Chatbot/ChatConversation.tsx
index fc12bfac..8833ed8e 100644
--- a/ui/src/pages/Chatbot/ChatConversation.tsx
+++ b/ui/src/pages/Chatbot/ChatConversation.tsx
@@ -1503,7 +1503,7 @@ export const ChatConversation: React.FC<{ surface: ChatSurface }> = ({ surface }
                               

{msg.actionCard.description}