feat(summarizer): resilient endpoint chain with local→cloud fallback (ADR-022)
The first friendly-pilot live run produced zero summaries: koala/phi4-mini hit three silent failure modes — 8k context overflow on long transcripts (HTTP 400), intermittent malformed JSON (highlights as a bare string), and no fallback wired at all (summarizer.New(primary, nil)). Keep phi4-mini as the fast primary and add resilience around it: - Ordered endpoint chain (summarizer.NewChain): phi4-mini → koala/phi4-14b (local) → berget/mistral-small (worst-case external). All reached through the one LiteLLM gateway by alias. - A parse failure now advances the chain like a transport error — the old Primary→Fallback shape returned the parse error without trying anyone else. - Tolerant parse: highlights/takeaways coerce string→[]string, absorbing the common small-model quirk without spending a fallback round-trip. - Transcript truncation (TAPIR_MAX_TRANSCRIPT_CHARS=18000) prevents the overflow rather than recovering from it; validated to fit phi4-mini's 8k window. - Bounded completion budget (TAPIR_SUMMARY_MAX_TOKENS=1500) — the old 8192 budget itself contributed to the overflow. Local-first guarantee preserved by ordering: external endpoint is tried only after every local one fails. TAPIR_CLOUD_FALLBACK_MODEL="" disables it entirely for client/NDA deployments. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -34,15 +34,35 @@ type Client struct {
|
||||
httpClient *http.Client
|
||||
}
|
||||
|
||||
// Option configures a Client at construction. Variadic so the existing 4-arg
|
||||
// call sites stay valid as new knobs are added.
|
||||
type Option func(*Client)
|
||||
|
||||
// WithMaxTokens overrides the per-request completion budget. The summarizer uses
|
||||
// this to cap completion for small-context models (e.g. koala/phi4-mini, 8k):
|
||||
// with the default 8192 budget, prompt + max_tokens overflows an 8k context and
|
||||
// the gateway returns HTTP 400. A non-positive n is ignored (keeps the default).
|
||||
func WithMaxTokens(n int) Option {
|
||||
return func(c *Client) {
|
||||
if n > 0 {
|
||||
c.maxTokens = n
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// New constructs a Client.
|
||||
func New(baseURL, apiKey, model string, timeout time.Duration) *Client {
|
||||
return &Client{
|
||||
func New(baseURL, apiKey, model string, timeout time.Duration, opts ...Option) *Client {
|
||||
c := &Client{
|
||||
baseURL: strings.TrimRight(baseURL, "/"),
|
||||
apiKey: apiKey,
|
||||
model: model,
|
||||
maxTokens: defaultMaxTokens,
|
||||
httpClient: &http.Client{Timeout: timeout},
|
||||
}
|
||||
for _, opt := range opts {
|
||||
opt(c)
|
||||
}
|
||||
return c
|
||||
}
|
||||
|
||||
type chatRequest struct {
|
||||
|
||||
@@ -64,6 +64,27 @@ func TestClient_SendsMaxTokens(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestClient_WithMaxTokens overrides the completion budget — the summarizer caps
|
||||
// it small so prompt + max_tokens fits a small-context model's window (8k).
|
||||
func TestClient_WithMaxTokens(t *testing.T) {
|
||||
var body chatRequest
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
_ = json.NewDecoder(r.Body).Decode(&body)
|
||||
_ = json.NewEncoder(w).Encode(map[string]any{
|
||||
"choices": []map[string]any{{"message": map[string]any{"content": "ok"}}},
|
||||
})
|
||||
}))
|
||||
defer srv.Close()
|
||||
|
||||
c := New(srv.URL, "", "test-model", 10*time.Second, WithMaxTokens(1500))
|
||||
if _, err := c.Complete(context.Background(), "sys", "user"); err != nil {
|
||||
t.Fatalf("Complete: %v", err)
|
||||
}
|
||||
if body.MaxTokens != 1500 {
|
||||
t.Errorf("max_tokens = %d, want 1500", body.MaxTokens)
|
||||
}
|
||||
}
|
||||
|
||||
func TestClient_ReturnsErrorOnNon200(t *testing.T) {
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
http.Error(w, "overloaded", http.StatusServiceUnavailable)
|
||||
|
||||
Reference in New Issue
Block a user