Infra ADR-0004 renamed the Gitea host. Bulk replace across go.mod and all .go import paths. Build and tests pass unchanged. Closes #20 Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Dt6aHEDWRjkK14Voi6HnGh
180 lines
6.5 KiB
Go
180 lines
6.5 KiB
Go
// Package chat implements the per-video deeper-dive chat (ADR-027): a read-only
|
|
// QA over a video's ALREADY-STORED transcript (ADR-021). It is the enforcement
|
|
// point for the feature's load-bearing safety property — stored-transcript-only:
|
|
// the Service has NO VideoSource and NO caption-fetch dependency, only a
|
|
// Completer factory, so it CANNOT reach YouTube or the rate gate by construction.
|
|
// The caller supplies the stored transcript text; chat never fetches.
|
|
//
|
|
// It reuses the same LiteLLM gateway as the summarizer (a chat is a different
|
|
// call, not a new integration) and the same transcript-truncation discipline
|
|
// (TAPIR_MAX_TRANSCRIPT_CHARS) so a long transcript fits a small-context model.
|
|
package chat
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"log/slog"
|
|
"strings"
|
|
"time"
|
|
"unicode/utf8"
|
|
|
|
"git.d-ma.be/mathias/tapir/internal/metrics"
|
|
)
|
|
|
|
// Completer is the minimal LLM chat surface the Service needs. *llm.Client
|
|
// satisfies it; tests use a fake. It is the SAME surface the summarizer uses.
|
|
type Completer interface {
|
|
Complete(ctx context.Context, system, user string) (string, error)
|
|
}
|
|
|
|
// Turn is one completed exchange in an ephemeral, session-only conversation
|
|
// (ADR-027 v1: nothing is persisted).
|
|
type Turn struct {
|
|
Question string
|
|
Answer string
|
|
}
|
|
|
|
// Request is one chat turn: the chosen model, the stored transcript text, the
|
|
// prior turns (for multi-turn context within the session), and the new question.
|
|
type Request struct {
|
|
Model string
|
|
Transcript string
|
|
History []Turn
|
|
Question string
|
|
}
|
|
|
|
// Reply is the model's answer plus whether the transcript was bounded to fit the
|
|
// model context (so the UI can be honest that an answer about the tail of a long
|
|
// video may be incomplete).
|
|
type Reply struct {
|
|
Answer string
|
|
Truncated bool
|
|
}
|
|
|
|
// Service answers questions against a stored transcript via a switchable set of
|
|
// models. models is the ordered, local-first list offered to the user (the cloud
|
|
// model is simply absent when disabled — see cmd wiring); maxChars bounds the
|
|
// transcript sent to any model (0 = unbounded). newClient builds a Completer for
|
|
// a chosen model alias (the same gateway, a different alias).
|
|
type Service struct {
|
|
newClient func(model string) Completer
|
|
models []string
|
|
maxChars int
|
|
}
|
|
|
|
// New constructs a Service. models must be non-empty and already filtered to the
|
|
// offerable set (cloud excluded when disabled) and de-duplicated by the caller.
|
|
func New(newClient func(model string) Completer, models []string, maxChars int) *Service {
|
|
return &Service{newClient: newClient, models: models, maxChars: maxChars}
|
|
}
|
|
|
|
// Models returns a copy of the offerable model list (local-first order).
|
|
func (s *Service) Models() []string {
|
|
out := make([]string, len(s.models))
|
|
copy(out, s.models)
|
|
return out
|
|
}
|
|
|
|
// offers reports whether model is in the offerable set — the guard that keeps an
|
|
// arbitrary, un-offered alias (e.g. a forged form value) from reaching the gateway.
|
|
func (s *Service) offers(model string) bool {
|
|
for _, m := range s.models {
|
|
if m == model {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// DefaultModel resolves the model a fresh chat opens with: the summary's own
|
|
// model when it is still an offered option (the ADR-027 default — chat continues
|
|
// in the model that produced the summary), otherwise the first offered model.
|
|
// Returns "" only when no models are configured.
|
|
func (s *Service) DefaultModel(summaryModel string) string {
|
|
if summaryModel != "" && s.offers(summaryModel) {
|
|
return summaryModel
|
|
}
|
|
if len(s.models) > 0 {
|
|
return s.models[0]
|
|
}
|
|
return ""
|
|
}
|
|
|
|
// Answer runs one chat turn. The model is forced back to a default if the request
|
|
// names an un-offered alias, so chat can never call the gateway with an arbitrary
|
|
// model. The transcript is truncated up front (reporting whether it was cut) and
|
|
// passed as system context; the running conversation is the user message.
|
|
func (s *Service) Answer(ctx context.Context, req Request) (Reply, error) {
|
|
if len(s.models) == 0 {
|
|
return Reply{}, fmt.Errorf("chat: no models configured")
|
|
}
|
|
model := req.Model
|
|
if !s.offers(model) {
|
|
model = s.DefaultModel("")
|
|
}
|
|
|
|
transcript, truncated := truncate(req.Transcript, s.maxChars)
|
|
system := buildSystem(transcript, truncated)
|
|
user := buildUser(req.History, req.Question)
|
|
|
|
start := time.Now()
|
|
out, err := s.newClient(model).Complete(ctx, system, user)
|
|
if err != nil {
|
|
return Reply{}, fmt.Errorf("chat: %s: %w", model, err)
|
|
}
|
|
dur := time.Since(start)
|
|
metrics.ObserveChat(model, dur)
|
|
slog.Default().Info("chat answer", "model", model, "elapsed_ms", dur.Milliseconds())
|
|
answer := strings.TrimSpace(out)
|
|
if answer == "" {
|
|
return Reply{}, fmt.Errorf("chat: %s returned an empty answer", model)
|
|
}
|
|
return Reply{Answer: answer, Truncated: truncated}, nil
|
|
}
|
|
|
|
const systemPreamble = `You are Tapir, answering questions about ONE video using ONLY the transcript below.
|
|
Ground every answer in the transcript. If the transcript does not contain the answer, say so plainly rather than guessing.`
|
|
|
|
const truncatedNote = `
|
|
The transcript below is truncated to fit the model — if a question seems to concern something missing, note it may be beyond the available portion.`
|
|
|
|
// buildSystem frames the model as a transcript-grounded QA assistant and embeds
|
|
// the (possibly truncated) transcript as context.
|
|
func buildSystem(transcript string, truncated bool) string {
|
|
var b strings.Builder
|
|
b.WriteString(systemPreamble)
|
|
if truncated {
|
|
b.WriteString(truncatedNote)
|
|
}
|
|
b.WriteString("\n\nTranscript:\n")
|
|
b.WriteString(transcript)
|
|
return b.String()
|
|
}
|
|
|
|
// buildUser renders the running conversation as the user message: prior turns as
|
|
// Q/A pairs followed by the new question. Folding history into one message keeps
|
|
// the Completer surface (a single system+user call) unchanged — no new llm method.
|
|
func buildUser(history []Turn, question string) string {
|
|
var b strings.Builder
|
|
for _, t := range history {
|
|
fmt.Fprintf(&b, "Q: %s\nA: %s\n\n", t.Question, t.Answer)
|
|
}
|
|
fmt.Fprintf(&b, "Q: %s", question)
|
|
return b.String()
|
|
}
|
|
|
|
// truncate caps content to max bytes on a UTF-8 rune boundary, reporting whether
|
|
// it cut. It mirrors the summarizer's truncation discipline (ADR-022) but returns
|
|
// the cut flag so the chat UI can be honest about a bounded transcript. A
|
|
// non-positive max (or content already within budget) returns content unchanged.
|
|
func truncate(content string, max int) (string, bool) {
|
|
if max <= 0 || len(content) <= max {
|
|
return content, false
|
|
}
|
|
cut := max
|
|
for cut > 0 && !utf8.RuneStart(content[cut]) {
|
|
cut--
|
|
}
|
|
return content[:cut], true
|
|
}
|