feat(discovery): drop Shorts and livestreams before the caption fetch (ADR-023)
CI / Lint / Test / Vet (push) Successful in 12s
CI / Build & Import (push) Successful in 10s

The scarce resource is the per-IP timedtext caption fetch (ADR-014); the pilot's
candidate set was mostly Shorts/clips/livestreams, each burning a fetch (a "none"
result is a completed fetch — it costs budget even when it yields nothing).

NewVideos now enriches candidates with one cheap Data API videos.list call
(contentDetails.duration + snippet.liveBroadcastContent — the quota API, a
DIFFERENT limit from the timedtext 429) and drops, before returning: videos
shorter than TAPIR_MIN_VIDEO_SECONDS (default 60) and any live/upcoming
broadcast. Dropped videos are never persisted, so the list declutters too.

Degrade-open: MinVideoSeconds=0 disables it (no quota call); a videos.list error
returns candidates unfiltered so discovery never breaks on a metadata hiccup. The
paste-a-URL path (VideoByID) is not filtered — an explicit request is honoured.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
2026-06-10 19:34:45 +02:00
co-authored by Claude Opus 4.8
parent 1aa8a97f95
commit 9db06d8a63
7 changed files with 250 additions and 4 deletions
+106 -4
View File
@@ -66,6 +66,13 @@ type Config struct {
// poll. Zero means defaultMaxVideos.
MaxVideosPerSubscription int
// MinVideoSeconds drops videos shorter than this from discovery (Shorts/clips,
// ADR-023). NewVideos enriches candidates with a single cheap videos.list call
// (contentDetails.duration + snippet.liveBroadcastContent) and filters before
// returning, so the scarce caption-fetch budget is never spent on them. Live
// and upcoming broadcasts are dropped too. Zero disables the filter.
MinVideoSeconds int
// BaseURL overrides the Data API root. Empty means defaultBaseURL.
BaseURL string
@@ -242,7 +249,97 @@ func (a *Adapter) NewVideos(ctx context.Context, sub domain.Subscription) ([]dom
break
}
}
return videos, nil
// Drop Shorts/sub-minute clips and live/upcoming broadcasts before they ever
// reach the rate-limited caption path (ADR-023). One cheap videos.list call
// (quota API, not the timedtext throttle) supplies duration + live status.
return a.filterLowValue(ctx, client, videos), nil
}
// filterLowValue removes videos shorter than cfg.MinVideoSeconds and any live or
// upcoming broadcast, using a single videos.list lookup for duration +
// liveBroadcastContent. The filter is best-effort: if MinVideoSeconds is 0 (off)
// or the lookup fails, the input is returned unfiltered — discovery must not break
// because a metadata call hiccuped; the worst case is the pre-ADR-023 behaviour.
func (a *Adapter) filterLowValue(ctx context.Context, client *http.Client, videos []domain.Video) []domain.Video {
if a.cfg.MinVideoSeconds <= 0 || len(videos) == 0 {
return videos
}
ids := make([]string, 0, len(videos))
for _, v := range videos {
ids = append(ids, v.ProviderVideoID)
}
q := url.Values{
"part": {"contentDetails,snippet"},
"id": {strings.Join(ids, ",")},
}
var resp videoListResponse
if err := a.getJSON(ctx, client, "/videos", q, &resp); err != nil {
// Degrade open: keep the candidates rather than lose discovery.
return videos
}
type meta struct {
seconds int
live string
}
byID := make(map[string]meta, len(resp.Items))
for _, it := range resp.Items {
byID[it.ID] = meta{seconds: parseISO8601Seconds(it.ContentDetails.Duration), live: it.Snippet.LiveBroadcastContent}
}
kept := videos[:0]
for _, v := range videos {
m, ok := byID[v.ProviderVideoID]
if !ok {
kept = append(kept, v) // unknown metadata: keep, let the fetch decide
continue
}
if m.live != "" && m.live != "none" {
continue // live or upcoming broadcast
}
if m.seconds > 0 && m.seconds < a.cfg.MinVideoSeconds {
continue // Short / sub-threshold clip
}
kept = append(kept, v)
}
return kept
}
// parseISO8601Seconds parses an ISO 8601 duration as returned by the YouTube Data
// API (e.g. "PT1H2M3S", "PT45S", "PT3M") into seconds. Only the hour/minute/second
// components YouTube emits are handled; an unparseable or zero value returns 0,
// which the caller treats as "unknown" (not filtered on duration).
func parseISO8601Seconds(d string) int {
if !strings.HasPrefix(d, "PT") {
return 0
}
d = d[2:]
total, num := 0, 0
seen := false
for _, r := range d {
switch {
case r >= '0' && r <= '9':
num = num*10 + int(r-'0')
seen = true
case r == 'H':
total += num * 3600
num, seen = 0, false
case r == 'M':
total += num * 60
num, seen = 0, false
case r == 'S':
total += num
num, seen = 0, false
default:
return 0 // unexpected component (days/weeks) — treat as unknown
}
}
if seen {
return 0 // trailing digits without a unit: malformed
}
return total
}
// VideoByID fetches a single video's metadata (videos.list, snippet) for an
@@ -377,11 +474,16 @@ type playlistItemListResponse struct {
type videoListResponse struct {
Items []struct {
ID string `json:"id"`
Snippet struct {
Title string `json:"title"`
ChannelTitle string `json:"channelTitle"`
PublishedAt time.Time `json:"publishedAt"`
Title string `json:"title"`
ChannelTitle string `json:"channelTitle"`
PublishedAt time.Time `json:"publishedAt"`
LiveBroadcastContent string `json:"liveBroadcastContent"`
} `json:"snippet"`
ContentDetails struct {
Duration string `json:"duration"` // ISO 8601, e.g. "PT1M30S"
} `json:"contentDetails"`
} `json:"items"`
}
+85
View File
@@ -186,6 +186,91 @@ func TestNewVideosCapsAtMax(t *testing.T) {
}
}
// TestNewVideosFiltersShortsAndLive: with MinVideoSeconds set, discovery enriches
// candidates via videos.list and drops sub-threshold clips (Shorts) and
// live/upcoming broadcasts before they reach the rate-limited caption path.
func TestNewVideosFiltersShortsAndLive(t *testing.T) {
a, _ := newTestAdapter(t, func(w http.ResponseWriter, r *http.Request) {
switch r.URL.Path {
case "/playlistItems":
_, _ = w.Write([]byte(`{
"items": [
{"snippet": {"title": "Real Talk", "publishedAt": "2026-06-03T10:00:00Z", "resourceId": {"videoId": "long1"}}},
{"snippet": {"title": "A Short", "publishedAt": "2026-06-03T09:00:00Z", "resourceId": {"videoId": "short1"}}},
{"snippet": {"title": "Live Now", "publishedAt": "2026-06-03T08:00:00Z", "resourceId": {"videoId": "live1"}}}
]
}`))
case "/videos":
if got := r.URL.Query().Get("part"); got != "contentDetails,snippet" {
t.Errorf("videos.list part=%q, want contentDetails,snippet", got)
}
_, _ = w.Write([]byte(`{
"items": [
{"id": "long1", "contentDetails": {"duration": "PT12M30S"}, "snippet": {"liveBroadcastContent": "none"}},
{"id": "short1", "contentDetails": {"duration": "PT45S"}, "snippet": {"liveBroadcastContent": "none"}},
{"id": "live1", "contentDetails": {"duration": "PT0S"}, "snippet": {"liveBroadcastContent": "live"}}
]
}`))
default:
t.Errorf("unexpected path %q", r.URL.Path)
}
})
a.cfg.MinVideoSeconds = 60
vids, err := a.NewVideos(context.Background(), domain.Subscription{ID: "s1", UserID: "u1", ChannelID: "UC_acme"})
if err != nil {
t.Fatalf("NewVideos: %v", err)
}
if len(vids) != 1 || vids[0].ProviderVideoID != "long1" {
t.Fatalf("expected only long1 to survive the filter, got %+v", vids)
}
}
// TestNewVideosNoFilterWhenDisabled: MinVideoSeconds=0 keeps the pre-ADR-023
// behaviour — no videos.list call, no filtering.
func TestNewVideosNoFilterWhenDisabled(t *testing.T) {
a, _ := newTestAdapter(t, func(w http.ResponseWriter, r *http.Request) {
if r.URL.Path == "/videos" {
t.Errorf("videos.list must not be called when MinVideoSeconds is 0")
}
_, _ = w.Write([]byte(`{"items": [
{"snippet": {"title": "A Short", "publishedAt": "2026-06-03T09:00:00Z", "resourceId": {"videoId": "short1"}}}
]}`))
})
a.cfg.MinVideoSeconds = 0
vids, err := a.NewVideos(context.Background(), domain.Subscription{ID: "s1", UserID: "u1", ChannelID: "UC_acme"})
if err != nil {
t.Fatalf("NewVideos: %v", err)
}
if len(vids) != 1 {
t.Fatalf("filter disabled must keep all videos, got %d", len(vids))
}
}
func TestParseISO8601Seconds(t *testing.T) {
cases := []struct {
in string
want int
}{
{"PT45S", 45},
{"PT1M30S", 90},
{"PT3M", 180},
{"PT1H2M3S", 3723},
{"PT2H", 7200},
{"PT0S", 0},
{"", 0},
{"garbage", 0},
{"P1D", 0}, // days component not handled → unknown
{"PT10", 0}, // trailing digits without a unit → malformed
}
for _, c := range cases {
if got := parseISO8601Seconds(c.in); got != c.want {
t.Errorf("parseISO8601Seconds(%q) = %d, want %d", c.in, got, c.want)
}
}
}
// TestUploadsPlaylistID covers the zero-cost UC->UU derivation, including
// non-standard ids that must fall through unchanged (handled via fallback).
func TestUploadsPlaylistID(t *testing.T) {
+16
View File
@@ -46,6 +46,12 @@ type Config struct {
// MaxTranscriptChars bounds the transcript text sent to the model so a long
// transcript does not overflow a small-context primary. 0 disables truncation.
MaxTranscriptChars int
// MinVideoSeconds drops videos shorter than this from discovery (Shorts and
// other sub-minute clips that are noise and waste the scarce caption-fetch
// budget, ADR-014/ADR-023). Enforced via a cheap Data API videos.list lookup at
// discovery, never the rate-limited caption path. 0 disables the filter.
MinVideoSeconds int
// SummarizerTimeout bounds a single completion call. Thinking models are
// slow, so the default is generous.
SummarizerTimeout time.Duration
@@ -138,6 +144,7 @@ const (
defaultCloudFallbackModel = "berget/mistral-small"
defaultSummaryMaxTokens = 1500
defaultMaxTranscriptChars = 18000
defaultMinVideoSeconds = 60
defaultSummarizerTimeout = 5 * time.Minute
defaultYTTokenRef = "youtube/refresh_token"
defaultYTConnectRedirectURL = "https://tapir.d-ma.be/oauth/youtube/callback"
@@ -230,6 +237,15 @@ func Load() (Config, error) {
}
c.MaxTranscriptChars = maxChars
minVideo, err := intOr("TAPIR_MIN_VIDEO_SECONDS", defaultMinVideoSeconds)
if err != nil {
return Config{}, err
}
if minVideo < 0 {
minVideo = 0
}
c.MinVideoSeconds = minVideo
onboard, err := intOr("TAPIR_ONBOARD_SUMMARIZE_COUNT", defaultOnboardSummarizeCount)
if err != nil {
return Config{}, err