feat(youtube): acquire captions via player/timedtext baseUrl (ADR-010)
CI / Lint / Test / Vet (push) Successful in 6s
CI / Build & Import (push) Failing after 1s
CI / Mirror to GitHub (push) Has been skipped

The Data API captions.download endpoint is owner-only: every subscription
video the user does not own returned HTTP 403, producing 0 summaries and a
~150-line error spew in the first live Stage-0 run. Captions-first (ADR-007)
is sound; only the acquisition mechanism was wrong.

FetchTranscript now resolves caption tracks from the InnerTube player
response (ANDROID client, unauthenticated) and GETs the chosen track's
timedtext baseUrl with a plain http.Client — no OAuth token, which can break
the endpoint. The srv3 XML, json3, and legacy <transcript> formats all parse;
non-asr tracks in a preferred language win. Watch-page ytInitialPlayerResponse
scrape is the fallback when InnerTube returns no tracks.

Degrade, don't error (explicit quick-fix): no captionTracks, empty baseUrl, a
non-200 fetch, or an unparseable body yield Source=none, not an error. Only
genuine transport faults error — this kills the spew. OAuth stays on
ListSubscriptions/NewVideos (Data API); only transcript fetch goes unauthed.

Validated live from koala: the ANDROID client returned working baseUrls and
real transcript text for public videos the run identity does not own.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
2026-06-02 23:08:13 +02:00
co-authored by Claude Opus 4.8
parent fa16a62e6d
commit 8b14ef4add
4 changed files with 530 additions and 127 deletions
+350 -93
View File
@@ -1,150 +1,365 @@
package youtube
import (
"bytes"
"context"
"encoding/json"
"encoding/xml"
"fmt"
"io"
"net/http"
"net/url"
"regexp"
"strings"
"gitea.d-ma.be/mathias/tapir/internal/domain"
)
// FetchTranscript resolves a transcript captions-first (ADR-007):
//
// - List the video's caption tracks. No track => domain.SourceNone, no error.
// - Select a track (preferred language first, else the first track) and
// download it as WebVTT, stripping cue timing/markup to plain text.
//
// Audio download and speech-to-text are deliberately absent: no captions means
// SourceNone, full stop.
func (a *Adapter) FetchTranscript(ctx context.Context, v domain.Video) (domain.Transcript, error) {
client, err := a.httpClient(ctx, a.cfg.TokenSecretRef)
if err != nil {
return domain.Transcript{}, err
}
// defaultPlayerBaseURL is the InnerTube / watch-page host. Overridable via
// Config.PlayerBaseURL (tests point it at an httptest server).
const defaultPlayerBaseURL = "https://www.youtube.com"
tracks, err := a.listCaptionTracks(ctx, client, v.ProviderVideoID)
// androidUserAgent identifies the InnerTube ANDROID client. ANDROID historically
// returns caption baseUrls without a PoToken requirement (ADR-010).
const androidUserAgent = "com.google.android.youtube/20.10.38 (Linux; U; Android 11) gzip"
// maxCaptionBytes bounds a single timedtext/player body read.
const maxCaptionBytes = 16 << 20 // 16 MiB
// FetchTranscript resolves a transcript captions-first (ADR-007) via the player
// response + timedtext baseUrl, unauthenticated (ADR-010 — Data API
// captions.download is owner-only and 403s for videos the user does not own):
//
// - POST the InnerTube player endpoint (ANDROID client) and read
// captions.playerCaptionsTracklistRenderer.captionTracks[]. None => SourceNone.
// - Select a track (preferred language, non-asr first) and GET its baseUrl with
// a plain http.Client (no OAuth token — it can break the timedtext endpoint),
// parsing srv3 XML / json3 / legacy XML to plain text.
//
// Degrade, don't error (ADR-010): no captionTracks, empty baseUrl, a non-200
// fetch, or an unparseable body all yield SourceNone rather than an error. Only
// genuine transport (network) faults return an error. Audio download and
// speech-to-text remain absent (ADR-007).
func (a *Adapter) FetchTranscript(ctx context.Context, v domain.Video) (domain.Transcript, error) {
client := a.plainClient()
tracks, err := a.captionTracks(ctx, client, v.ProviderVideoID)
if err != nil {
return domain.Transcript{}, fmt.Errorf("list captions for %q: %w", v.ProviderVideoID, err)
return domain.Transcript{}, fmt.Errorf("resolve caption tracks for %q: %w", v.ProviderVideoID, err)
}
track, ok := a.selectTrack(tracks)
if !ok {
// No usable caption track: a recorded "checked, none available", not an error.
return domain.Transcript{
VideoID: v.ID,
UserID: v.UserID,
Source: domain.SourceNone,
}, nil
if !ok || strings.TrimSpace(track.BaseURL) == "" {
return noTranscript(v), nil
}
raw, err := a.getRaw(ctx, client, "/captions/"+track.ID, url.Values{"tfmt": {"vtt"}})
raw, status, err := a.httpGet(ctx, client, track.BaseURL, nil)
if err != nil {
return domain.Transcript{}, fmt.Errorf("download caption track %q: %w", track.ID, err)
return domain.Transcript{}, fmt.Errorf("download caption track for %q: %w", v.ProviderVideoID, err)
}
if status != http.StatusOK {
// Owner-only 403, region/age gate, or transient unavailability: not an error.
return noTranscript(v), nil
}
text := vttToText(string(raw))
text := timedtextToText(string(raw))
if text == "" {
// Track existed but carried no text: treat as no usable transcript.
return domain.Transcript{
VideoID: v.ID,
UserID: v.UserID,
Source: domain.SourceNone,
}, nil
return noTranscript(v), nil
}
return domain.Transcript{
VideoID: v.ID,
UserID: v.UserID,
Source: domain.SourceCaptions,
Language: track.Snippet.Language,
Language: track.LanguageCode,
Content: text,
}, nil
}
func (a *Adapter) listCaptionTracks(ctx context.Context, client *http.Client, videoID string) ([]captionTrack, error) {
q := url.Values{
"part": {"snippet"},
"videoId": {videoID},
}
var resp captionListResponse
if err := a.getJSON(ctx, client, "/captions", q, &resp); err != nil {
return nil, err
}
return resp.Items, nil
// noTranscript is the recorded "checked, none usable" result — not an error.
func noTranscript(v domain.Video) domain.Transcript {
return domain.Transcript{VideoID: v.ID, UserID: v.UserID, Source: domain.SourceNone}
}
// selectTrack picks the best caption track: the first track whose language
// matches a configured preferred language, else the first track. Returns ok ==
// false when there are no tracks at all.
// captionTracks resolves a video's caption tracks from the player response. It
// tries the InnerTube ANDROID client first and falls back to scraping
// ytInitialPlayerResponse from the watch page. A nil slice (no tracks) is a
// degrade, not an error; only transport faults return an error.
func (a *Adapter) captionTracks(ctx context.Context, client *http.Client, videoID string) ([]captionTrack, error) {
tracks, err := a.playerCaptionTracks(ctx, client, videoID)
if err != nil {
return nil, err
}
if len(tracks) > 0 {
return tracks, nil
}
// Fallback: scrape the watch page (no PoToken/InnerTube context needed).
return a.scrapeCaptionTracks(ctx, client, videoID)
}
// playerCaptionTracks POSTs the InnerTube player endpoint with the ANDROID client
// context (unauthenticated) and returns its caption tracks.
func (a *Adapter) playerCaptionTracks(ctx context.Context, client *http.Client, videoID string) ([]captionTrack, error) {
reqBody, err := json.Marshal(playerRequest{
Context: innertubeContext{Client: innertubeClient{
ClientName: "ANDROID",
ClientVersion: "20.10.38",
AndroidSDKVersion: 30,
HL: "en",
GL: "US",
}},
VideoID: videoID,
})
if err != nil {
return nil, fmt.Errorf("encode player request: %w", err)
}
body, status, err := a.httpPost(ctx, client, a.playerBaseURL+"/youtubei/v1/player", reqBody)
if err != nil {
return nil, err
}
if status != http.StatusOK {
return nil, nil // degrade
}
var resp playerResponse
if err := json.Unmarshal(body, &resp); err != nil {
return nil, nil // unparseable => degrade
}
return resp.Captions.PlayerCaptionsTracklistRenderer.CaptionTracks, nil
}
// ytInitialMarker locates the embedded player JSON on the watch page.
const ytInitialMarker = "ytInitialPlayerResponse"
// scrapeCaptionTracks fetches the watch page and extracts caption tracks from the
// embedded ytInitialPlayerResponse JSON. Any failure degrades to no tracks.
func (a *Adapter) scrapeCaptionTracks(ctx context.Context, client *http.Client, videoID string) ([]captionTrack, error) {
body, status, err := a.httpGet(ctx, client, a.playerBaseURL+"/watch?v="+videoID, map[string]string{
"User-Agent": androidUserAgent,
})
if err != nil {
return nil, err
}
if status != http.StatusOK {
return nil, nil
}
obj, ok := extractJSONObject(body, ytInitialMarker)
if !ok {
return nil, nil
}
var resp playerResponse
if err := json.Unmarshal(obj, &resp); err != nil {
return nil, nil
}
return resp.Captions.PlayerCaptionsTracklistRenderer.CaptionTracks, nil
}
// selectTrack picks the best caption track: a preferred-language non-asr track
// first, then a preferred-language asr track, then any non-asr track, else the
// first track. Returns ok == false when there are no tracks at all.
func (a *Adapter) selectTrack(tracks []captionTrack) (captionTrack, bool) {
if len(tracks) == 0 {
return captionTrack{}, false
}
for _, pref := range a.cfg.PreferredLanguages {
for _, t := range tracks {
if strings.EqualFold(t.Snippet.Language, pref) {
return t, true
for _, asr := range []bool{false, true} {
for _, pref := range a.cfg.PreferredLanguages {
for _, t := range tracks {
if t.isASR() == asr && matchLang(t.LanguageCode, pref) {
return t, true
}
}
}
}
for _, t := range tracks {
if !t.isASR() {
return t, true
}
}
return tracks[0], true
}
// --- caption response shapes ------------------------------------------------
// matchLang matches a track language against a preferred code, tolerating region
// suffixes (preferred "en" matches "en", "en-US", "en-GB").
func matchLang(code, pref string) bool {
if strings.EqualFold(code, pref) {
return true
}
return strings.HasPrefix(strings.ToLower(code), strings.ToLower(pref)+"-")
}
type captionListResponse struct {
Items []captionTrack `json:"items"`
// --- HTTP (plain, unauthenticated) ------------------------------------------
// plainClient returns an unauthenticated HTTP client. No OAuth token is attached:
// the player/timedtext endpoints can reject authenticated requests (ADR-010).
// Tests inject the httptest transport via a.transport.
func (a *Adapter) plainClient() *http.Client {
if a.transport != nil {
return &http.Client{Transport: a.transport}
}
return &http.Client{}
}
func (a *Adapter) httpGet(ctx context.Context, client *http.Client, url string, headers map[string]string) ([]byte, int, error) {
return a.httpDo(ctx, client, http.MethodGet, url, nil, headers)
}
func (a *Adapter) httpPost(ctx context.Context, client *http.Client, url string, body []byte) ([]byte, int, error) {
return a.httpDo(ctx, client, http.MethodPost, url, body, map[string]string{
"Content-Type": "application/json",
"User-Agent": androidUserAgent,
})
}
// httpDo issues a request and returns (body, status, err). A non-nil err is a
// genuine transport fault; a non-200 status is returned to the caller to decide
// (callers degrade rather than error per ADR-010).
func (a *Adapter) httpDo(ctx context.Context, client *http.Client, method, url string, body []byte, headers map[string]string) ([]byte, int, error) {
var rdr io.Reader
if body != nil {
rdr = bytes.NewReader(body)
}
req, err := http.NewRequestWithContext(ctx, method, url, rdr)
if err != nil {
return nil, 0, fmt.Errorf("build %s %s: %w", method, url, err)
}
for k, v := range headers {
req.Header.Set(k, v)
}
resp, err := client.Do(req)
if err != nil {
return nil, 0, fmt.Errorf("%s %s: %w", method, url, err)
}
defer func() { _ = resp.Body.Close() }()
b, err := io.ReadAll(io.LimitReader(resp.Body, maxCaptionBytes))
if err != nil {
return nil, 0, fmt.Errorf("read %s %s body: %w", method, url, err)
}
return b, resp.StatusCode, nil
}
// --- player / caption response shapes ---------------------------------------
type playerRequest struct {
Context innertubeContext `json:"context"`
VideoID string `json:"videoId"`
}
type innertubeContext struct {
Client innertubeClient `json:"client"`
}
type innertubeClient struct {
ClientName string `json:"clientName"`
ClientVersion string `json:"clientVersion"`
AndroidSDKVersion int `json:"androidSdkVersion,omitempty"`
HL string `json:"hl"`
GL string `json:"gl"`
}
type playerResponse struct {
Captions struct {
PlayerCaptionsTracklistRenderer struct {
CaptionTracks []captionTrack `json:"captionTracks"`
} `json:"playerCaptionsTracklistRenderer"`
} `json:"captions"`
}
type captionTrack struct {
ID string `json:"id"`
Snippet struct {
Language string `json:"language"`
TrackKind string `json:"trackKind"` // "standard" | "ASR" | "forced"
Name string `json:"name"`
Status string `json:"status"`
} `json:"snippet"`
BaseURL string `json:"baseUrl"`
LanguageCode string `json:"languageCode"`
Kind string `json:"kind"` // "asr" for auto-generated
}
// --- WebVTT -> plain text ---------------------------------------------------
func (t captionTrack) isASR() bool { return strings.EqualFold(t.Kind, "asr") }
var (
vttCueTime = regexp.MustCompile(`-->`)
vttTag = regexp.MustCompile(`<[^>]+>`) // inline <00:00:01.000>, <c> markup
vttIndex = regexp.MustCompile(`^\d+$`) // SRT-style numeric cue index
vttSetting = regexp.MustCompile(`^(NOTE|STYLE|REGION)`) // VTT block headers
)
// --- timedtext -> plain text ------------------------------------------------
// vttToText reduces a WebVTT (or SRT-ish) caption file to plain transcript text:
// the WEBVTT header, NOTE/STYLE blocks, cue-timing lines, numeric indices, and
// inline markup are dropped; consecutive duplicate lines (common in rolling
// auto-captions) are collapsed.
func vttToText(raw string) string {
raw = strings.ReplaceAll(raw, "\r\n", "\n")
// timedtextToText reduces a timedtext caption body to plain transcript text. It
// auto-detects the format: json3 (a JSON object), else XML (srv3 <p> cues or the
// legacy <transcript><text> form). Whitespace within a cue is normalised and
// consecutive duplicate lines (common in rolling auto-captions) are collapsed.
func timedtextToText(raw string) string {
trimmed := strings.TrimSpace(raw)
if strings.HasPrefix(trimmed, "{") {
return collapse(json3Lines(trimmed))
}
return collapse(xmlLines(trimmed))
}
type json3Body struct {
Events []struct {
Segs []struct {
Utf8 string `json:"utf8"`
} `json:"segs"`
} `json:"events"`
}
func json3Lines(raw string) []string {
var doc json3Body
if err := json.Unmarshal([]byte(raw), &doc); err != nil {
return nil
}
var lines []string
for _, ev := range doc.Events {
var b strings.Builder
for _, s := range ev.Segs {
b.WriteString(s.Utf8)
}
if line := normalize(b.String()); line != "" {
lines = append(lines, line)
}
}
return lines
}
type timedtextXML struct {
Ps []struct {
Chardata string `xml:",chardata"`
Segs []struct {
Chardata string `xml:",chardata"`
} `xml:"s"`
} `xml:"body>p"`
// Legacy format: <transcript><text start dur>...</text></transcript>.
Texts []string `xml:"text"`
}
func xmlLines(raw string) []string {
var doc timedtextXML
if err := xml.Unmarshal([]byte(raw), &doc); err != nil {
return nil
}
var lines []string
for _, p := range doc.Ps {
var b strings.Builder
b.WriteString(p.Chardata)
for _, s := range p.Segs {
b.WriteString(s.Chardata)
}
if line := normalize(b.String()); line != "" {
lines = append(lines, line)
}
}
if len(lines) == 0 {
for _, t := range doc.Texts {
if line := normalize(t); line != "" {
lines = append(lines, line)
}
}
}
return lines
}
// normalize collapses internal whitespace (incl. intra-cue newlines) to single
// spaces and trims. encoding/xml and encoding/json already decode entities.
func normalize(s string) string {
return strings.Join(strings.Fields(s), " ")
}
// collapse drops consecutive duplicate lines and joins with newlines.
func collapse(lines []string) string {
var out []string
var prev string
for _, line := range strings.Split(raw, "\n") {
line = strings.TrimSpace(line)
if line == "" {
continue
}
if strings.HasPrefix(line, "WEBVTT") {
continue
}
if vttSetting.MatchString(line) {
continue
}
if vttCueTime.MatchString(line) {
continue
}
if vttIndex.MatchString(line) {
continue
}
line = strings.TrimSpace(vttTag.ReplaceAllString(line, ""))
if line == "" || line == prev {
for _, line := range lines {
if line == prev {
continue
}
out = append(out, line)
@@ -152,3 +367,45 @@ func vttToText(raw string) string {
}
return strings.Join(out, "\n")
}
// extractJSONObject finds marker in body and returns the first balanced JSON
// object that follows it (brace-matched, string-aware).
func extractJSONObject(body []byte, marker string) ([]byte, bool) {
i := bytes.Index(body, []byte(marker))
if i < 0 {
return nil, false
}
s := body[i+len(marker):]
j := bytes.IndexByte(s, '{')
if j < 0 {
return nil, false
}
s = s[j:]
depth, inStr, esc := 0, false, false
for k := 0; k < len(s); k++ {
c := s[k]
if inStr {
switch {
case esc:
esc = false
case c == '\\':
esc = true
case c == '"':
inStr = false
}
continue
}
switch c {
case '"':
inStr = true
case '{':
depth++
case '}':
depth--
if depth == 0 {
return s[:k+1], true
}
}
}
return nil, false
}