Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c5f556d1d6 | ||
|
|
a884e7e9c5 | ||
|
|
27fd33c99c | ||
|
|
2c96926ff7 |
@@ -60,8 +60,14 @@ These caused real mistakes that were caught and corrected; the corrections are l
|
|||||||
stores, and sinks are adapters. Adding a video provider or a sink = a new adapter implementing
|
stores, and sinks are adapters. Adding a video provider or a sink = a new adapter implementing
|
||||||
the interface, nothing in the engine changes. This is what keeps "standalone vs homelab" a
|
the interface, nothing in the engine changes. This is what keeps "standalone vs homelab" a
|
||||||
wiring choice (ADR-003).
|
wiring choice (ADR-003).
|
||||||
- **BDD.** The `docs/use-cases/*.feature` files are the behavior spec. New behavior gets a
|
- **BDD.** The `docs/use-cases/*.feature` files are the behavior spec (design records — there is
|
||||||
scenario; the use-case core is tested through fake adapters, not live YouTube/brain.
|
no godog runner). New behavior gets a scenario; the use-case core is tested through fake
|
||||||
|
adapters, not live YouTube/brain. A name-coverage gate (`test/acceptance/scenario_coverage_test.go`,
|
||||||
|
`TestScenarioCoverage`) keeps the two from drifting: every non-`@pending` scenario must be
|
||||||
|
mapped to an existing Go test in `scenarioCoverage`. When you add a scenario, either map it to
|
||||||
|
its covering test or tag it `@pending` in the `.feature` with a one-line reason. It checks the
|
||||||
|
*link*, not that the test exercises the scenario — that's the deliberate trade for not running
|
||||||
|
godog (see issue #5 / the BDD-runner decision).
|
||||||
|
|
||||||
## Skills (engineering discipline)
|
## Skills (engineering discipline)
|
||||||
|
|
||||||
@@ -79,7 +85,7 @@ Skills live in the canonical library `mathias/skills` and are wired into this re
|
|||||||
|
|
||||||
## Current build state (start here for the first task)
|
## Current build state (start here for the first task)
|
||||||
|
|
||||||
The repo is **green and shipping** — last tag `v0.4.0`. `task check` passes (fmt, vet, lint,
|
The repo is **green and shipping** — last tag `v0.8.0`. `task check` passes (fmt, vet, lint,
|
||||||
`go test -p 1 ./...`). Go is `1.26.1` (see `go.mod`).
|
`go test -p 1 ./...`). Go is `1.26.1` (see `go.mod`).
|
||||||
|
|
||||||
- Clean Architecture core is implemented: `internal/domain` (entities), `internal/ports`
|
- Clean Architecture core is implemented: `internal/domain` (entities), `internal/ports`
|
||||||
|
|||||||
@@ -75,8 +75,10 @@ password or Google); a new Dex subject is routed to `/register` to create a Tapi
|
|||||||
Stage 1 is multi-user: each user connects their own YouTube account from the browser and
|
Stage 1 is multi-user: each user connects their own YouTube account from the browser and
|
||||||
manages their own summaries under DB-enforced RLS isolation. When `TAPIR_DISCOVERY_INTERVAL`
|
manages their own summaries under DB-enforced RLS isolation. When `TAPIR_DISCOVERY_INTERVAL`
|
||||||
is set (e.g. `2h`), the serve process runs a scheduled discovery pass for every registered
|
is set (e.g. `2h`), the serve process runs a scheduled discovery pass for every registered
|
||||||
user automatically — no CronJob required. See `docs/homelab-integration.md` for the full
|
user automatically — no CronJob required. In auto mode only videos published within
|
||||||
config reference.
|
`TAPIR_AUTO_SUMMARIZE_WINDOW` (default ~7d, ADR-020) are summarised automatically; older videos
|
||||||
|
are listed and summarised on demand, so a large back-catalogue doesn't keep re-driving the
|
||||||
|
caption rate gate. See `docs/homelab-integration.md` for the full config reference.
|
||||||
|
|
||||||
### Headless on koala
|
### Headless on koala
|
||||||
|
|
||||||
|
|||||||
+16
-2
@@ -49,10 +49,12 @@ func buildUserRunner(cfg config.Config, st *store.Store, secretStore ports.Secre
|
|||||||
runner.WithAutoWindow(cfg.AutoSummarizeWindow)), nil
|
runner.WithAutoWindow(cfg.AutoSummarizeWindow)), nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// userLister enumerates every registered user. *store.Store satisfies it via
|
// userLister enumerates every registered user and reports a user's video
|
||||||
// ListAllUsers. A small local interface keeps the scheduler testable with a fake.
|
// connections. *store.Store satisfies it via ListAllUsers + ConnectionsForUser.
|
||||||
|
// A small local interface keeps the scheduler testable with a fake.
|
||||||
type userLister interface {
|
type userLister interface {
|
||||||
ListAllUsers(ctx context.Context) ([]store.UserIdentity, error)
|
ListAllUsers(ctx context.Context) ([]store.UserIdentity, error)
|
||||||
|
ConnectionsForUser(ctx context.Context, userID string) ([]store.Connection, error)
|
||||||
}
|
}
|
||||||
|
|
||||||
// runDiscoveryPass runs one discovery pass for every user. runUser performs a
|
// runDiscoveryPass runs one discovery pass for every user. runUser performs a
|
||||||
@@ -78,6 +80,18 @@ func runDiscoveryPass(
|
|||||||
if ctx.Err() != nil {
|
if ctx.Err() != nil {
|
||||||
break // shutting down: stop enumerating
|
break // shutting down: stop enumerating
|
||||||
}
|
}
|
||||||
|
// Skip users with no video connection. A discovery pass for them only
|
||||||
|
// attempts to resolve a token that was never minted, logging a spurious
|
||||||
|
// "ref not found" every tick (e.g. stale Dex-era orphan identities).
|
||||||
|
conns, err := lister.ConnectionsForUser(ctx, u.UserID)
|
||||||
|
if err != nil {
|
||||||
|
log.Warn("scheduler: list connections failed", "user", u.UserID, "err", err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if len(conns) == 0 {
|
||||||
|
log.Debug("scheduler: skipping user with no video connections", "user", u.UserID)
|
||||||
|
continue
|
||||||
|
}
|
||||||
stats, err := runUser(ctx, u.UserID)
|
stats, err := runUser(ctx, u.UserID)
|
||||||
total = sumStats(total, stats)
|
total = sumStats(total, stats)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
|
|||||||
@@ -21,14 +21,25 @@ func quietLog() *slog.Logger {
|
|||||||
|
|
||||||
// fakeLister returns a fixed user set (or an error) for the scheduler under test.
|
// fakeLister returns a fixed user set (or an error) for the scheduler under test.
|
||||||
type fakeLister struct {
|
type fakeLister struct {
|
||||||
users []store.UserIdentity
|
users []store.UserIdentity
|
||||||
err error
|
err error
|
||||||
|
noConn map[string]bool // users that have NOT connected a video source
|
||||||
}
|
}
|
||||||
|
|
||||||
func (f fakeLister) ListAllUsers(context.Context) ([]store.UserIdentity, error) {
|
func (f fakeLister) ListAllUsers(context.Context) ([]store.UserIdentity, error) {
|
||||||
return f.users, f.err
|
return f.users, f.err
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ConnectionsForUser reports a single youtube connection for every user except
|
||||||
|
// those in noConn, which return zero — the connection-less case the scheduler
|
||||||
|
// must skip instead of running (and failing to resolve a token for).
|
||||||
|
func (f fakeLister) ConnectionsForUser(_ context.Context, userID string) ([]store.Connection, error) {
|
||||||
|
if f.noConn[userID] {
|
||||||
|
return nil, nil
|
||||||
|
}
|
||||||
|
return []store.Connection{{Provider: "youtube"}}, nil
|
||||||
|
}
|
||||||
|
|
||||||
// countingRunUser records how many passes each user got, optionally failing for
|
// countingRunUser records how many passes each user got, optionally failing for
|
||||||
// specific users, under a mutex so it is safe across the scheduler goroutine.
|
// specific users, under a mutex so it is safe across the scheduler goroutine.
|
||||||
type countingRunUser struct {
|
type countingRunUser struct {
|
||||||
@@ -91,6 +102,21 @@ func TestDiscoveryPassRunsEveryUserOnce(t *testing.T) {
|
|||||||
require.Equal(t, 3, stats.Summarized, "stats are summed across users")
|
require.Equal(t, 3, stats.Summarized, "stats are summed across users")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestDiscoveryPassSkipsUsersWithoutConnections(t *testing.T) {
|
||||||
|
// b never connected a video source (e.g. a stale Dex-era orphan identity).
|
||||||
|
// It must be skipped silently — not run and logged as a token error every pass.
|
||||||
|
lister := fakeLister{users: usersN("a", "b", "c"), noConn: map[string]bool{"b": true}}
|
||||||
|
rc := newCountingRunUser()
|
||||||
|
|
||||||
|
stats := runDiscoveryPass(context.Background(), lister, rc.run, quietLog())
|
||||||
|
|
||||||
|
require.Equal(t, 1, rc.count("a"))
|
||||||
|
require.Equal(t, 0, rc.count("b"), "a user with no connection must be skipped, not run")
|
||||||
|
require.Equal(t, 1, rc.count("c"))
|
||||||
|
require.Equal(t, 2, stats.Summarized, "only connected users contribute")
|
||||||
|
require.Equal(t, 0, stats.Errors, "skipping is silent — no spurious error stat")
|
||||||
|
}
|
||||||
|
|
||||||
func TestDiscoveryPassOneUserFailureDoesNotStopOthers(t *testing.T) {
|
func TestDiscoveryPassOneUserFailureDoesNotStopOthers(t *testing.T) {
|
||||||
lister := fakeLister{users: usersN("a", "b", "c")}
|
lister := fakeLister{users: usersN("a", "b", "c")}
|
||||||
rc := newCountingRunUser("b") // user b's pass errors
|
rc := newCountingRunUser("b") // user b's pass errors
|
||||||
|
|||||||
@@ -147,9 +147,17 @@ graph TB
|
|||||||
rate-limiting — ADR-014).
|
rate-limiting — ADR-014).
|
||||||
- **Summarization mode** — `users.auto_summarize` (migration 006). Default is **true** for new
|
- **Summarization mode** — `users.auto_summarize` (migration 006). Default is **true** for new
|
||||||
users (migration 011, ADR-018); all existing rows were back-filled via migration 012. Auto:
|
users (migration 011, ADR-018); all existing rows were back-filled via migration 012. Auto:
|
||||||
every new video is summarized. Manual: new videos appear unsummarized; the button sets
|
new videos **published within the recency window** (`TAPIR_AUTO_SUMMARIZE_WINDOW`, default ~7d,
|
||||||
`videos.summarize_requested`, which the next `tapir run` processes and clears. Both the click
|
ADR-020) are summarized automatically; older videos are discovered and listed but wait for an
|
||||||
path and the batch `tapir run` drive the same unchanged engine.
|
explicit "Summarize". Manual: new videos appear unsummarized; the button sets
|
||||||
|
`videos.summarize_requested`, which the next `tapir run` processes and clears. A manual request
|
||||||
|
bypasses the recency bound. Both the click path and the batch `tapir run` drive the same
|
||||||
|
unchanged engine.
|
||||||
|
- **List surface (ADR-020)** — the list reads `ListVideos` ordered summarized-first, then
|
||||||
|
`published_at DESC NULLS LAST`. The web layer collapses the noise so summaries are not buried:
|
||||||
|
un-summarized videos older than the recency window fold into one "Show N older videos"
|
||||||
|
disclosure, and caption-less videos collapse to a single count line. Copy surfaces scarcity
|
||||||
|
honestly (queue counts, gradual-fill note) — it never implies the feed is fuller than it is.
|
||||||
|
|
||||||
The engine, ports, and sink adapters are **untouched** by all of the above — the web surface only
|
The engine, ports, and sink adapters are **untouched** by all of the above — the web surface only
|
||||||
reads the store and triggers the existing engine. Adding it changed wiring, not the core (ADR-003).
|
reads the store and triggers the existing engine. Adding it changed wiring, not the core (ADR-003).
|
||||||
@@ -174,6 +182,7 @@ sequenceDiagram
|
|||||||
loop per user
|
loop per user
|
||||||
S->>DB: GetAutoSummarize(userID)
|
S->>DB: GetAutoSummarize(userID)
|
||||||
S->>YT: ListSubscriptions + NewVideos
|
S->>YT: ListSubscriptions + NewVideos
|
||||||
|
Note over S,DB: auto: skip videos published before<br/>TAPIR_AUTO_SUMMARIZE_WINDOW (ADR-020);<br/>older ones listed, await manual request
|
||||||
Note over S,YT: WaitFetchGate(ctx) throttles<br/>all fetches to TAPIR_FETCH_RATE
|
Note over S,YT: WaitFetchGate(ctx) throttles<br/>all fetches to TAPIR_FETCH_RATE
|
||||||
alt transcript available
|
alt transcript available
|
||||||
S->>LLM: Summarize
|
S->>LLM: Summarize
|
||||||
@@ -214,19 +223,22 @@ Both paths share `globalFetchGate` — rate limiting is **respected in both**, n
|
|||||||
|
|
||||||
| Path | Trigger | Order | Rationale |
|
| Path | Trigger | Order | Rationale |
|
||||||
|------|---------|-------|-----------|
|
|------|---------|-------|-----------|
|
||||||
| **Foreground** | User clicks "Summarize now" on any non-summarized card (`POST /v/{id}/retry-now` for rate-limited; `POST /v/{id}/summarize` for pending) | Single chosen video | On-demand value: user picks a specific video to read now |
|
| **Foreground** | User clicks "Summarize" on any non-summarized card (`POST /v/{id}/retry-now` for rate-limited; `POST /v/{id}/summarize` for pending) | Single chosen video | On-demand value: user picks a specific video to read now — bypasses the recency bound |
|
||||||
| **Background batch** | Scheduled discovery pass every `TAPIR_DISCOVERY_INTERVAL` | **Newest-first across all channels** (see below) | Onboarding prioritisation: most recent, relevant videos surface first |
|
| **Background batch** | Scheduled discovery pass every `TAPIR_DISCOVERY_INTERVAL` | **Newest-first across all channels** (see below), **bounded to the recency window** (ADR-020) | Onboarding prioritisation within bounded load: recent videos auto-fill; the older back-catalogue stays on-demand |
|
||||||
|
|
||||||
The rationale for both paths is **onboarding prioritisation** — a new user should get summaries
|
The rationale for both paths is **onboarding prioritisation under an honest, bounded load** — a
|
||||||
of their most recent, relevant videos quickly while the older back-catalogue fills in behind,
|
new user gets summaries of their most recent videos automatically, while the older back-catalogue
|
||||||
all within the honest shared rate limit.
|
is listed but summarised only on demand, so it never re-drives the shared rate gate every cycle.
|
||||||
|
|
||||||
### Newest-first batch ordering (ADR-018)
|
### Newest-first batch ordering (ADR-018)
|
||||||
|
|
||||||
Within each scheduled pass, `RunOnce` uses a three-phase structure:
|
Within each scheduled pass, `RunOnce` uses a three-phase structure:
|
||||||
|
|
||||||
1. **Discover + persist**: walk all channels, `UpsertVideo` every candidate (so it appears in
|
1. **Discover + persist**: walk all channels, `UpsertVideo` every candidate (so it appears in
|
||||||
the list), apply pre-filters (seen/manual/backoff), collect surviving candidates.
|
the list), apply pre-filters (seen/manual/backoff/**recency**), collect surviving candidates.
|
||||||
|
The recency pre-filter (ADR-020) drops auto-mode videos published before
|
||||||
|
`now - TAPIR_AUTO_SUMMARIZE_WINDOW` unless they are explicitly requested; an undated video is
|
||||||
|
never aged out. They remain persisted/listed — only auto-summarisation is skipped.
|
||||||
2. **Sort**: order candidates `published_at DESC, NULLS LAST, discovery_pos ASC`. Videos with
|
2. **Sort**: order candidates `published_at DESC, NULLS LAST, discovery_pos ASC`. Videos with
|
||||||
no publish date (schema 001: nullable) sort after all dated content. The sort is in-memory
|
no publish date (schema 001: nullable) sort after all dated content. The sort is in-memory
|
||||||
(`slices.SortStableFunc`) — at current scale this is fine.
|
(`slices.SortStableFunc`) — at current scale this is fine.
|
||||||
@@ -235,7 +247,8 @@ Within each scheduled pass, `RunOnce` uses a three-phase structure:
|
|||||||
Before (per-channel inline): `[chanA-old, chanA-mid, chanB-new, chanB-null]`
|
Before (per-channel inline): `[chanA-old, chanA-mid, chanB-new, chanB-null]`
|
||||||
After (newest-first): `[chanB-new, chanA-mid, chanA-old, chanB-null]`
|
After (newest-first): `[chanB-new, chanA-mid, chanA-old, chanB-null]`
|
||||||
|
|
||||||
The set of processed videos is identical; only the order within a pass changes.
|
The set of *processed* videos now also excludes auto-mode back-catalogue beyond the recency
|
||||||
|
window (those stay listed, summarised on demand); within the processed set, only order changes.
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
|
|||||||
+4
-3
@@ -147,9 +147,10 @@ mechanism.
|
|||||||
|
|
||||||
- **USER** — one row per registered user (Stage 1, ADR-012; no longer single-row). The Tapir-side
|
- **USER** — one row per registered user (Stage 1, ADR-012; no longer single-row). The Tapir-side
|
||||||
profile; the Dex identity is held separately in `USER_IDENTITY`, not on this row. `auto_summarize`
|
profile; the Dex identity is held separately in `USER_IDENTITY`, not on this row. `auto_summarize`
|
||||||
(migration 006) is the per-user mode flag: `TRUE` = auto-summarize every new video. Default is
|
(migration 006) is the per-user mode flag: `TRUE` = auto-summarize new videos **published within
|
||||||
**true** for new users (migration 011, ADR-018); existing rows were back-filled via migration 012
|
the recency window** (`TAPIR_AUTO_SUMMARIZE_WINDOW`, default ~7d, ADR-020); older videos are
|
||||||
with RLS bypass.
|
listed but summarised on demand. Default is **true** for new users (migration 011, ADR-018);
|
||||||
|
existing rows were back-filled via migration 012 with RLS bypass.
|
||||||
- **USER_IDENTITY** (migration 004) — the `dex_subject → user_id` map. `dex_subject` is the PK,
|
- **USER_IDENTITY** (migration 004) — the `dex_subject → user_id` map. `dex_subject` is the PK,
|
||||||
`user_id` a `UNIQUE` FK to `users` with `ON DELETE CASCADE`. This is the bridge resolved at login
|
`user_id` a `UNIQUE` FK to `users` with `ON DELETE CASCADE`. This is the bridge resolved at login
|
||||||
*before* a `user_id` is known, so it is **deliberately not RLS-enabled** (it holds no user data;
|
*before* a `user_id` is known, so it is **deliberately not RLS-enabled** (it holds no user data;
|
||||||
|
|||||||
@@ -1,5 +1,11 @@
|
|||||||
# Spec — Newest-first batch ordering + honest "Try now" / prioritisation docs
|
# Spec — Newest-first batch ordering + honest "Try now" / prioritisation docs
|
||||||
|
|
||||||
|
> **Extended by ADR-020 (2026-06-08).** This spec covers the *batch processing* order within a
|
||||||
|
> pass. ADR-020 adds (a) a recency pre-filter — auto mode skips videos published before
|
||||||
|
> `TAPIR_AUTO_SUMMARIZE_WINDOW`, listed but summarised on demand — and (b) the same
|
||||||
|
> `published_at DESC NULLS LAST` ordering on the **list read** (`ListVideos`), which previously
|
||||||
|
> sorted by `seen_at`. See `DECISIONS.md` ADR-020.
|
||||||
|
|
||||||
**Repo:** tapir · **Size:** small · **Solo session.**
|
**Repo:** tapir · **Size:** small · **Solo session.**
|
||||||
|
|
||||||
**Why.** Product intent (maintainer, 2026-06-06): a new user should get summaries of their
|
**Why.** Product intent (maintainer, 2026-06-06): a new user should get summaries of their
|
||||||
|
|||||||
@@ -1,5 +1,10 @@
|
|||||||
# Spec — In-process scheduled discovery + auto-summarize + rate-gate finish
|
# Spec — In-process scheduled discovery + auto-summarize + rate-gate finish
|
||||||
|
|
||||||
|
> **Extended by ADR-020 (2026-06-08).** Auto-summarize is no longer "every unseen video": the
|
||||||
|
> scheduler now skips videos published before `TAPIR_AUTO_SUMMARIZE_WINDOW` (default ~7d) unless
|
||||||
|
> explicitly requested, so a back-catalogue does not re-drive the rate gate every cycle. See
|
||||||
|
> `DECISIONS.md` ADR-020.
|
||||||
|
|
||||||
**Repo:** tapir · **Size:** medium · **Solo session** (not a swarm).
|
**Repo:** tapir · **Size:** medium · **Solo session** (not a swarm).
|
||||||
|
|
||||||
**Why this exists.** The Stage-0 gate ("me or a friend returns and reads/acts in ≥2 separate
|
**Why this exists.** The Stage-0 gate ("me or a friend returns and reads/acts in ≥2 separate
|
||||||
|
|||||||
@@ -1,5 +1,12 @@
|
|||||||
# Spec — Unify video-card states: one "Summarize now" verb, honest no-captions state
|
# Spec — Unify video-card states: one "Summarize now" verb, honest no-captions state
|
||||||
|
|
||||||
|
> **Superseded in part by ADR-020 (2026-06-08).** The five card states still hold, but the copy
|
||||||
|
> changed: the nudge verb is now **"Summarize"** (not "Summarize now"), the rate-limited state
|
||||||
|
> reads **"In queue"** (not "Fetching soon…"), and the queued state reads **"summarizing
|
||||||
|
> shortly"** (not "waiting for the next run"). The list also now collapses older un-summarized
|
||||||
|
> and caption-less videos. See `DECISIONS.md` ADR-020 and `views.templ` (`VideoCard`) for the
|
||||||
|
> current copy; this doc is kept as the original design record.
|
||||||
|
|
||||||
**Repo:** tapir · **Size:** small, **view-layer only** (`views.templ` + a little CSS in
|
**Repo:** tapir · **Size:** small, **view-layer only** (`views.templ` + a little CSS in
|
||||||
`view.go`; regenerate `views_templ.go`). No handler, store, or DB change. The two existing
|
`view.go`; regenerate `views_templ.go`). No handler, store, or DB change. The two existing
|
||||||
handlers (`/summarize`, `/retry-now`) stay exactly as they are — only what the card *shows*
|
handlers (`/summarize`, `/retry-now`) stay exactly as they are — only what the card *shows*
|
||||||
|
|||||||
@@ -179,3 +179,4 @@ distinguishable.
|
|||||||
| **"Summarize now" foreground path** | Unified quiet nudge button on actionable non-summarized cards. Five explicit card states — (1) summarized: chip + no button; (2) no captions (`transcript_status = 'none'`): "No transcript available", no button; (3) queued: "Queued" chip, no button; (4) rate-limited: "Fetching soon…" + "Summarize now" → `POST /v/{id}/retry-now` (clears `rate_limited_at`, triggers engine); (5) pending: "Not summarized" + "Summarize now" → `POST /v/{id}/summarize` (queues + triggers engine). One verb, one style (`.btn-quiet`); backend difference invisible to user. Both handlers call `ProcessVideo` through `globalFetchGate`. Rate gate respected, not bypassed — this is onboarding prioritisation. | Fast onboarding value; honest dead-end for no-captions videos (no button that fails). | `internal/web/handlers.go` (`handleRetryNow`, `handleRequestSummarize`); `internal/web/views.templ` (`VideoCard`) |
|
| **"Summarize now" foreground path** | Unified quiet nudge button on actionable non-summarized cards. Five explicit card states — (1) summarized: chip + no button; (2) no captions (`transcript_status = 'none'`): "No transcript available", no button; (3) queued: "Queued" chip, no button; (4) rate-limited: "Fetching soon…" + "Summarize now" → `POST /v/{id}/retry-now` (clears `rate_limited_at`, triggers engine); (5) pending: "Not summarized" + "Summarize now" → `POST /v/{id}/summarize` (queues + triggers engine). One verb, one style (`.btn-quiet`); backend difference invisible to user. Both handlers call `ProcessVideo` through `globalFetchGate`. Rate gate respected, not bypassed — this is onboarding prioritisation. | Fast onboarding value; honest dead-end for no-captions videos (no button that fails). | `internal/web/handlers.go` (`handleRetryNow`, `handleRequestSummarize`); `internal/web/views.templ` (`VideoCard`) |
|
||||||
| **Pipeline stats bar** | A one-line status bar above the video list: `N summarized · M fetching soon · K no captions`. Computed from the unfiltered row set; hidden when all videos are summarized. Gives the user a clear read on pipeline state without any interaction. | Replaces the "why is nothing happening?" confusion when most videos are pending or rate-limited. | `internal/web/view.go` (`PipelineStats`, `pipelineStats`) |
|
| **Pipeline stats bar** | A one-line status bar above the video list: `N summarized · M fetching soon · K no captions`. Computed from the unfiltered row set; hidden when all videos are summarized. Gives the user a clear read on pipeline state without any interaction. | Replaces the "why is nothing happening?" confusion when most videos are pending or rate-limited. | `internal/web/view.go` (`PipelineStats`, `pipelineStats`) |
|
||||||
| **Unavailable channels (account page)** | The `/account` page shows a "Unavailable channels" section when any channels returned HTTP 404 on the last discovery pass. Lists channel name, an "unavailable" badge, and the first-seen date. Data sourced from the `channel_errors` table (migration 013). | Surfaces silent failures so users know why some subscribed channels produce no new videos. | migration 013; `internal/web/account.go`; `internal/adapters/youtube/youtube.go` (`domain.ErrChannelUnavailable`) |
|
| **Unavailable channels (account page)** | The `/account` page shows a "Unavailable channels" section when any channels returned HTTP 404 on the last discovery pass. Lists channel name, an "unavailable" badge, and the first-seen date. Data sourced from the `channel_errors` table (migration 013). | Surfaces silent failures so users know why some subscribed channels produce no new videos. | migration 013; `internal/web/account.go`; `internal/adapters/youtube/youtube.go` (`domain.ErrChannelUnavailable`) |
|
||||||
|
| **Recency window + sparse-state honesty (ADR-020)** | Supersedes the copy/sort in the rows above. Auto-summarize is bounded to videos published within `TAPIR_AUTO_SUMMARIZE_WINDOW` (~7d); older un-summarized videos collapse behind a single "Show N older videos — summarize on demand" disclosure, and caption-less videos collapse to a one-line count (not N cards). List order is now `summarized-first, published_at DESC NULLS LAST`. Copy reframed for honest scarcity: pipeline bar reads "N ready · M in queue · K no captions" (no "fetching soon"); a gradual-fill note explains the rate limit; the nudge verb is "Summarize" (not "Summarize now"); the queued card says "summarizing shortly"; the empty-connected state drops the impossible `tapir run` instruction. Detail leads with Takeaways. Filters slimmed (no date pickers; hidden when empty); watched/skipped segmented; back link on detail; empty terms checkbox removed. | Make the sparse reality legible and honest instead of implying abundance/imminence; bound auto load so the back-catalogue doesn't re-drive the caption gate. Never fetch harder — scarcity is surfaced, not engineered around. | ADR-020; `2384c47`, `3df0459`, `40b703e`, `a1a5217`, `4a0a56e`, `9bf1c31`, `980638d`, `12fb031`, `f775441`, `51aa5d9` |
|
||||||
|
|||||||
@@ -10,6 +10,8 @@ Feature: Connect and manage video accounts
|
|||||||
And my refresh token is stored only as a secret reference
|
And my refresh token is stored only as a secret reference
|
||||||
And my subscriptions are synced
|
And my subscriptions are synced
|
||||||
|
|
||||||
|
@pending
|
||||||
|
# Vimeo connect is not built yet (provider label exists; no connect flow or test).
|
||||||
Scenario: Connect a Vimeo account
|
Scenario: Connect a Vimeo account
|
||||||
Given I have no connected video accounts
|
Given I have no connected video accounts
|
||||||
When I connect my Vimeo account
|
When I connect my Vimeo account
|
||||||
@@ -28,6 +30,9 @@ Feature: Connect and manage video accounts
|
|||||||
And no new videos are watched for that connection
|
And no new videos are watched for that connection
|
||||||
And my existing summaries remain readable
|
And my existing summaries remain readable
|
||||||
|
|
||||||
|
@pending
|
||||||
|
# Per-provider BYO credential config is not built as a web flow yet (the summarizer
|
||||||
|
# supports a fallback endpoint, but there is no user-facing BYO setup + its test).
|
||||||
Scenario Outline: BYO AI credential is optional and per-provider
|
Scenario Outline: BYO AI credential is optional and per-provider
|
||||||
When I configure a BYO provider "<provider>"
|
When I configure a BYO provider "<provider>"
|
||||||
Then the credential is stored only as a secret reference
|
Then the credential is stored only as a secret reference
|
||||||
|
|||||||
@@ -19,6 +19,9 @@ Feature: Public landing page
|
|||||||
Then I see a link to my summaries
|
Then I see a link to my summaries
|
||||||
And I see a way to log out
|
And I see a way to log out
|
||||||
|
|
||||||
|
@pending
|
||||||
|
# Behaviour ships (logout redirects to /welcome) but is not unit-tested: logout lives in
|
||||||
|
# the OIDC Auth impl and StubAuth has no routes to exercise it cheaply.
|
||||||
Scenario: Logging out returns to the welcome page
|
Scenario: Logging out returns to the welcome page
|
||||||
Given I am logged in
|
Given I am logged in
|
||||||
When I log out
|
When I log out
|
||||||
|
|||||||
@@ -34,6 +34,9 @@ Feature: Register and manage a multi-user account
|
|||||||
And the other user's data remains intact
|
And the other user's data remains intact
|
||||||
And my Dex identity is left intact
|
And my Dex identity is left intact
|
||||||
|
|
||||||
|
@pending
|
||||||
|
# Re-registration after delete is supported by design (delete leaves the Dex identity,
|
||||||
|
# ADR-013) but has no dedicated end-to-end test yet.
|
||||||
Scenario: A deleted user can register again as a fresh account
|
Scenario: A deleted user can register again as a fresh account
|
||||||
Given I deleted my Tapir account but my Dex identity still exists
|
Given I deleted my Tapir account but my Dex identity still exists
|
||||||
When I sign in again
|
When I sign in again
|
||||||
|
|||||||
@@ -6,15 +6,25 @@ Feature: Choose how new videos get summarized
|
|||||||
Background:
|
Background:
|
||||||
Given I am a registered user with a connected video account
|
Given I am a registered user with a connected video account
|
||||||
|
|
||||||
Scenario: Auto mode summarizes every new video
|
Scenario: Auto mode summarizes recent new videos automatically
|
||||||
Given my summarization mode is "auto"
|
Given my summarization mode is "auto"
|
||||||
When a subscribed channel posts a new video with captions
|
When a subscribed channel posts a new video with captions within the recency window
|
||||||
Then Tapir summarizes it without my asking
|
Then Tapir summarizes it without my asking
|
||||||
And the summary appears in my list
|
And the summary appears in my list
|
||||||
|
|
||||||
Scenario: Manual mode is the default and leaves new videos unsummarized
|
Scenario: Auto mode lists older videos without summarizing them
|
||||||
Given I have not changed my summarization mode
|
Given my summarization mode is "auto"
|
||||||
Then my mode is "manual"
|
When discovery finds a video published before the recency window
|
||||||
|
Then the video appears in my list with no summary
|
||||||
|
And it is not summarized automatically
|
||||||
|
And I can still summarize it on demand with "Summarize"
|
||||||
|
|
||||||
|
Scenario: Automatic is the default for a new user
|
||||||
|
Given I have just registered
|
||||||
|
Then my summarization mode is "auto"
|
||||||
|
|
||||||
|
Scenario: Manual mode leaves new videos unsummarized
|
||||||
|
Given my summarization mode is "manual"
|
||||||
When a subscribed channel posts a new video with captions
|
When a subscribed channel posts a new video with captions
|
||||||
Then the video appears in my list with no summary
|
Then the video appears in my list with no summary
|
||||||
And nothing is summarized until I request it
|
And nothing is summarized until I request it
|
||||||
@@ -30,3 +40,8 @@ Feature: Choose how new videos get summarized
|
|||||||
# auto_summarize is a per-user setting and summarize_requested is a per-video queue
|
# auto_summarize is a per-user setting and summarize_requested is a per-video queue
|
||||||
# flag (migration 006). The web button sets the flag; `tapir run` processes both the
|
# flag (migration 006). The web button sets the flag; `tapir run` processes both the
|
||||||
# auto videos and the manually queued ones, then clears the flag.
|
# auto videos and the manually queued ones, then clears the flag.
|
||||||
|
#
|
||||||
|
# Recency bound (ADR-020): in auto mode the scheduler only summarizes videos published
|
||||||
|
# within TAPIR_AUTO_SUMMARIZE_WINDOW (default ~7d); older videos are discovered and
|
||||||
|
# listed but wait for an explicit "Summarize" — so a back-catalogue does not re-drive
|
||||||
|
# the per-IP caption gate (ADR-014) every cycle. A manual request bypasses the bound.
|
||||||
|
|||||||
@@ -0,0 +1,211 @@
|
|||||||
|
package acceptance
|
||||||
|
|
||||||
|
// This is the name-coverage gate for the BDD spec (see docs/use-cases/*.feature).
|
||||||
|
// There is no godog runner — the .feature files are design records, and the real
|
||||||
|
// behaviour is covered by the hand-written Go tests across the module. This test
|
||||||
|
// keeps the two from drifting in the cheapest honest way: every non-@pending
|
||||||
|
// Scenario must have an entry in scenarioCoverage pointing at a Go test that
|
||||||
|
// actually exists. It does NOT prove the test exercises the scenario (only godog
|
||||||
|
// could); it catches the common drift — "added a scenario, forgot the test", a
|
||||||
|
// renamed/deleted covering test, or a scenario removed without cleaning the map.
|
||||||
|
//
|
||||||
|
// When you add a Scenario: either map it here to its covering test, or tag it
|
||||||
|
// @pending in the .feature with a one-line reason for why it has no test yet.
|
||||||
|
|
||||||
|
import (
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"regexp"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// scenarioCoverage maps each non-@pending Scenario name to the Go test that
|
||||||
|
// covers it. Keep it in sync with docs/use-cases/*.feature — the test below
|
||||||
|
// fails if a scenario is unmapped, a mapped test is missing, or an entry no
|
||||||
|
// longer matches a real non-pending scenario.
|
||||||
|
var scenarioCoverage = map[string]string{
|
||||||
|
// ai_routing.feature
|
||||||
|
"Local AI produces the summary": "TestSummarize_LocalSucceeds",
|
||||||
|
"Local AI fails and the user has a BYO provider configured": "TestSummarize_FallsBackToBYO",
|
||||||
|
"Local AI fails and the user has no BYO provider": "TestSummarize_LocalFailsNoBYO_NoExternalSend",
|
||||||
|
"A user without BYO never has content sent externally": "TestSummarize_NoBYO_ContentOnlyLocal",
|
||||||
|
|
||||||
|
// landing_page.feature
|
||||||
|
"An unauthenticated visit to the root is sent to the welcome page": "TestUnauthenticatedRootRedirectsToWelcome",
|
||||||
|
"The welcome page invites an unauthenticated visitor to start": "TestWelcomeLoggedOut",
|
||||||
|
"An authenticated user on the welcome page sees their way in and out": "TestWelcomeLoggedIn",
|
||||||
|
|
||||||
|
// connect_account.feature
|
||||||
|
"Connect a YouTube account": "TestCallbackExchangesAndRecordsConnection",
|
||||||
|
"Tokens are never stored in the clear": "TestCallbackExchangesAndRecordsConnection",
|
||||||
|
"Revoking a connection stops watching but keeps history": "TestDisconnectRemovesTokenAndConnectionKeepsAccount",
|
||||||
|
|
||||||
|
// summarize_mode.feature
|
||||||
|
"Auto mode summarizes recent new videos automatically": "TestRunOnce_AutoMode_SkipsOldVideos",
|
||||||
|
"Auto mode lists older videos without summarizing them": "TestRunOnce_AutoMode_SkipsOldVideos",
|
||||||
|
"Automatic is the default for a new user": "TestRegisteredUserDefaultsAutoSummarizeOn",
|
||||||
|
"Manual mode leaves new videos unsummarized": "TestRunOnce_ManualMode_SkipsUnrequested",
|
||||||
|
"Requesting a summary in manual mode queues it for the next run": "TestRunOnce_ManualMode_ProcessesRequested",
|
||||||
|
|
||||||
|
// registration.feature
|
||||||
|
"A new Dex subject is routed to registration": "TestUnregisteredSubjectRedirectedToRegister",
|
||||||
|
"Registering creates the account and its identity mapping": "TestRegisterCreatesExactlyOneUserAndIdentity",
|
||||||
|
"A returning subject passes straight through": "TestRegisteredSubjectPassesThrough",
|
||||||
|
"Deleting an account removes only my data and leaves other users untouched": "TestDeleteAccountWipesDataAndSecretsAndLogsOut",
|
||||||
|
|
||||||
|
// summarize_new_video.feature
|
||||||
|
"A subscribed channel posts a video that has captions": "TestSubscribedVideoWithCaptionsIsSummarizedAndDelivered",
|
||||||
|
"A subscribed channel posts a video with no usable transcript": "TestVideoWithNoTranscriptIsSkipped",
|
||||||
|
"A channel I am not subscribed to posts a video": "TestUnsubscribedChannelVideoIsNotProcessed",
|
||||||
|
"The same video is not summarized twice": "TestAlreadySummarizedVideoIsNotReprocessed",
|
||||||
|
}
|
||||||
|
|
||||||
|
var (
|
||||||
|
scenarioRe = regexp.MustCompile(`^\s*Scenario(?: Outline)?:\s*(.+?)\s*$`)
|
||||||
|
testFuncRe = regexp.MustCompile(`^func (Test\w+)\(`)
|
||||||
|
)
|
||||||
|
|
||||||
|
// scenario is one parsed Gherkin scenario and whether it is @pending.
|
||||||
|
type scenario struct {
|
||||||
|
name string
|
||||||
|
pending bool
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestScenarioCoverage(t *testing.T) {
|
||||||
|
root := moduleRoot(t)
|
||||||
|
|
||||||
|
scenarios := parseScenarios(t, filepath.Join(root, "docs", "use-cases"))
|
||||||
|
if len(scenarios) == 0 {
|
||||||
|
t.Fatal("no scenarios parsed from docs/use-cases — wrong path?")
|
||||||
|
}
|
||||||
|
tests := allTestFuncNames(t, root)
|
||||||
|
|
||||||
|
// Index scenario names for the reverse (stale-entry) check.
|
||||||
|
active := map[string]bool{} // non-pending scenario names
|
||||||
|
var pending []string
|
||||||
|
for _, s := range scenarios {
|
||||||
|
if s.pending {
|
||||||
|
pending = append(pending, s.name)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
active[s.name] = true
|
||||||
|
|
||||||
|
// 1. Every non-pending scenario must be mapped.
|
||||||
|
fn, ok := scenarioCoverage[s.name]
|
||||||
|
if !ok {
|
||||||
|
t.Errorf("scenario %q has no coverage entry — map it in scenarioCoverage to a covering test, or tag it @pending in the .feature", s.name)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// 2. The mapped test must actually exist.
|
||||||
|
if !tests[fn] {
|
||||||
|
t.Errorf("scenario %q maps to %q, which is not a Test function anywhere in the module", s.name, fn)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// 3. No stale entries: every map key must be a real, non-pending scenario.
|
||||||
|
for name := range scenarioCoverage {
|
||||||
|
if !active[name] {
|
||||||
|
t.Errorf("scenarioCoverage has entry %q, which is not a current non-pending scenario (renamed, removed, or now @pending?)", name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if len(pending) > 0 {
|
||||||
|
t.Logf("%d @pending scenario(s) without a test (tracked, not required): %s",
|
||||||
|
len(pending), strings.Join(pending, "; "))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// parseScenarios reads every *.feature under dir and returns its scenarios with
|
||||||
|
// their @pending status. A scenario is @pending when a `@pending` tag line
|
||||||
|
// precedes it (tags survive intervening comment lines, the layout these files
|
||||||
|
// use); the flag is consumed at the Scenario line and reset afterward.
|
||||||
|
func parseScenarios(t *testing.T, dir string) []scenario {
|
||||||
|
t.Helper()
|
||||||
|
entries, err := os.ReadDir(dir)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("read use-cases dir: %v", err)
|
||||||
|
}
|
||||||
|
var out []scenario
|
||||||
|
for _, e := range entries {
|
||||||
|
if e.IsDir() || !strings.HasSuffix(e.Name(), ".feature") {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
b, err := os.ReadFile(filepath.Join(dir, e.Name()))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("read %s: %v", e.Name(), err)
|
||||||
|
}
|
||||||
|
pending := false
|
||||||
|
for _, line := range strings.Split(string(b), "\n") {
|
||||||
|
trimmed := strings.TrimSpace(line)
|
||||||
|
if strings.HasPrefix(trimmed, "@") {
|
||||||
|
if strings.Contains(trimmed, "@pending") {
|
||||||
|
pending = true
|
||||||
|
}
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if m := scenarioRe.FindStringSubmatch(line); m != nil {
|
||||||
|
out = append(out, scenario{name: m[1], pending: pending})
|
||||||
|
pending = false
|
||||||
|
}
|
||||||
|
// comment (#) and step lines leave a set @pending intact until the
|
||||||
|
// scenario consumes it; a blank line between scenarios is harmless.
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
// allTestFuncNames walks the module for `func TestXxx(` declarations, excluding
|
||||||
|
// this file (whose regex literal would otherwise look like a definition).
|
||||||
|
func allTestFuncNames(t *testing.T, root string) map[string]bool {
|
||||||
|
t.Helper()
|
||||||
|
self := "scenario_coverage_test.go"
|
||||||
|
names := map[string]bool{}
|
||||||
|
err := filepath.WalkDir(root, func(path string, d os.DirEntry, err error) error {
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if d.IsDir() {
|
||||||
|
if d.Name() == ".git" {
|
||||||
|
return filepath.SkipDir
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
if !strings.HasSuffix(path, "_test.go") || filepath.Base(path) == self {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
b, err := os.ReadFile(path)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
for _, line := range strings.Split(string(b), "\n") {
|
||||||
|
if m := testFuncRe.FindStringSubmatch(line); m != nil {
|
||||||
|
names[m[1]] = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("walk module: %v", err)
|
||||||
|
}
|
||||||
|
return names
|
||||||
|
}
|
||||||
|
|
||||||
|
// moduleRoot walks up from the working directory to the dir containing go.mod.
|
||||||
|
func moduleRoot(t *testing.T) string {
|
||||||
|
t.Helper()
|
||||||
|
dir, err := os.Getwd()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("getwd: %v", err)
|
||||||
|
}
|
||||||
|
for {
|
||||||
|
if _, err := os.Stat(filepath.Join(dir, "go.mod")); err == nil {
|
||||||
|
return dir
|
||||||
|
}
|
||||||
|
parent := filepath.Dir(dir)
|
||||||
|
if parent == dir {
|
||||||
|
t.Fatal("go.mod not found walking up from cwd")
|
||||||
|
}
|
||||||
|
dir = parent
|
||||||
|
}
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user