Compare commits
6
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
98cfae595c | ||
|
|
2a595b5a92 | ||
|
|
b7a2cc5fdf | ||
|
|
db638cca11 | ||
|
|
38579598e0 | ||
|
|
d6fa92b176 |
@@ -0,0 +1,189 @@
|
||||
// Package classification defines the data-sensitivity taxonomy and the
|
||||
// per-wing / per-repo tagging the capture server reads to enforce the I1
|
||||
// sovereignty gate (issue #50, capture spec §4.1).
|
||||
//
|
||||
// The single load-bearing property is fail-safe-to-strictest: a target
|
||||
// with no explicit tag and no known default classifies as Confidential,
|
||||
// never as something more permissive. A missing tag must never silently
|
||||
// downgrade — that would turn the I1 gate into theatre.
|
||||
//
|
||||
// Classification is read from an optional classification.yaml at the
|
||||
// brain root. A central, Flux-reconcilable file is deliberate: it is
|
||||
// auditable in one place (I2/I5), it does not require a live Gitea client
|
||||
// to classify a repo (so this package has no dependency on the gitea
|
||||
// tracker work), and it avoids tagging a wing's _index.md frontmatter —
|
||||
// which BuildWingIndex regenerates and would clobber.
|
||||
package classification
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
|
||||
"gopkg.in/yaml.v3"
|
||||
)
|
||||
|
||||
// Level is a data-sensitivity tier. Higher is stricter, so the "stricter
|
||||
// wins" rule (spec §4.1 model C) is a plain max.
|
||||
type Level int
|
||||
|
||||
const (
|
||||
Public Level = iota
|
||||
Internal
|
||||
Confidential
|
||||
)
|
||||
|
||||
// String returns the canonical lowercase token for a level.
|
||||
func (l Level) String() string {
|
||||
switch l {
|
||||
case Public:
|
||||
return "public"
|
||||
case Internal:
|
||||
return "internal"
|
||||
case Confidential:
|
||||
return "confidential"
|
||||
default:
|
||||
return fmt.Sprintf("level(%d)", int(l))
|
||||
}
|
||||
}
|
||||
|
||||
// ParseLevel parses a level token (case-insensitive, surrounding space
|
||||
// tolerated). An unknown token is an error — callers must decide what to
|
||||
// do with bad input rather than have it silently coerced.
|
||||
func ParseLevel(s string) (Level, error) {
|
||||
switch strings.ToLower(strings.TrimSpace(s)) {
|
||||
case "public":
|
||||
return Public, nil
|
||||
case "internal":
|
||||
return Internal, nil
|
||||
case "confidential":
|
||||
return Confidential, nil
|
||||
default:
|
||||
return Confidential, fmt.Errorf("unknown classification level %q (want public/internal/confidential)", s)
|
||||
}
|
||||
}
|
||||
|
||||
// Stricter returns the more restrictive of two levels.
|
||||
func Stricter(a, b Level) Level {
|
||||
if a > b {
|
||||
return a
|
||||
}
|
||||
return b
|
||||
}
|
||||
|
||||
// TargetKind distinguishes the two kinds of capture destination.
|
||||
type TargetKind int
|
||||
|
||||
const (
|
||||
WingTarget TargetKind = iota // a brain wing (insights land here)
|
||||
RepoTarget // a Gitea repo (tickets / summaries land here)
|
||||
)
|
||||
|
||||
// Target names a capture destination to classify.
|
||||
type Target struct {
|
||||
Kind TargetKind
|
||||
Name string
|
||||
}
|
||||
|
||||
// Config holds the explicit per-wing / per-repo classification tags read
|
||||
// from classification.yaml. Absent entries fall through to the built-in
|
||||
// defaults in defaultFor. The zero value (no file) is valid and applies
|
||||
// defaults to everything.
|
||||
type Config struct {
|
||||
wings map[string]Level
|
||||
repos map[string]Level
|
||||
}
|
||||
|
||||
// rawConfig is the on-disk YAML shape: string→string maps, parsed into
|
||||
// validated levels by Load.
|
||||
type rawConfig struct {
|
||||
Wings map[string]string `yaml:"wings"`
|
||||
Repos map[string]string `yaml:"repos"`
|
||||
}
|
||||
|
||||
// Load reads classification.yaml from brainDir. An absent file is not an
|
||||
// error — it yields an empty config where every target classifies by the
|
||||
// built-in defaults. A malformed file, or any unparseable level token in
|
||||
// it, is a hard error: a classification source the server cannot trust
|
||||
// must fail loud, not degrade silently.
|
||||
func Load(brainDir string) (*Config, error) {
|
||||
cfg := &Config{wings: map[string]Level{}, repos: map[string]Level{}}
|
||||
|
||||
data, err := os.ReadFile(filepath.Join(brainDir, "classification.yaml"))
|
||||
if err != nil {
|
||||
if os.IsNotExist(err) {
|
||||
return cfg, nil
|
||||
}
|
||||
return nil, fmt.Errorf("read classification.yaml: %w", err)
|
||||
}
|
||||
|
||||
var raw rawConfig
|
||||
if err := yaml.Unmarshal(data, &raw); err != nil {
|
||||
return nil, fmt.Errorf("parse classification.yaml: %w", err)
|
||||
}
|
||||
for name, lvl := range raw.Wings {
|
||||
parsed, perr := ParseLevel(lvl)
|
||||
if perr != nil {
|
||||
return nil, fmt.Errorf("wing %q: %w", name, perr)
|
||||
}
|
||||
cfg.wings[normalise(name)] = parsed
|
||||
}
|
||||
for name, lvl := range raw.Repos {
|
||||
parsed, perr := ParseLevel(lvl)
|
||||
if perr != nil {
|
||||
return nil, fmt.Errorf("repo %q: %w", name, perr)
|
||||
}
|
||||
cfg.repos[normalise(name)] = parsed
|
||||
}
|
||||
return cfg, nil
|
||||
}
|
||||
|
||||
// Derive returns the classification for any target — the function the
|
||||
// capture use-case calls per item.
|
||||
func (c *Config) Derive(t Target) Level {
|
||||
if t.Kind == RepoTarget {
|
||||
return c.Repo(t.Name)
|
||||
}
|
||||
return c.Wing(t.Name)
|
||||
}
|
||||
|
||||
// Wing classifies a brain wing: an explicit tag wins, else defaults.
|
||||
func (c *Config) Wing(name string) Level {
|
||||
if lvl, ok := c.wings[normalise(name)]; ok {
|
||||
return lvl
|
||||
}
|
||||
return defaultFor(name)
|
||||
}
|
||||
|
||||
// Repo classifies a Gitea repo: an explicit tag wins, else defaults.
|
||||
func (c *Config) Repo(name string) Level {
|
||||
if lvl, ok := c.repos[normalise(name)]; ok {
|
||||
return lvl
|
||||
}
|
||||
return defaultFor(name)
|
||||
}
|
||||
|
||||
// defaultFor applies the built-in defaulting rules when a target has no
|
||||
// explicit tag:
|
||||
// - client-* → Confidential (client work is confidential by default)
|
||||
// - hyperguild / homelab → Internal (the operator's own infra)
|
||||
// - everything else → Confidential (fail safe to strictest)
|
||||
func defaultFor(name string) Level {
|
||||
n := normalise(name)
|
||||
if strings.HasPrefix(n, "client-") {
|
||||
return Confidential
|
||||
}
|
||||
switch n {
|
||||
case "hyperguild", "homelab":
|
||||
return Internal
|
||||
default:
|
||||
return Confidential
|
||||
}
|
||||
}
|
||||
|
||||
// normalise lowercases and trims a wing/repo name so matching and the
|
||||
// client-* prefix check are case-insensitive.
|
||||
func normalise(name string) string {
|
||||
return strings.ToLower(strings.TrimSpace(name))
|
||||
}
|
||||
@@ -0,0 +1,112 @@
|
||||
package classification
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func TestLevelOrderingAndString(t *testing.T) {
|
||||
assert.True(t, Public < Internal)
|
||||
assert.True(t, Internal < Confidential)
|
||||
assert.Equal(t, "public", Public.String())
|
||||
assert.Equal(t, "internal", Internal.String())
|
||||
assert.Equal(t, "confidential", Confidential.String())
|
||||
}
|
||||
|
||||
func TestParseLevel(t *testing.T) {
|
||||
for s, want := range map[string]Level{
|
||||
"public": Public, "internal": Internal, "confidential": Confidential,
|
||||
"PUBLIC": Public, " Confidential ": Confidential,
|
||||
} {
|
||||
got, err := ParseLevel(s)
|
||||
require.NoError(t, err, s)
|
||||
assert.Equal(t, want, got, s)
|
||||
}
|
||||
_, err := ParseLevel("secret")
|
||||
require.Error(t, err, "unknown level must error, not silently default")
|
||||
_, err = ParseLevel("")
|
||||
require.Error(t, err)
|
||||
}
|
||||
|
||||
func TestStricterReturnsMax(t *testing.T) {
|
||||
assert.Equal(t, Confidential, Stricter(Internal, Confidential))
|
||||
assert.Equal(t, Confidential, Stricter(Confidential, Public))
|
||||
assert.Equal(t, Internal, Stricter(Public, Internal))
|
||||
assert.Equal(t, Public, Stricter(Public, Public))
|
||||
}
|
||||
|
||||
func TestLoadAbsentFileIsDefaultsOnly(t *testing.T) {
|
||||
cfg, err := Load(t.TempDir())
|
||||
require.NoError(t, err, "absent classification.yaml must not be an error — defaults apply")
|
||||
require.NotNil(t, cfg)
|
||||
// Pure defaulting still works.
|
||||
assert.Equal(t, Internal, cfg.Wing("hyperguild"))
|
||||
assert.Equal(t, Confidential, cfg.Wing("anything-unknown"))
|
||||
}
|
||||
|
||||
func TestLoadParsesExplicitTags(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
require.NoError(t, os.WriteFile(filepath.Join(dir, "classification.yaml"), []byte(
|
||||
"wings:\n research-public: public\n hyperguild: confidential\nrepos:\n infra: internal\n research-public: public\n",
|
||||
), 0o644))
|
||||
|
||||
cfg, err := Load(dir)
|
||||
require.NoError(t, err)
|
||||
// Explicit tag wins over the built-in default (hyperguild default is internal).
|
||||
assert.Equal(t, Confidential, cfg.Wing("hyperguild"))
|
||||
// Explicit public is honoured.
|
||||
assert.Equal(t, Public, cfg.Wing("research-public"))
|
||||
assert.Equal(t, Internal, cfg.Repo("infra"))
|
||||
assert.Equal(t, Public, cfg.Repo("research-public"))
|
||||
}
|
||||
|
||||
func TestLoadRejectsUnknownLevelInFile(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
require.NoError(t, os.WriteFile(filepath.Join(dir, "classification.yaml"),
|
||||
[]byte("wings:\n x: top-secret\n"), 0o644))
|
||||
_, err := Load(dir)
|
||||
require.Error(t, err, "an unparseable level in the config must fail loud, not be ignored")
|
||||
}
|
||||
|
||||
func TestWingDefaulting(t *testing.T) {
|
||||
cfg, err := Load(t.TempDir())
|
||||
require.NoError(t, err)
|
||||
cases := map[string]Level{
|
||||
"client-seb": Confidential, // client-* → confidential
|
||||
"client-mastercard": Confidential,
|
||||
"hyperguild": Internal,
|
||||
"homelab": Internal,
|
||||
"jepa-fx": Confidential, // unknown → fail safe to strictest
|
||||
"": Confidential, // empty → fail safe
|
||||
}
|
||||
for wing, want := range cases {
|
||||
assert.Equal(t, want, cfg.Wing(wing), "wing %q", wing)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRepoDefaulting(t *testing.T) {
|
||||
cfg, err := Load(t.TempDir())
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, Confidential, cfg.Repo("client-seb-pipeline"))
|
||||
assert.Equal(t, Internal, cfg.Repo("hyperguild"))
|
||||
assert.Equal(t, Confidential, cfg.Repo("some-unknown-repo"), "untagged repo → confidential (fail safe)")
|
||||
}
|
||||
|
||||
func TestDeriveUnifiedTarget(t *testing.T) {
|
||||
cfg, err := Load(t.TempDir())
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, Internal, cfg.Derive(Target{Kind: WingTarget, Name: "homelab"}))
|
||||
assert.Equal(t, Confidential, cfg.Derive(Target{Kind: RepoTarget, Name: "client-x"}))
|
||||
assert.Equal(t, Confidential, cfg.Derive(Target{Kind: WingTarget, Name: "untagged"}))
|
||||
}
|
||||
|
||||
func TestCaseInsensitiveMatching(t *testing.T) {
|
||||
cfg, err := Load(t.TempDir())
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, Confidential, cfg.Wing("Client-SEB"), "client- prefix match is case-insensitive")
|
||||
assert.Equal(t, Internal, cfg.Wing("HyperGuild"))
|
||||
}
|
||||
@@ -0,0 +1,109 @@
|
||||
---
|
||||
name: close-session
|
||||
description: Disciplined end-of-session closeout for a Claude.ai chat before archiving it. Harvests the session's decisions, artifacts, and open threads and durably persists them to the brain MCP and the right Gitea repo so nothing is lost when context resets. Use this whenever the user signals they are wrapping up — phrases like "close this out", "let's wrap up", "before I archive", "session retro", "capture this before I go", "did we lose anything", or any end-of-session/handoff cue — even if they don't say the word "close". Also use when the user explicitly asks to retro, archive, or hand off a working session.
|
||||
---
|
||||
|
||||
# close-session
|
||||
|
||||
Capture a finishing Claude.ai work session into durable storage before the chat is archived and its context is lost. The goal is simple and load-bearing: **after this runs, a fresh session (or another agent) can reconstruct what was decided, what was shipped, and what is still open — without the original chat.**
|
||||
|
||||
This skill is **batch**: one session in, findings out, done. It does not loop or re-read its own fresh output semantically (see Phase 5). Run the phases in order. Stop at any confirmation gate that says STOP.
|
||||
|
||||
## Operating constraints (read first)
|
||||
|
||||
- **Gitea owner is always `mathias`.** Never guess another owner.
|
||||
- **Ground-truth at HEAD before acting.** Issue bodies and doc references rot — stale hostnames, retired services, moved endpoints. Before closing/commenting on any issue, `gitea:issue_get` it fresh. Before asserting an infra fact, verify it; do not copy it from memory or from a stale issue body.
|
||||
- **Current infra truths** (verify rather than trust, but these are the known-good baseline): Gitea is `git.d-ma.be` (not `gitea.d-ma.be`). LiteLLM is `http://koala:30401/v1/` (public `https://llm-api.d-ma.be`); piguard runs NGINX Proxy Manager only — never reference `piguard:4000` or `koala:4000`. Identity provider is Authentik (Dex migration complete).
|
||||
- **Side-effects need a confirmation gate.** Closing issues, committing files, and writing to the brain are all real writes. Surface exactly what will happen and get a clear yes before doing it. Reads are free; writes are gated.
|
||||
- **Never fabricate.** If the session didn't produce a decision worth persisting, say so and skip that write. An empty-but-honest closeout beats an invented one.
|
||||
|
||||
## Phase 1 — Harvest
|
||||
|
||||
Reconstruct what actually happened this session from the conversation itself. Produce, in working memory:
|
||||
|
||||
- **Decisions taken** — what was decided and the reasoning, not just the outcome.
|
||||
- **Artifacts produced** — issues filed/closed, PRs opened/merged, files committed, brain notes written, ADRs. Capture identifiers (issue numbers, PR numbers, paths, commit SHAs) as you go.
|
||||
- **Open threads** — what was deferred, what's blocked, what the next session should pick up.
|
||||
- **Generalizable learnings** — reusable patterns or footguns that would bite anyone again (these are brain-worthy; project status is not).
|
||||
|
||||
Be honest about fidelity: a long session compresses harder at the start than the end. Flag anything you're reconstructing rather than certain of.
|
||||
|
||||
## Phase 2 — Ground-truth Gitea state
|
||||
|
||||
For every repo touched this session, get its true current state before proposing any change. `gitea:repo_status` (owner `mathias`) gives branches + open PRs + protection in one call. For each issue you intend to close, comment on, or reference: `gitea:issue_get` it fresh and compare to what the session assumed. Note any drift (closed-already, body rotted, renamed) — you'll surface it in Phase 3.
|
||||
|
||||
Do not write anything in this phase. This is the read pass.
|
||||
|
||||
## Phase 3 — Confirm and act on issue changes
|
||||
|
||||
Present a single consolidated plan of issue actions: which to close (with closing comment), which to file (discovered-but-deferred work — token-budget gaps, recorded limitations, v2 follow-ups), which to comment on. Include the exact title/body for any new issue and the closing rationale for any close.
|
||||
|
||||
**GATE — STOP and get explicit confirmation before any issue write.** Issue closes and new issues are side-effects. Once confirmed, execute them (`gitea:issue_close`, `gitea:issue_create`, `gitea:issue_comment`, all owner `mathias`), correcting any rotted references you found in Phase 2 as you go.
|
||||
|
||||
## Phase 4 — Commit the canonical session summary
|
||||
|
||||
Write one summary file to `mathias/ai-sessions`, committed directly to `main` via `gitea:file_write_branch` (no PR — this repo is solo and unprotected; if branch protection is ever added, fall back to a branch + PR).
|
||||
|
||||
**Path:** `summaries/claudeai/<YYYY-MM>/<YYYY-MM-DD>-<topic-slug>-<chatid8>.md`
|
||||
where `<chatid8>` is the first 8 chars of the chat's UUID if known, else a short stable slug. `claudeai` has no host segment — Claude.ai is Anthropic-side, not a homelab host.
|
||||
|
||||
**Frontmatter — the REDUCED live-capture schema.** A live close-session capture cannot populate the batch-export telemetry (token counts, message counts, duration_ms, permission_mode) — those only exist in the account export pipeline. Write only what's truthfully known, and mark fidelity so a reader (or the batch pipeline) can tell a live capture from an export:
|
||||
|
||||
```yaml
|
||||
---
|
||||
title: "<concise session title>"
|
||||
client: "claudeai"
|
||||
interface: "claudeai-chat"
|
||||
date: "<YYYY-MM-DD>"
|
||||
repos_touched: [<repo slugs>]
|
||||
topic_tags: [<tags>]
|
||||
outcome: "<shipped|in-progress|abandoned>"
|
||||
fidelity: "live-capture" # NOT an export; reconstructed live from chat
|
||||
captured_by: "close-session-skill"
|
||||
---
|
||||
```
|
||||
|
||||
Do not invent the export-only fields. `fidelity: live-capture` is the honest signal; if the batch export later produces a richer summary for the same session, the export is source of truth and supersedes this.
|
||||
|
||||
**Body** (keep it reconstructable, not exhaustive):
|
||||
```markdown
|
||||
## One-paragraph summary
|
||||
## Decisions
|
||||
## Key artifacts
|
||||
## Open threads
|
||||
```
|
||||
|
||||
**GATE — STOP, show the full file (path + frontmatter + body), get explicit confirmation before committing.**
|
||||
|
||||
## Phase 5 — Brain orientation note (the durable "where we are" record)
|
||||
|
||||
Write one brain note so a fresh session can orient without the chat. This uses the `brain_update`/`brain_get` verbs (live since 2026-06).
|
||||
|
||||
**Target:** `wing: <domain>` (the project/topic domain, e.g. `hyperguild`, `jepa-fx`), `hall: decisions`. The note is a knowledge-type record (a decision/orientation), grouped by knowledge-type, not by interface surface.
|
||||
|
||||
**Batch read-after-write discipline (important — do these in order, do not interleave):**
|
||||
|
||||
1. **Read first, before any write.** Check whether an orientation note already exists for this wing/topic. Do your "does this already exist / what should I supersede" reads NOW, up front. BM25/keyword search and `brain_get` are immediate; semantic/vector search may lag up to ~5 min after a write, so never rely on a semantic query to find something you wrote earlier in this same run.
|
||||
2. **Write or supersede:**
|
||||
- **New note** → `brain_write` (wing, hall: decisions). Returns `{id, path, content_hash}`.
|
||||
- **Superseding a prior orientation note** → `brain_update` (slug or path, wing, hall, content, reason). Whole-note replace; stamps `supersedes`/`updated_at`; returns `{id, path, content_hash, superseded}`. Use this instead of a second `brain_write` to the same slug — blind re-write creates duplicates/contradictions, which is the exact failure brain_update exists to prevent.
|
||||
3. **Confirm it landed** via `brain_get(id)` and check the returned `content_hash` matches what the write returned. This is the read-after-write confirmation — do it with `brain_get`, never a semantic query.
|
||||
|
||||
**RULE: no semantic/vector brain query after the first `brain_update` in this run.** The batch shape makes this natural — read up front, write, confirm by id. If you ever find the skill wanting to semantic-search a just-superseded note, stop and flag it (that's the signal the staleness window matters and needs the synchronous-reembed follow-up).
|
||||
|
||||
**GATE — STOP, show the note (target wing/hall, new-vs-supersede, full content), get explicit confirmation before the brain write.**
|
||||
|
||||
After the note lands, if it relates to a note in another wing, create the cross-link inline with `brain_tunnel(source, target)` (idempotent; both paths brain-relative, must be in different wings). Optionally append a `session_log` entry (`session_id`, `skill: close-session`, `phase`, `final_status`) for telemetry. Both are now callable directly from Claude.ai — no Claude Code/Crush handoff needed.
|
||||
|
||||
## Phase 6 — Verdict
|
||||
|
||||
Deliver a final "safe to archive" verdict in the chat. Either:
|
||||
|
||||
- **SAFE TO ARCHIVE** — list what landed (issues closed/filed with numbers, summary path, brain note id, any tunnels) so the trail is auditable. Then list anything still in the user's queue (e.g. a PR awaiting their merge, a decision owed next session).
|
||||
- **NOT YET** — name the specific gate that wasn't passed or the write that failed, and what to do about it.
|
||||
|
||||
Never claim safe-to-archive if any gated write was declined or errored. The verdict is the skill's contract: if it says safe, the session can be lost without losing the work.
|
||||
|
||||
## Why the gates and the batch discipline matter
|
||||
|
||||
The whole point is durability across a context reset. Every gate is a place where a wrong write would silently corrupt the record (close the wrong issue, overwrite a good brain note, commit a half-truth). The batch read-discipline in Phase 5 exists because the brain's vector index refreshes out-of-band: write-then-semantically-reread in the same run can read stale, so the skill front-loads reads and confirms writes by id. Get those right and the skill does what it promises — nothing important is lost when the chat goes away.
|
||||
@@ -0,0 +1,248 @@
|
||||
# Capture capability — use-case & BDD specification
|
||||
|
||||
**Status:** Decisions resolved 2026-06-22 (§4). Ready for implementation scoping. `capture` is a
|
||||
privileged cross-harness write path touching brain + Gitea + ai-sessions.
|
||||
**Tracks:** hyperguild #49.
|
||||
**Governed by:** `infra/docs/architecture/01-invariants.md` (I1–I5), the admissibility test in
|
||||
`00-synthesis-model.md`, and the distributed-consolidation shape mandated by
|
||||
`brain/wiki/homelab/decisions/no-centralized-cross-harness-observer-2026-06-17.md`.
|
||||
|
||||
---
|
||||
|
||||
## 1. Use-case (Clean Architecture form)
|
||||
|
||||
**Name:** CaptureSession
|
||||
**Actor:** A harness acting on the user's behalf (claude.ai Chat/Cowork/Code/Design, Claude Code
|
||||
CLI, Crush, Pi, LLM Council, Agentsquad executor/reviewer) — or the user directly.
|
||||
**Goal:** Durably persist a finished session's valuable output — insights → brain, action items →
|
||||
Gitea tickets, optional summary → ai-sessions — with one uniform invocation, identical core
|
||||
behaviour across harnesses.
|
||||
|
||||
**Primary success scenario (essential steps):**
|
||||
1. Caller assembles capture input (insights, tickets, optional summary) + context (harness,
|
||||
session_ref, fidelity, actor, **data-classification**).
|
||||
2. System validates the whole request (fail-closed).
|
||||
3. System resolves **effective classification** (stricter of caller-declared and target-derived)
|
||||
and the **server-derived harness origin** (from the authenticated principal). It checks the
|
||||
**sovereignty gate** (I1): if effective classification is confidential AND the origin is a
|
||||
non-sovereign (us-nexus) surface, the capture is **refused** before any write.
|
||||
4. System persists insights (write or supersede), tickets (create/close/comment), summary — each
|
||||
best-effort, recording per-item outcome.
|
||||
5. System emits an **audit record** (I5) of who/what captured what, when, via which principal.
|
||||
6. System returns a structured, partial-aware receipt.
|
||||
|
||||
**Architectural shape:** the *logic* is a shared use-case (`CaptureService`), invoked **per-harness
|
||||
against the caller's own credentials** (distributed consolidation — no high-degree observer node).
|
||||
A central authenticated relay endpoint exists ONLY as a fallback for harnesses that cannot run the
|
||||
use-case in-process (Crush/Pi/headless); the relay holds no standing visibility and retains nothing
|
||||
beyond the I5 audit log.
|
||||
|
||||
---
|
||||
|
||||
## 2. Invariant obligations (acceptance gates, not nice-to-haves)
|
||||
|
||||
| Invariant | Obligation on `capture` |
|
||||
|---|---|
|
||||
| **I1 sovereign containment** | A confidential-classified session MUST NOT be captured through a us-nexus harness. Harness origin is **server-derived from the authenticated principal** (not caller-asserted). Classification uses **model (C)**: caller declares, server cross-checks the target's tag, **stricter wins**, mismatch logged. See §4.1–4.2. |
|
||||
| **I2 deliberate acceptance** | The *distributed-library* form opens no new acceptance. IF a central relay node is deployed, its cross-harness reach MUST be entered in `infra/docs/security-baseline.md` with Why-accepted / Revisit-if before it ships. |
|
||||
| **I3 GitOps reconcilability** | IF `capture` runs as a deployed service, its manifest lives under `infra/k3s/apps/**` (sovereign source, Flux-reconciled). No untracked runtime. |
|
||||
| **I4 decisions captured** | The distributed-vs-central decision and the intent-named-verb pattern are recorded (ADR + brain). |
|
||||
| **I5 auditability** | Every capture emits a request-level audit record (actor/principal, harness, items written, timestamp) to the alloy/loki substrate. **Classification-aware degradation** (§4.4): confidential + sink-down → hard-refuse; internal/public + sink-down → durable local buffer + ntfy + reconcile. Floor: refuse if nothing can record the audit. |
|
||||
|
||||
---
|
||||
|
||||
## 3. BDD scenarios (Gherkin)
|
||||
|
||||
```gherkin
|
||||
Feature: Capture session value uniformly across harnesses
|
||||
As an operator working across many AI harnesses
|
||||
I want one uniform command to persist insights and file tickets
|
||||
So that valuable session output is never lost and is always auditable
|
||||
|
||||
Background:
|
||||
Given a brain store, a Gitea issue tracker, and an ai-sessions summary writer
|
||||
And the caller is authenticated with a principal
|
||||
And the session context declares a harness, a fidelity, and a data classification
|
||||
|
||||
# --- Core happy path ---
|
||||
Scenario: Capture insights and tickets from a non-confidential session
|
||||
Given a session classified as "internal"
|
||||
And the capture input has 2 insights and 1 ticket to create
|
||||
When capture is invoked
|
||||
Then both insights are written to the brain and their ids and content hashes are returned
|
||||
And the ticket is created in the named repo under owner "mathias"
|
||||
And an audit record is emitted naming the principal, harness, and items written
|
||||
And the receipt reports every item as ok
|
||||
|
||||
# --- I1: sovereignty gate (the load-bearing refusal) ---
|
||||
# Harness origin is server-derived from the authenticated principal, never from context.harness.
|
||||
Scenario: Refuse capture of a confidential session through a us-nexus harness
|
||||
Given a session whose effective classification is "confidential"
|
||||
And the authenticated principal resolves to a us-nexus harness origin
|
||||
When capture is invoked
|
||||
Then the capture is refused before any write
|
||||
And no insight, ticket, or summary is persisted
|
||||
And the refusal names the sovereignty invariant as the reason
|
||||
|
||||
Scenario: Allow capture of a confidential session through a sovereign harness
|
||||
Given a session whose effective classification is "confidential"
|
||||
And the authenticated principal resolves to a sovereign-soil harness origin
|
||||
When capture is invoked
|
||||
Then the capture proceeds and persists normally
|
||||
|
||||
Scenario: Ignore a caller-asserted harness label and use the server-derived origin
|
||||
Given the request context asserts harness "sovereign-soil"
|
||||
But the authenticated principal resolves to a us-nexus origin
|
||||
And the session classification is "confidential"
|
||||
When capture is invoked
|
||||
Then the capture is refused
|
||||
And the server-derived origin is used, not the asserted label
|
||||
And the asserted-vs-derived discrepancy is logged as a security event
|
||||
|
||||
# --- I1: classification model (C) — stricter of declared vs target-derived wins ---
|
||||
Scenario: Take the stricter classification when caller and target disagree
|
||||
Given the caller declares classification "internal"
|
||||
But the target wing/repo is tagged "confidential"
|
||||
When capture is invoked
|
||||
Then the effective classification is "confidential"
|
||||
And the declared-vs-derived mismatch is logged as a security event
|
||||
And the I1 gate is evaluated against "confidential"
|
||||
|
||||
Scenario: Honour a caller raising sensitivity above the target's tag
|
||||
Given the caller declares classification "confidential"
|
||||
And the target wing/repo is tagged "internal"
|
||||
When capture is invoked
|
||||
Then the effective classification is "confidential"
|
||||
And the capture is gated as confidential
|
||||
|
||||
# --- Supersession + staleness discipline (reuses #45 / #47 resolution) ---
|
||||
Scenario: Supersede a prior insight rather than duplicating it
|
||||
Given an insight whose context names an existing note to supersede
|
||||
When capture is invoked
|
||||
Then the existing note is updated in place, not duplicated
|
||||
And the prior content hash is recorded in the superseding note
|
||||
And read-after-write confirmation uses a direct fetch, never a semantic query
|
||||
|
||||
# --- Validation: fail-closed ---
|
||||
Scenario: Reject a malformed request before any write
|
||||
Given a capture input with an invalid wing/hall or unknown repo
|
||||
When capture is invoked
|
||||
Then the request is rejected with a validation error
|
||||
And nothing is written to the brain, Gitea, or ai-sessions
|
||||
|
||||
# --- Partial failure: best-effort + honest receipt ---
|
||||
Scenario: Report partial success when one item fails mid-capture
|
||||
Given a capture input with 2 insights and 1 ticket
|
||||
And the second insight write will fail
|
||||
When capture is invoked
|
||||
Then the first insight and the ticket are persisted
|
||||
And the second insight is reported as failed in the receipt
|
||||
And no rollback is attempted
|
||||
And the audit record reflects exactly what landed
|
||||
|
||||
# --- Dry run ---
|
||||
Scenario: Preview a capture without writing
|
||||
Given a valid capture input with dry_run true
|
||||
When capture is invoked
|
||||
Then the would-be receipt is returned
|
||||
And nothing is written anywhere
|
||||
|
||||
# --- I5: auditability is classification-aware (confidential fails closed) ---
|
||||
Scenario: Confidential capture hard-refuses when the central audit sink is down
|
||||
Given the effective classification is "confidential"
|
||||
And the central audit substrate (loki) cannot be written to
|
||||
When capture is invoked
|
||||
Then the capture is refused before any write
|
||||
And the reason names the auditability invariant
|
||||
# Confidential work must be centrally auditable at write time — no buffered exception.
|
||||
|
||||
Scenario: Internal capture degrades to a durable local buffer when the sink is down
|
||||
Given the effective classification is "internal" or "public"
|
||||
And the central audit substrate (loki) cannot be written to
|
||||
When capture is invoked
|
||||
Then the capture proceeds
|
||||
And the audit record is written to a durable LOCAL fallback buffer
|
||||
And an ntfy alert is emitted naming the degraded audit state
|
||||
And the receipt flags that audit was buffered locally, not centrally recorded
|
||||
|
||||
Scenario: Locally buffered audit records reconcile to the central sink on recovery
|
||||
Given internal-tier audit records were buffered locally during a sink outage
|
||||
When the central audit substrate becomes reachable again
|
||||
Then the buffered records are replayed to the central sink
|
||||
And the local buffer is cleared only after confirmed central write
|
||||
|
||||
Scenario: Even internal capture refuses if neither sink nor local buffer can be written
|
||||
Given the effective classification is "internal" or "public"
|
||||
And neither the central sink nor the local fallback buffer can be written
|
||||
When capture is invoked
|
||||
Then the capture is refused
|
||||
And the reason names the auditability invariant
|
||||
# Degrade-and-warn has a floor: if NOTHING can record the audit, do not write.
|
||||
|
||||
# --- Summary fidelity (collision rule from the retro work) ---
|
||||
Scenario: A richer-fidelity summary supersedes a thinner one for the same session
|
||||
Given a summary already exists for session_ref X at fidelity "live-capture"
|
||||
And a new summary arrives for session_ref X at fidelity "transcript-parse"
|
||||
When capture is invoked
|
||||
Then the transcript-parse summary supersedes the live-capture one
|
||||
And the live-capture summary is not left as a contradicting duplicate
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 4. Resolved decisions (2026-06-22)
|
||||
|
||||
These were open questions at draft; resolved in the 2026-06-22 review session. Recorded here as
|
||||
binding design decisions for the build.
|
||||
|
||||
1. **Classification trust — model (C): caller-declares + server-cross-checks, stricter wins.**
|
||||
The caller declares `context.classification`; the server **independently derives** the target's
|
||||
classification (from the target wing/repo's classification tag) and gates on the **stricter of
|
||||
the two**. The caller can voluntarily *raise* sensitivity but can never *lower* it below the
|
||||
target's floor. A declared-vs-derived **mismatch is logged as a security event** (I5).
|
||||
- **Prerequisite (new build work):** a classification taxonomy (e.g. `public` /
|
||||
`internal` / `confidential`) and a per-wing / per-repo classification tag the server can read.
|
||||
This must exist before the I1 gate is load-bearing. Tracked as a sub-task of #49.
|
||||
- **Implemented (#50):** taxonomy `public < internal < confidential` (ordered so "stricter wins"
|
||||
is `max`) in `ingestion/internal/classification/`. Tags are read from an optional
|
||||
`classification.yaml` at the brain root (`wings:` / `repos:` maps); absent entries fall to
|
||||
built-in defaults (`client-*` → confidential; `hyperguild`/`homelab` → internal; everything
|
||||
else → **confidential, fail-safe**). `Config.Derive(Target)` is the function the use-case
|
||||
calls. See brain `wiki/hyperguild/decisions/capture-classification-taxonomy`.
|
||||
- Rationale: composes with decision 2; fails safe; honours a caller flagging something *more*
|
||||
sensitive than its destination. Pure caller-trust (A) was rejected — it makes the gate theatre.
|
||||
|
||||
2. **Sovereign-harness determination — server-derived, not caller-asserted.**
|
||||
"Is this harness us-nexus / sovereign?" is derived from the **authenticated principal/origin**
|
||||
(the OAuth2 identity), never from `context.harness`. `context.harness` survives only as a
|
||||
self-reported label for the audit log — descriptive telemetry, **never a gate input**. A control
|
||||
keyed on an attacker-suppliable value is not a control.
|
||||
|
||||
3. **Central relay — ships in v1, with the I2 ledger entry.**
|
||||
The relay is required, not optional: claude.ai (Chat/Cowork/Design), Crush, Pi, and LLM Council
|
||||
cannot run the use-case library in-process, and those are primary day-to-day surfaces. Deferring
|
||||
the relay would ship a capability that doesn't work from the interfaces actually in use. Because
|
||||
the relay is a (thin, no-standing-visibility, audit-only-retention) central node, its cross-harness
|
||||
reach **must be entered in `infra/docs/security-baseline.md`** with Why-accepted / Revisit-if
|
||||
**before it ships** (I2). That ledger entry is v1 work, not a follow-up.
|
||||
|
||||
4. **Audit-sink-down — classification-aware: confidential fails closed, internal/public degrades.**
|
||||
The posture inherits from the effective classification (decision 1), so there is one coherent
|
||||
sensitivity model rather than a separate availability policy:
|
||||
- **Confidential + central audit sink unreachable → hard-refuse.** No buffer, no proceed.
|
||||
Confidential work must be centrally auditable *at write time*; "buffer and reconcile later"
|
||||
introduces a buffer-integrity question (can a write tamper with its own pending audit record?)
|
||||
that must not exist for confidential data. The simplicity of "refuse" is itself the assurance
|
||||
asset — trivially true, nothing to poke holes in.
|
||||
- **Internal / public + central sink unreachable → degrade-and-warn** with a durable local buffer
|
||||
+ ntfy alert + reconcile-on-recovery (the earlier Q4 design, now scoped to lower tiers). Keeps
|
||||
capture available for your own homelab work during an observability outage; negligible risk
|
||||
since the buffered record is still durable and the data isn't client-confidential.
|
||||
- **Floor (all tiers):** if *nothing* — neither central sink nor (for internal/public) the local
|
||||
buffer — can record the audit, capture **refuses**. No tier writes wholly un-audited.
|
||||
- Rationale: matches assurance cost to data sensitivity, exactly as the I1/sovereignty model
|
||||
does for placement. Presentable to a due-diligence client as "audit posture is
|
||||
classification-aware: confidential fails closed, internal degrades gracefully" — which
|
||||
demonstrates the judgment, not just a binary. Couples Q4 to Q1's classification machinery
|
||||
(being built anyway) and removes the buffer-integrity rabbit hole for the only tier where it
|
||||
mattered.
|
||||
Reference in New Issue
Block a user