"""train.py — autoresearch agent file (only this may be edited). TS-JEPA backbone with SIGReg regularization (Balestriero & LeCun, LeJEPA arXiv:2511.08544; time-series placement from ChronoJEPA arXiv: 2505.XXXXX). PatchTST-style encoder over windowed daily [return, realized_vol] → FREEZE → linear probe predicts NEXT-day realized vol → val_vol_r2 (OOS R²). Writes metrics.json — the single scalar the loop reads. Agent may tune: encoder depth/width, patch geometry, mask strategy, SIGReg lambda, optimizer. Do NOT touch prepare_data.py, loop.py, or the data pipeline. """ import json import math import numpy as np import pandas as pd import torch import torch.nn as nn # --- agent-tunable knobs --- WINDOW = 60 # INCREASED lookback for better volatility persistence capture PATCH_LEN = 5 # time-patch size (must divide WINDOW) STRIDE = 5 D_MODEL = 64 # transformer hidden dim - INCREASED for capacity DEPTH = 2 # transformer layers N_HEADS = 4 MASK_FRAC = 0.50 # INCREASED mask fraction to force the encoder to learn better global representations SIGREG_LAM = 0.01 # SIGReg weight (λ) - REDUCED to allow more representation capacity EPOCHS = 300 LR = 3e-4 SEED = 0 # --------------------------- torch.manual_seed(SEED) np.random.seed(SEED) dev = "cuda" if torch.cuda.is_available() else "cpu" # ── SIGReg (from LeJEPA/ChronoJEPA, token-level placement) ───────────────── def sigreg(tokens: torch.Tensor, knots: int = 17) -> torch.Tensor: """Epps-Pulley test statistic pushes token embeddings toward isotropic Gaussian. tokens: (B, T, D) — applied per-token, averaged across B and T. """ B, T, D = tokens.shape z = tokens.reshape(B * T, D) # (N, D) t = torch.linspace(0, 3, knots, device=z.device, dtype=z.float().dtype) dt = 3.0 / (knots - 1) w = torch.full((knots,), 2 * dt, device=z.device, dtype=z.float().dtype) w[0] = dt; w[-1] = dt phi = torch.exp(-t.square() / 2.0) A = torch.randn(D, 256, device=z.device, dtype=z.float().dtype) A = A / A.norm(p=2, dim=0) x_t = (z.float() @ A).unsqueeze(-1) * t # (N, 256, knots) err = (x_t.cos().mean(0) - phi).square() + x_t.sin().mean(0).square() return ((err @ (w * phi)) * z.shape[0]).mean() # ── Encoder + Predictor ───────────────────────────────────────────────────── class PatchEncoder(nn.Module): """PatchTST-style encoder for univariate windows.""" def __init__(self, in_feats, patch_len, stride, d_model, depth, n_heads): super().__init__() self.patch_len = patch_len self.stride = stride self.d_model = d_model self.embed = nn.Linear(patch_len * in_feats, d_model) layer = nn.TransformerEncoderLayer(d_model, n_heads, 2 * d_model, dropout=0.0, batch_first=True) self.tf = nn.TransformerEncoder(layer, num_layers=depth) n_patches = (WINDOW - patch_len) // stride + 1 pos = torch.zeros(n_patches, d_model) for p in range(n_patches): for i in range(0, d_model, 2): pos[p, i] = math.sin(p / 10000 ** (i / d_model)) if i + 1 < d_model: pos[p, i+1] = math.cos(p / 10000 ** (i / d_model)) self.register_buffer("pos", pos) def forward(self, x: torch.Tensor) -> torch.Tensor: # x: (B, W, F) → patches → (B, T, D) B, W, F = x.shape n_patches = (W - self.patch_len) // self.stride + 1 patches = torch.stack([x[:, i*self.stride:i*self.stride+self.patch_len, :] .reshape(B, -1) for i in range(n_patches)], dim=1) tokens = self.embed(patches) + self.pos[:n_patches] return self.tf(tokens) # (B, T, D) class Predictor(nn.Module): def __init__(self, d_model): super().__init__() self.net = nn.Sequential(nn.Linear(d_model, d_model), nn.GELU(), nn.Linear(d_model, d_model)) def forward(self, x): return self.net(x) # ── Data ──────────────────────────────────────────────────────────────────── def build(): """Year-based split: encoder trains on 2019-2021; probe evaluates on 2022-2023 OOS.""" df = pd.read_parquet("data/processed/eurusd_daily.parquet").reset_index(drop=True) df["date"] = pd.to_datetime(df["date"]) feats = df[["ret", "realized_vol"]].to_numpy(np.float32) target = df["realized_vol"].to_numpy(np.float32) tr_idx = df.index[df["date"].dt.year <= 2021].tolist() te_idx = df.index[df["date"].dt.year >= 2022].tolist() mu = feats[:tr_idx[-1]+1].mean(0) sd = feats[:tr_idx[-1]+1].std(0) + 1e-8 fn = (feats - mu) / sd def windows(idx): X, y = [], [] for t in idx: if t - WINDOW >= 0 and t + 1 < len(df): X.append(fn[t - WINDOW:t]); y.append(target[t + 1]) return np.stack(X).astype(np.float32), np.array(y, np.float32) return windows(tr_idx), windows(te_idx) # ── Training ───────────────────────────────────────────────────────────────── def main(): (Xtr, ytr), (Xte, yte) = build() n_feats = Xtr.shape[2] Xtr_t = torch.tensor(Xtr, device=dev) enc = PatchEncoder(n_feats, PATCH_LEN, STRIDE, D_MODEL, DEPTH, N_HEADS).to(dev) pred = Predictor(D_MODEL).to(dev) opt = torch.optim.AdamW(list(enc.parameters()) + list(pred.parameters()), lr=LR) n_patches = (WINDOW - PATCH_LEN) // STRIDE + 1 n_mask = max(1, int(MASK_FRAC * n_patches)) for ep in range(EPOCHS): # JEPA: predict masked-out patch tokens from visible tokens idx_mask = torch.randperm(n_patches)[:n_mask] ctx_mask = torch.ones(n_patches, dtype=torch.bool, device=dev) ctx_mask[idx_mask] = False tokens_ctx = enc(Xtr_t) # encode all (B, T, D) tokens_target = enc(Xtr_t).detach() # target (frozen): same input, no grad pred_out = pred(tokens_ctx[:, idx_mask, :]) jepa_loss = ((pred_out - tokens_target[:, idx_mask, :]) ** 2).mean() reg_loss = sigreg(tokens_ctx) loss = jepa_loss + SIGREG_LAM * reg_loss opt.zero_grad(); loss.backward(); opt.step() enc.eval() with torch.no_grad(): def embed(X_np): t = torch.tensor(X_np, device=dev) return enc(t).mean(1).cpu().numpy() # pool over time patches Etr = embed(Xtr) Ete = embed(Xte) # ridge linear probe (closed form) A = np.hstack([Etr, np.ones((len(Etr), 1))]) w = np.linalg.solve(A.T @ A + 1e-3 * np.eye(A.shape[1]), A.T @ ytr) pred_np = np.hstack([Ete, np.ones((len(Ete), 1))]) @ w ss_res = ((yte - pred_np) ** 2).sum() ss_tot = ((yte - yte.mean()) ** 2).sum() val_vol_r2 = float(1 - ss_res / ss_tot) json.dump({ "val_vol_r2": val_vol_r2, "n_test": len(yte), "knobs": {"WINDOW": WINDOW, "PATCH_LEN": PATCH_LEN, "STRIDE": STRIDE, "D_MODEL": D_MODEL, "DEPTH": DEPTH, "MASK_FRAC": MASK_FRAC, "SIGREG_LAM": SIGREG_LAM, "EPOCHS": EPOCHS}, }, open("metrics.json", "w"), indent=2) print("val_vol_r2 = %.4f (n_test=%d, dev=%s)" % (val_vol_r2, len(yte), dev)) # ── EXPORT BLOCK — do NOT edit (agent boundary) ────────────────────────── # Set EXPORT_EMBEDDINGS=1 to write embeddings.json for the Go eval harness. # Uses year-based split (train≤2021, OOS≥2022) regardless of probe split. import os if os.environ.get("EXPORT_EMBEDDINGS") == "1": df2 = pd.read_parquet("data/processed/eurusd_daily.parquet").reset_index(drop=True) df2["date"] = pd.to_datetime(df2["date"]) tr_mask = df2["date"].dt.year <= 2021 feats2 = df2[["ret", "realized_vol"]].to_numpy(np.float32) mu2 = feats2[tr_mask].mean(0); sd2 = feats2[tr_mask].std(0) + 1e-8 fn2 = (feats2 - mu2) / sd2 def _export_windows(year_mask): idx = df2.index[year_mask].tolist() Xs, dates, rvs = [], [], [] for t in idx: if t - WINDOW >= 0: Xs.append(fn2[t - WINDOW:t]) dates.append(str(df2["date"].iloc[t].date())) rvs.append(float(df2["realized_vol"].iloc[t])) if not Xs: return [], [], [] with torch.no_grad(): E = enc(torch.tensor(np.stack(Xs), device=dev)).mean(1).cpu().numpy().tolist() return E, dates, rvs Etr, dates_tr, rv_tr = _export_windows(tr_mask) Eoos, dates_oos, rv_oos = _export_windows(df2["date"].dt.year >= 2022) hv_thr = float(np.percentile(rv_oos, 67)) hv_label = [1 if v >= hv_thr else 0 for v in rv_oos] json.dump({"embeddings": Eoos, "dates": dates_oos, "realized_vol": rv_oos, "hv_label": hv_label, "train_embeddings": Etr, "train_realized_vol": rv_tr}, open("embeddings.json", "w")) print("exported embeddings.json train=%d oos=%d HV=%d/%d" % ( len(Etr), len(Eoos), sum(hv_label), len(hv_label))) # ── END EXPORT BLOCK ───────────────────────────────────────────────────── if __name__ == "__main__": main()