Files
OpenFin/backend/services/pattern_lab.py
OpenSquared 303ecc2a3a feat: instrument picker + Pattern Lab instrument scan mode
- instruments.ts: 90 IB-options-tradable instruments in 12 categories
  (US Indices, Europe, Asia, EM, Sectors, Forex, Bonds, Metals, Energy,
   Agriculture, Crypto, Volatility) — EUR/CHF, Cotton, etc. all included

- PatternExplorer: replace text input in Instrument Lens with categorised
  grid picker (category pill filters + search + custom ticker fallback)

- PatternLab: add Instrument Scan tab alongside Event Presets
  - Pick any instrument from the shared categorised picker
  - Set period (start/end date) + horizon per pattern
  - AI scans the full period: identifies 4-6 key pattern instances each with
    their own entry date, expected move, strategy
  - 'Evaluate outcomes' fetches actual price at T+horizon per pattern
  - 'Save pattern' promotes any instance to the Pattern Library

- backend/services/pattern_lab.py: run_instrument_scan() + evaluate_instrument_outcomes()
  (per-pattern analysis_date vs shared date in event mode)
- backend/routers/pattern_lab.py: POST /instrument-scan + POST /evaluate-instrument/{id}
- useApi.ts: useInstrumentScan + useEvaluateInstrumentScan hooks

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-06-22 18:20:22 +02:00

431 lines
17 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""
Pattern Lab — Historical backtest engine for pattern discovery.
Workflow:
1. build_historical_context(date, tickers) — pull prices + technicals via yfinance
2. run_ai_backtest(context, hint, horizon) — GPT-4o acts as analyst on that past date
3. evaluate_outcomes(run) — fetch actual prices at T+horizon, score each idea
"""
import json
import logging
from datetime import datetime, timedelta
from typing import Optional
import numpy as np
import pandas as pd
import yfinance as yf
_log = logging.getLogger(__name__)
# ── Context builder ────────────────────────────────────────────────────────────
def _fetch_ticker(ticker: str, start: str, end: str) -> Optional[pd.DataFrame]:
try:
df = yf.download(ticker, start=start, end=end, progress=False, auto_adjust=True)
if isinstance(df.columns, pd.MultiIndex):
df.columns = df.columns.get_level_values(0)
return df if not df.empty else None
except Exception as e:
_log.warning(f"[PatternLab] yfinance error for {ticker}: {e}")
return None
def _rsi(closes: pd.Series, period: int = 14) -> Optional[float]:
if len(closes) < period + 1:
return None
delta = closes.diff()
gain = delta.clip(lower=0).rolling(period).mean()
loss = (-delta.clip(upper=0)).rolling(period).mean()
rs = gain / loss
val = 100 - (100 / (1 + rs.iloc[-1]))
return round(float(val), 1) if not np.isnan(val) else None
def build_historical_context(analysis_date: str, tickers: list) -> dict:
dt = datetime.strptime(analysis_date, "%Y-%m-%d")
hist_start = (dt - timedelta(days=400)).strftime("%Y-%m-%d")
hist_end = (dt + timedelta(days=3)).strftime("%Y-%m-%d")
context: dict = {"analysis_date": analysis_date, "assets": {}}
all_tickers = list(tickers)
if "^VIX" not in all_tickers:
all_tickers.append("^VIX")
for ticker in all_tickers:
df = _fetch_ticker(ticker, hist_start, hist_end)
if df is None or "Close" not in df.columns:
context["assets"][ticker] = {"error": "no data"}
continue
df.index = pd.to_datetime(df.index).tz_localize(None)
sub = df[df.index.normalize() <= dt].copy()
if sub.empty:
context["assets"][ticker] = {"error": "no data before date"}
continue
closes = sub["Close"].dropna()
price = float(closes.iloc[-1])
def ret(days: int) -> Optional[float]:
past = sub[sub.index.normalize() <= dt - timedelta(days=days)]
if past.empty:
return None
p0 = float(past["Close"].dropna().iloc[-1])
return round((price / p0 - 1) * 100, 2) if p0 else None
ma50 = round(float(closes.tail(50).mean()), 2) if len(closes) >= 50 else None
ma200 = round(float(closes.tail(200).mean()), 2) if len(closes) >= 200 else None
context["assets"][ticker] = {
"price": round(price, 4),
"ret_1w": ret(5),
"ret_1m": ret(21),
"ret_3m": ret(63),
"ret_ytd": ret(int((dt - datetime(dt.year, 1, 1)).days)),
"rsi_14": _rsi(closes.tail(60)),
"ma50": ma50,
"ma200": ma200,
"pct_from_ma200": round((price / ma200 - 1) * 100, 2) if ma200 else None,
}
return context
# ── AI analysis ────────────────────────────────────────────────────────────────
async def run_ai_backtest(context: dict, theme_hint: str, horizon_days: int, openai_key: str) -> dict:
from openai import AsyncOpenAI
client = AsyncOpenAI(api_key=openai_key)
analysis_date = context["analysis_date"]
assets = context.get("assets", {})
lines = []
for ticker, d in assets.items():
if "error" in d:
continue
line = f" {ticker}: {d.get('price', '?')}"
if d.get("ret_1w") is not None: line += f" 1W:{d['ret_1w']:+.1f}%"
if d.get("ret_1m") is not None: line += f" 1M:{d['ret_1m']:+.1f}%"
if d.get("ret_3m") is not None: line += f" 3M:{d['ret_3m']:+.1f}%"
if d.get("rsi_14") is not None: line += f" RSI:{d['rsi_14']:.0f}"
if d.get("pct_from_ma200") is not None: line += f" vsMA200:{d['pct_from_ma200']:+.1f}%"
lines.append(line)
market_block = "\n".join(lines) if lines else "No market data available."
prompt = f"""You are a senior macro/options trader. Today is {analysis_date}.
You have full knowledge of what was happening in markets and geopolitics on this specific date.
## REAL MARKET DATA — {analysis_date}
{market_block}
## CONTEXT / THEME
{theme_hint}
## INSTRUCTIONS
Using the market data above AND your knowledge of the historical context on {analysis_date}:
- Identify 2 to 4 high-conviction trade patterns that were present at this time.
- For each pattern, recommend the best options/futures strategy.
- The investment horizon is {horizon_days} days from {analysis_date}.
For each pattern you MUST specify:
- The primary underlying ticker (must be a real, liquid, yfinance-compatible symbol)
- The expected percentage move in the UNDERLYING over {horizon_days} days
- The expected direction: "up", "down", or "any" (for volatility plays)
- A confidence score 0100
Return ONLY this valid JSON (no markdown, no explanation):
{{
"patterns": [
{{
"name": "Short descriptive pattern name",
"description": "What you observed and why this matters",
"category": "géopolitique|macro_monétaire|technique|commodités_supply|risk_off|flux_saisonnier|géo_économique|crédit_stress",
"signal_direction": "bullish|bearish|volatility|neutral",
"underlying": "TICKER",
"strategy": "e.g. long put, long call, straddle, call spread",
"expected_move_pct": 12.5,
"expected_direction": "up|down|any",
"horizon_days": {horizon_days},
"confidence": 72,
"rationale": "Concise explanation of the trade thesis at this historical moment"
}}
]
}}"""
try:
resp = await client.chat.completions.create(
model="gpt-4o",
messages=[{"role": "user", "content": prompt}],
temperature=0.2,
response_format={"type": "json_object"},
timeout=60,
)
raw = resp.choices[0].message.content or "{}"
return json.loads(raw)
except Exception as e:
_log.error(f"[PatternLab] AI backtest failed: {e}")
return {"patterns": [], "error": str(e)}
# ── Outcome evaluator ──────────────────────────────────────────────────────────
def evaluate_outcomes(run: dict) -> list:
analysis_date = run["analysis_date"]
horizon_days = int(run["horizon_days"])
ai_result = json.loads(run.get("ai_result") or "{}")
patterns = ai_result.get("patterns", [])
dt = datetime.strptime(analysis_date, "%Y-%m-%d")
eval_dt = dt + timedelta(days=horizon_days)
# Don't evaluate if eval date is in the future
if eval_dt.date() >= datetime.utcnow().date():
return [{"error": f"Evaluation date {eval_dt.date()} is in the future"}]
fetch_start = (dt - timedelta(days=5)).strftime("%Y-%m-%d")
fetch_end = (eval_dt + timedelta(days=5)).strftime("%Y-%m-%d")
outcomes = []
for pat in patterns:
ticker = pat.get("underlying", "")
if not ticker:
continue
df = _fetch_ticker(ticker, fetch_start, fetch_end)
if df is None or "Close" not in df.columns:
outcomes.append({"pattern_name": pat.get("name"), "underlying": ticker, "error": "no data"})
continue
df.index = pd.to_datetime(df.index).tz_localize(None)
entry_sub = df[df.index.normalize() <= dt]
exit_sub = df[df.index.normalize() <= eval_dt]
if entry_sub.empty or exit_sub.empty:
outcomes.append({"pattern_name": pat.get("name"), "underlying": ticker, "error": "insufficient history"})
continue
entry_price = float(entry_sub["Close"].dropna().iloc[-1])
exit_price = float(exit_sub["Close"].dropna().iloc[-1])
actual_move = round((exit_price / entry_price - 1) * 100, 2)
expected_dir = pat.get("expected_direction", "any")
expected_move = float(pat.get("expected_move_pct", 0))
threshold = max(expected_move * 0.5, 3.0) # at least 50% of expected, min 3%
if expected_dir == "up":
hit = actual_move >= threshold
elif expected_dir == "down":
hit = actual_move <= -threshold
else: # any / volatility
hit = abs(actual_move) >= threshold
outcomes.append({
"pattern_name": pat.get("name"),
"underlying": ticker,
"strategy": pat.get("strategy", ""),
"signal_direction": pat.get("signal_direction", ""),
"expected_direction": expected_dir,
"expected_move_pct": expected_move,
"actual_move_pct": actual_move,
"entry_price": round(entry_price, 4),
"exit_price": round(exit_price, 4),
"entry_date": analysis_date,
"exit_date": eval_dt.strftime("%Y-%m-%d"),
"hit": hit,
"confidence": pat.get("confidence", 0),
})
return outcomes
# ── Instrument Scan ────────────────────────────────────────────────────────────
async def run_instrument_scan(
ticker: str, instrument_name: str,
start_date: str, end_date: str,
horizon_days: int, openai_key: str,
) -> dict:
"""
Analyse a specific instrument over a historical period.
The AI identifies 4-6 key pattern instances (each with its own entry date)
that drove significant moves in the underlying.
"""
from openai import AsyncOpenAI
client = AsyncOpenAI(api_key=openai_key)
# Fetch full period price history
fetch_start = (datetime.strptime(start_date, "%Y-%m-%d") - timedelta(days=5)).strftime("%Y-%m-%d")
fetch_end = (datetime.strptime(end_date, "%Y-%m-%d") + timedelta(days=5)).strftime("%Y-%m-%d")
df = _fetch_ticker(ticker, fetch_start, fetch_end)
monthly_lines = []
weekly_spikes = []
if df is not None and "Close" in df.columns:
df.index = pd.to_datetime(df.index).tz_localize(None)
s_dt = pd.Timestamp(start_date)
e_dt = pd.Timestamp(end_date)
sub = df[(df.index >= s_dt) & (df.index <= e_dt)].copy()
if not sub.empty:
# Monthly returns
monthly = sub["Close"].resample("ME").last().dropna()
for i in range(1, len(monthly)):
ret = (monthly.iloc[i] / monthly.iloc[i - 1] - 1) * 100
monthly_lines.append(f" {monthly.index[i].strftime('%Y-%m')}: {ret:+.1f}% (price {monthly.iloc[i]:.2f})")
# Weekly moves > 3%
weekly = sub["Close"].resample("W").last().dropna()
for i in range(1, len(weekly)):
wret = (weekly.iloc[i] / weekly.iloc[i - 1] - 1) * 100
if abs(wret) >= 3.0:
weekly_spikes.append(
f" week of {weekly.index[i].strftime('%Y-%m-%d')}: {wret:+.1f}%"
)
monthly_block = "\n".join(monthly_lines) if monthly_lines else " (no monthly data)"
spikes_block = "\n".join(weekly_spikes[:20]) if weekly_spikes else " (none > 3%)"
prompt = f"""You are a senior macro/options trader with deep historical market knowledge.
Instrument: {instrument_name} [{ticker}]
Period: {start_date}{end_date}
## ACTUAL MONTHLY PRICE RETURNS
{monthly_block}
## SIGNIFICANT WEEKLY MOVES (>3%)
{spikes_block}
## YOUR TASK
Based on the actual price data above and your historical knowledge of this period:
1. Identify 4 to 6 DISTINCT pattern/event instances that drove the most significant moves in {instrument_name}
2. Each pattern must have a specific ENTRY DATE (the trading signal date) within the period
3. Each pattern must be unique — different catalysts on different dates
4. The trade horizon per pattern is {horizon_days} days
For each pattern instance you MUST specify:
- `analysis_date`: the exact signal/entry date (YYYY-MM-DD, must be within {start_date} to {end_date})
- `underlying`: use "{ticker}" as the underlying
- `expected_direction`: "up", "down", or "any" (for volatility)
- `expected_move_pct`: realistic % move over {horizon_days} days (consistent with what actually happened)
- `confidence`: 0-100
Return ONLY this valid JSON:
{{
"instrument": "{instrument_name}",
"ticker": "{ticker}",
"period": "{start_date} to {end_date}",
"patterns": [
{{
"name": "Pattern name",
"description": "What caused this move and why it was tradeable",
"category": "géopolitique|macro_monétaire|technique|commodités_supply|risk_off|flux_saisonnier|géo_économique|crédit_stress",
"signal_direction": "bullish|bearish|volatility|neutral",
"analysis_date": "YYYY-MM-DD",
"underlying": "{ticker}",
"strategy": "long call|long put|straddle|call spread|put spread|...",
"expected_direction": "up|down|any",
"expected_move_pct": 5.0,
"horizon_days": {horizon_days},
"confidence": 75,
"rationale": "Why this was a high-conviction trade at this specific moment"
}}
]
}}"""
try:
resp = await client.chat.completions.create(
model="gpt-4o",
messages=[{"role": "user", "content": prompt}],
temperature=0.2,
response_format={"type": "json_object"},
timeout=90,
)
raw = resp.choices[0].message.content or "{}"
return json.loads(raw)
except Exception as e:
_log.error(f"[PatternLab] Instrument scan AI failed: {e}")
return {"patterns": [], "error": str(e)}
def evaluate_instrument_outcomes(run: dict) -> list:
"""
Like evaluate_outcomes but each pattern has its own analysis_date.
Used for instrument scans where the AI generates time-distributed pattern instances.
"""
ai_result = json.loads(run.get("ai_result") or "{}")
patterns = ai_result.get("patterns", [])
default_horizon = int(run["horizon_days"])
outcomes = []
for pat in patterns:
ticker = pat.get("underlying", "")
pat_date = pat.get("analysis_date") or run["analysis_date"]
pat_horizon = int(pat.get("horizon_days") or default_horizon)
if not ticker or not pat_date:
continue
try:
dt = datetime.strptime(pat_date, "%Y-%m-%d")
eval_dt = dt + timedelta(days=pat_horizon)
except ValueError:
continue
if eval_dt.date() >= datetime.utcnow().date():
outcomes.append({"pattern_name": pat.get("name"), "underlying": ticker,
"analysis_date": pat_date, "error": "eval date in the future"})
continue
fetch_start = (dt - timedelta(days=5)).strftime("%Y-%m-%d")
fetch_end = (eval_dt + timedelta(days=5)).strftime("%Y-%m-%d")
df = _fetch_ticker(ticker, fetch_start, fetch_end)
if df is None or "Close" not in df.columns:
outcomes.append({"pattern_name": pat.get("name"), "underlying": ticker,
"analysis_date": pat_date, "error": "no data"})
continue
df.index = pd.to_datetime(df.index).tz_localize(None)
entry_sub = df[df.index.normalize() <= dt]
exit_sub = df[df.index.normalize() <= eval_dt]
if entry_sub.empty or exit_sub.empty:
outcomes.append({"pattern_name": pat.get("name"), "underlying": ticker,
"analysis_date": pat_date, "error": "insufficient history"})
continue
entry_price = float(entry_sub["Close"].dropna().iloc[-1])
exit_price = float(exit_sub["Close"].dropna().iloc[-1])
actual_move = round((exit_price / entry_price - 1) * 100, 2)
expected_dir = pat.get("expected_direction", "any")
expected_move = float(pat.get("expected_move_pct", 0))
threshold = max(expected_move * 0.5, 3.0)
if expected_dir == "up":
hit = actual_move >= threshold
elif expected_dir == "down":
hit = actual_move <= -threshold
else:
hit = abs(actual_move) >= threshold
outcomes.append({
"pattern_name": pat.get("name"),
"underlying": ticker,
"strategy": pat.get("strategy", ""),
"signal_direction": pat.get("signal_direction", ""),
"expected_direction": expected_dir,
"expected_move_pct": expected_move,
"actual_move_pct": actual_move,
"entry_price": round(entry_price, 4),
"exit_price": round(exit_price, 4),
"analysis_date": pat_date,
"entry_date": pat_date,
"exit_date": eval_dt.strftime("%Y-%m-%d"),
"hit": hit,
"confidence": pat.get("confidence", 0),
})
return outcomes