feat: FF HTML scraper for upcoming weeks with forecasts
- scrape_upcoming(weeks_ahead=5) in ff_calendar.py: fetches forexfactory.com/calendar?week=... HTML for N weeks ahead, parses calendar__table (date/time/currency/impact/event/forecast/previous), converts ET times to UTC, upserts into ff_calendar - Daily scheduler in main.py: runs scrape_upcoming at startup (after 30s delay) then every 24h — no manual action needed - New endpoints: POST /api/eco/ff-scrape?weeks=5, GET /api/eco/ff-scrape/status - CalendarPage: "Scrape Upcoming (5w)" button (indigo) with polling + result Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
@@ -95,6 +95,22 @@ def startup():
|
||||
from services.institutional_scheduler import start_institutional_scheduler
|
||||
start_institutional_scheduler()
|
||||
|
||||
# Daily FF scraper — runs once at startup, then every 24h
|
||||
import threading, time as _time
|
||||
def _ff_scrape_loop():
|
||||
_time.sleep(30) # let server fully boot first
|
||||
while True:
|
||||
try:
|
||||
print("[Startup] FF scrape upcoming weeks …", flush=True)
|
||||
from services.ff_calendar import scrape_upcoming
|
||||
result = scrape_upcoming(weeks_ahead=5)
|
||||
print(f"[FF scrape] done: {result.get('total_upserted', 0)} upserted", flush=True)
|
||||
except Exception as e:
|
||||
print(f"[FF scrape] loop error: {e}", flush=True)
|
||||
_time.sleep(86400) # 24h
|
||||
|
||||
threading.Thread(target=_ff_scrape_loop, daemon=True).start()
|
||||
|
||||
# Auto-import Forex Factory CSV if file is present (force, then delete CSV)
|
||||
import threading
|
||||
def _auto_import_ff():
|
||||
|
||||
@@ -295,6 +295,7 @@ def upcoming_events() -> List[Dict[str, Any]]:
|
||||
|
||||
_ff_import_status: Dict[str, Any] = {"running": False, "last_result": None}
|
||||
_ff_sync_status: Dict[str, Any] = {"running": False, "last_result": None}
|
||||
_ff_scrape_status: Dict[str, Any] = {"running": False, "last_result": None}
|
||||
|
||||
# CSV search order: Docker image (/app), uploaded to /tmp, local dev paths
|
||||
_FF_CSV_CANDIDATES = [
|
||||
@@ -387,6 +388,37 @@ def ff_sync_status_ep() -> Dict[str, Any]:
|
||||
return _ff_sync_status
|
||||
|
||||
|
||||
def _run_ff_scrape(weeks_ahead: int = 5):
|
||||
global _ff_scrape_status
|
||||
_ff_scrape_status["running"] = True
|
||||
try:
|
||||
from services.ff_calendar import scrape_upcoming
|
||||
result = scrape_upcoming(weeks_ahead=weeks_ahead)
|
||||
_ff_scrape_status["last_result"] = result
|
||||
except Exception as e:
|
||||
logger.error(f"[eco/ff-scrape] Failed: {e}")
|
||||
_ff_scrape_status["last_result"] = {"error": str(e)}
|
||||
finally:
|
||||
_ff_scrape_status["running"] = False
|
||||
|
||||
|
||||
@router.post("/ff-scrape")
|
||||
def ff_scrape(
|
||||
background_tasks: BackgroundTasks,
|
||||
weeks: int = Query(5, ge=1, le=8, description="Number of weeks ahead to scrape"),
|
||||
) -> Dict[str, Any]:
|
||||
"""Scrape forexfactory.com HTML calendar for upcoming N weeks (incl. forecasts)."""
|
||||
if _ff_scrape_status["running"]:
|
||||
raise HTTPException(409, "Scrape already running")
|
||||
background_tasks.add_task(_run_ff_scrape, weeks)
|
||||
return {"status": "started", "weeks_ahead": weeks}
|
||||
|
||||
|
||||
@router.get("/ff-scrape/status")
|
||||
def ff_scrape_status_ep() -> Dict[str, Any]:
|
||||
return _ff_scrape_status
|
||||
|
||||
|
||||
@router.get("/calendar")
|
||||
def ff_calendar(
|
||||
period: str = Query("recent", description="recent|today|tomorrow|yesterday|this_week|next_week|previous_week|this_month|next_month|previous_month"),
|
||||
|
||||
@@ -1,18 +1,23 @@
|
||||
"""
|
||||
Forex Factory calendar service.
|
||||
- CSV bulk import from forex_factory_cache.csv (2007-2025)
|
||||
- Live sync from nfs.faireconomy.media JSON endpoint (this week / next week)
|
||||
- Live sync from nfs.faireconomy.media JSON endpoint (this week)
|
||||
- HTML scraper for upcoming weeks (forexfactory.com/calendar?week=...)
|
||||
"""
|
||||
import csv
|
||||
import json
|
||||
import logging
|
||||
from datetime import datetime, timezone, timedelta
|
||||
from datetime import datetime, date, timezone, timedelta
|
||||
from typing import Any, Dict, List, Optional
|
||||
from zoneinfo import ZoneInfo
|
||||
|
||||
import httpx
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_ET = ZoneInfo("America/New_York") # FF publishes times in US Eastern
|
||||
|
||||
_FF_THIS_WEEK_URL = "https://nfs.faireconomy.media/ff_calendar_thisweek.json"
|
||||
_FF_NEXT_WEEK_URL = "https://nfs.faireconomy.media/ff_calendar_nextweek.json"
|
||||
|
||||
@@ -343,3 +348,210 @@ def get_calendar(
|
||||
}
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
|
||||
# ── FF HTML scraper for upcoming weeks ────────────────────────────────────────
|
||||
#
|
||||
# Forex Factory renders its calendar server-side; the HTML table is parseable
|
||||
# without JavaScript. Times are in US Eastern Time (ET).
|
||||
#
|
||||
# URL pattern: https://www.forexfactory.com/calendar?week=jun26.2026
|
||||
# (any date in the week works; we use each Monday)
|
||||
|
||||
_FF_BASE = "https://www.forexfactory.com"
|
||||
|
||||
_FF_HEADERS = {
|
||||
"User-Agent": (
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
||||
"AppleWebKit/537.36 (KHTML, like Gecko) "
|
||||
"Chrome/125.0.0.0 Safari/537.36"
|
||||
),
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
||||
"Accept-Language": "en-US,en;q=0.9",
|
||||
"Accept-Encoding": "gzip, deflate, br",
|
||||
"Connection": "keep-alive",
|
||||
"Upgrade-Insecure-Requests": "1",
|
||||
"Referer": "https://www.forexfactory.com/",
|
||||
}
|
||||
|
||||
# FF impact icon CSS class suffix → our normalized impact label
|
||||
_FF_IMPACT_CSS = {
|
||||
"red": "high",
|
||||
"ora": "medium", # orange
|
||||
"yel": "low", # yellow
|
||||
"gra": "non-economic",
|
||||
}
|
||||
|
||||
|
||||
def _week_url(monday: date) -> str:
|
||||
"""Return FF calendar URL for the week containing 'monday'."""
|
||||
month = monday.strftime("%b").lower() # "jun", "jul", …
|
||||
return f"{_FF_BASE}/calendar?week={month}{monday.day}.{monday.year}"
|
||||
|
||||
|
||||
def _parse_et_time(time_str: str, ref_date: date) -> tuple[str, str]:
|
||||
"""
|
||||
Parse an FF time string (e.g. '8:30am', 'Tentative', 'All Day')
|
||||
expressed in US Eastern Time and return (event_date_utc, event_time_utc).
|
||||
"""
|
||||
s = (time_str or "").strip().lower()
|
||||
if not s or s in ("all day", "tentative", ""):
|
||||
return str(ref_date), "00:00"
|
||||
try:
|
||||
# FF uses "8:30am" / "12:45pm" format
|
||||
fmt = "%I:%M%p" if ":" in s else "%I%p"
|
||||
t = datetime.strptime(s, fmt)
|
||||
dt_et = datetime(
|
||||
ref_date.year, ref_date.month, ref_date.day,
|
||||
t.hour, t.minute, tzinfo=_ET,
|
||||
)
|
||||
dt_utc = dt_et.astimezone(timezone.utc)
|
||||
return dt_utc.strftime("%Y-%m-%d"), dt_utc.strftime("%H:%M")
|
||||
except Exception:
|
||||
return str(ref_date), "00:00"
|
||||
|
||||
|
||||
def _parse_impact(impact_td) -> str:
|
||||
if impact_td is None:
|
||||
return "low"
|
||||
span = impact_td.find("span", class_=lambda c: c and "icon--ff-impact" in c)
|
||||
if span:
|
||||
cls = " ".join(span.get("class", []))
|
||||
for suffix, label in _FF_IMPACT_CSS.items():
|
||||
if suffix in cls:
|
||||
return label
|
||||
# Fallback: check title attribute
|
||||
title = (impact_td.get("title") or "").lower()
|
||||
if "high" in title: return "high"
|
||||
if "medium" in title: return "medium"
|
||||
return "low"
|
||||
|
||||
|
||||
def _scrape_week(client: httpx.Client, monday: date) -> list[tuple]:
|
||||
"""Fetch and parse one week of FF calendar. Returns list of row tuples."""
|
||||
url = _week_url(monday)
|
||||
print(f"[FF scrape] GET {url}", flush=True)
|
||||
try:
|
||||
resp = client.get(url, timeout=20)
|
||||
resp.raise_for_status()
|
||||
except Exception as e:
|
||||
print(f"[FF scrape] fetch error {url}: {e}", flush=True)
|
||||
return []
|
||||
|
||||
soup = BeautifulSoup(resp.text, "lxml")
|
||||
table = soup.find("table", class_="calendar__table")
|
||||
if not table:
|
||||
# Try alternate selector (FF occasionally restructures)
|
||||
table = soup.find("table", attrs={"id": "calendar"})
|
||||
if not table:
|
||||
print(f"[FF scrape] table not found for {url}", flush=True)
|
||||
return []
|
||||
|
||||
rows = table.find_all("tr", class_="calendar__row")
|
||||
batch: list[tuple] = []
|
||||
current_date: date = monday # carry forward
|
||||
|
||||
for tr in rows:
|
||||
# ── Date cell (only filled on day-breaker rows) ─────────────────
|
||||
date_td = tr.find("td", class_="calendar__date")
|
||||
if date_td and date_td.get_text(strip=True):
|
||||
raw = date_td.get_text(" ", strip=True) # e.g. "Fri Jun 27"
|
||||
for fmt in ("%a %b %d", "%A %b %d", "%a %B %d"):
|
||||
try:
|
||||
parsed = datetime.strptime(raw, fmt)
|
||||
current_date = date(monday.year, parsed.month, parsed.day)
|
||||
# Handle year wrap (Dec → Jan)
|
||||
if current_date < monday - timedelta(days=6):
|
||||
current_date = date(monday.year + 1, parsed.month, parsed.day)
|
||||
break
|
||||
except ValueError:
|
||||
continue
|
||||
|
||||
# ── Time ────────────────────────────────────────────────────────
|
||||
time_td = tr.find("td", class_="calendar__time")
|
||||
time_str = time_td.get_text(strip=True) if time_td else ""
|
||||
|
||||
# ── Currency ────────────────────────────────────────────────────
|
||||
cur_td = tr.find("td", class_="calendar__currency")
|
||||
currency = cur_td.get_text(strip=True).upper() if cur_td else ""
|
||||
if currency not in SUPPORTED_CURRENCIES:
|
||||
continue
|
||||
|
||||
# ── Impact ──────────────────────────────────────────────────────
|
||||
impact_td = tr.find("td", class_="calendar__impact")
|
||||
impact = _parse_impact(impact_td)
|
||||
if impact == "non-economic":
|
||||
continue
|
||||
|
||||
# ── Event name ──────────────────────────────────────────────────
|
||||
event_td = tr.find("td", class_="calendar__event")
|
||||
if not event_td:
|
||||
continue
|
||||
title_span = event_td.find("span", class_="calendar__event-title")
|
||||
event_name = (title_span or event_td).get_text(strip=True)
|
||||
if not event_name:
|
||||
continue
|
||||
|
||||
# ── Values ──────────────────────────────────────────────────────
|
||||
actual = _td_val(tr, "calendar__actual")
|
||||
forecast = _td_val(tr, "calendar__forecast")
|
||||
previous = _td_val(tr, "calendar__previous")
|
||||
|
||||
# ── Convert time ET → UTC ────────────────────────────────────────
|
||||
ev_date, ev_time = _parse_et_time(time_str, current_date)
|
||||
|
||||
series_id = FF_TO_FRED.get(event_name)
|
||||
batch.append((
|
||||
ev_date, ev_time, currency, impact, event_name,
|
||||
actual or None, forecast or None, previous or None,
|
||||
series_id, None, "ff_scrape",
|
||||
))
|
||||
|
||||
return batch
|
||||
|
||||
|
||||
def _td_val(tr, cls: str) -> str:
|
||||
td = tr.find("td", class_=cls)
|
||||
if not td:
|
||||
return ""
|
||||
span = td.find("span")
|
||||
text = (span or td).get_text(strip=True)
|
||||
return text if text not in ("", "\xa0") else ""
|
||||
|
||||
|
||||
def scrape_upcoming(weeks_ahead: int = 5) -> Dict[str, Any]:
|
||||
"""
|
||||
Scrape FF calendar HTML for the next N weeks and upsert into ff_calendar.
|
||||
Includes the current week so we catch events added after the JSON sync.
|
||||
"""
|
||||
from services.database import get_conn
|
||||
|
||||
today_date = date.today()
|
||||
monday = today_date - timedelta(days=today_date.weekday())
|
||||
|
||||
total_inserted = 0
|
||||
week_results = {}
|
||||
|
||||
conn = get_conn()
|
||||
with httpx.Client(headers=_FF_HEADERS, follow_redirects=True, timeout=30) as client:
|
||||
# Warm up session — get cookies from homepage
|
||||
try:
|
||||
client.get(_FF_BASE, timeout=10)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
for w in range(weeks_ahead):
|
||||
week_monday = monday + timedelta(weeks=w)
|
||||
batch = _scrape_week(client, week_monday)
|
||||
|
||||
if batch:
|
||||
_upsert_batch(conn, batch)
|
||||
conn.commit()
|
||||
total_inserted += len(batch)
|
||||
|
||||
week_results[str(week_monday)] = len(batch)
|
||||
print(f"[FF scrape] week {week_monday}: {len(batch)} events", flush=True)
|
||||
|
||||
conn.close()
|
||||
print(f"[FF scrape] total upserted: {total_inserted}", flush=True)
|
||||
return {"weeks_scraped": weeks_ahead, "total_upserted": total_inserted, "by_week": week_results}
|
||||
|
||||
@@ -159,11 +159,33 @@ function ImportPanel({ stats, onImported }: { stats: FFStats; onImported: () =>
|
||||
const s = await fetch(`${API}/api/eco/ff-sync/status`).then(x => x.json())
|
||||
const res = s.last_result || {}
|
||||
const tw = res.thisweek?.events ?? '?'
|
||||
setSyncResult(`✓ ${tw} events synced`)
|
||||
setSyncResult(`✓ live: ${tw} events`)
|
||||
onImported()
|
||||
} catch { setSyncResult('sync error') }
|
||||
}
|
||||
|
||||
const [scrapeMsg, setScrapeMsg] = useState<string | null>(null)
|
||||
const scrapeRef = useRef<ReturnType<typeof setInterval> | null>(null)
|
||||
|
||||
const startScrape = async () => {
|
||||
setScrapeMsg('scraping…')
|
||||
try {
|
||||
await fetch(`${API}/api/eco/ff-scrape`, { method: 'POST' })
|
||||
scrapeRef.current = setInterval(async () => {
|
||||
const s = await fetch(`${API}/api/eco/ff-scrape/status`).then(x => x.json())
|
||||
if (!s.running) {
|
||||
clearInterval(scrapeRef.current!)
|
||||
const r = s.last_result || {}
|
||||
if (r.error) { setScrapeMsg(`scrape error: ${r.error}`); return }
|
||||
setScrapeMsg(`✓ ${r.total_upserted ?? '?'} events (${r.weeks_scraped} weeks)`)
|
||||
onImported()
|
||||
}
|
||||
}, 3000)
|
||||
} catch { setScrapeMsg('scrape error') }
|
||||
}
|
||||
|
||||
useEffect(() => () => { if (scrapeRef.current) clearInterval(scrapeRef.current) }, [])
|
||||
|
||||
const hasData = stats.total > 0
|
||||
const res = importStatus.last_result
|
||||
|
||||
@@ -176,17 +198,30 @@ function ImportPanel({ stats, onImported }: { stats: FFStats; onImported: () =>
|
||||
? <span className="text-yellow-400 animate-pulse">Importing CSV… (auto-import on startup)</span>
|
||||
: hasData
|
||||
? <><span className="text-emerald-400">{stats.total.toLocaleString()} events</span> <span className="text-slate-500 text-xs">({stats.earliest?.slice(0,7)} → {stats.latest?.slice(0,7)})</span></>
|
||||
: <span className="text-slate-400 text-xs">Auto-import will run on next server start (CSV bundled in image)</span>
|
||||
: <span className="text-slate-400 text-xs">Auto-import will run on next server start</span>
|
||||
}
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{/* Live sync — this week from faireconomy.media JSON */}
|
||||
<button
|
||||
onClick={startSync}
|
||||
className="px-3 py-1.5 rounded bg-slate-600 hover:bg-slate-500 flex items-center gap-1.5 text-white"
|
||||
title="Update this week actuals from faireconomy.media (Forex Factory JSON)"
|
||||
>
|
||||
<RefreshCw size={13} />
|
||||
Sync Live (this week)
|
||||
Sync Live
|
||||
</button>
|
||||
|
||||
{/* HTML scrape — upcoming 5 weeks with forecasts */}
|
||||
<button
|
||||
onClick={startScrape}
|
||||
disabled={scrapeMsg === 'scraping…'}
|
||||
className="px-3 py-1.5 rounded bg-indigo-700 hover:bg-indigo-600 disabled:opacity-50 flex items-center gap-1.5 text-white"
|
||||
title="Scrape forexfactory.com for upcoming 5 weeks with consensus forecasts"
|
||||
>
|
||||
<RefreshCw size={13} className={clsx(scrapeMsg === 'scraping…' && 'animate-spin')} />
|
||||
Scrape Upcoming (5w)
|
||||
</button>
|
||||
|
||||
{res && !importStatus.running && (
|
||||
@@ -195,6 +230,11 @@ function ImportPanel({ stats, onImported }: { stats: FFStats; onImported: () =>
|
||||
: <span className="text-emerald-400 text-xs flex items-center gap-1"><Check size={12}/> {res.inserted?.toLocaleString()} inserted</span>
|
||||
)}
|
||||
{syncResult && <span className="text-sky-400 text-xs">{syncResult}</span>}
|
||||
{scrapeMsg && (
|
||||
<span className={clsx('text-xs', scrapeMsg.startsWith('✓') ? 'text-emerald-400' : scrapeMsg.includes('error') ? 'text-red-400' : 'text-yellow-400')}>
|
||||
{scrapeMsg}
|
||||
</span>
|
||||
)}
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user