Files
OpenFin/backend/services/ff_page_parser.py
2026-07-02 11:27:45 +02:00

167 lines
5.0 KiB
Python

"""
Parser for ForexFactory calendar page source (HTML).
FF embeds a rich JSON object directly in the page:
window.calendarComponentStates[1] = { days: [...] }
Each event has: dateline (unix ts UTC), currency (ISO), name, impactName,
actual, forecast, previous, revision.
Usage:
rows = parse_ff_page(html_string)
# → list of tuples ready for ff_calendar._upsert_batch()
"""
import json
import logging
import re
from datetime import datetime, timezone
logger = logging.getLogger(__name__)
# Currencies we care about (same set as ff_calendar.py)
_SUPPORTED = {"USD", "EUR", "GBP", "JPY", "AUD", "CAD", "NZD", "CHF", "CNY"}
# Impact labels FF uses (impactName field in the embedded JSON)
_IMPACT_OK = {"high", "medium", "low"}
def _extract_days_json(html: str) -> list:
"""
Pull the value of `days` array from window.calendarComponentStates[1].
Returns parsed list or raises ValueError.
"""
# Find the start of the days array
m = re.search(r'window\.calendarComponentStates\s*\[\s*1\s*\]\s*=\s*\{\s*days\s*:\s*(\[)', html)
if not m:
raise ValueError("window.calendarComponentStates[1] not found in HTML")
start = m.start(1)
depth = 0
end = start
in_string = False
escape = False
for i in range(start, len(html)):
ch = html[i]
if escape:
escape = False
continue
if ch == '\\' and in_string:
escape = True
continue
if ch == '"':
in_string = not in_string
continue
if in_string:
continue
if ch == '[':
depth += 1
elif ch == ']':
depth -= 1
if depth == 0:
end = i + 1
break
raw = html[start:end]
return json.loads(raw)
def parse_ff_page(html: str) -> tuple[list, dict]:
"""
Parse FF calendar HTML page source.
Returns (rows, stats) where rows are tuples for _upsert_batch().
"""
days = _extract_days_json(html)
rows = []
skipped = 0
date_min = date_max = None
for day in days:
for ev in day.get("events", []):
dateline = ev.get("dateline")
if not dateline:
skipped += 1
continue
currency = (ev.get("currency") or "").strip()
if currency not in _SUPPORTED:
skipped += 1
continue
impact = (ev.get("impactName") or "low").lower()
if impact not in _IMPACT_OK:
skipped += 1
continue
name = (ev.get("name") or "").strip()
if not name:
skipped += 1
continue
dt = datetime.fromtimestamp(int(dateline), tz=timezone.utc)
event_date = dt.strftime("%Y-%m-%d")
# "All Day" events: FF sets timeMasked=true — normalize to "00:00"
# so they match rows inserted by the HTML scraper (same convention)
event_time = "00:00" if ev.get("timeMasked") else dt.strftime("%H:%M")
if date_min is None or event_date < date_min:
date_min = event_date
if date_max is None or event_date > date_max:
date_max = event_date
actual = (ev.get("actual") or "").strip() or None
forecast = (ev.get("forecast") or "").strip() or None
revision = (ev.get("revision") or "").strip()
previous = revision or (ev.get("previous") or "").strip() or None
rows.append((
event_date, event_time, currency, impact, name,
actual, forecast, previous,
None, None, "ff_page",
))
stats = {
"total": len(rows) + skipped,
"inserted": len(rows),
"skipped": skipped,
"date_from": date_min,
"date_to": date_max,
}
return rows, stats
def import_ff_page_file(path: str) -> dict:
"""Read an FF calendar HTML file and upsert into ff_calendar."""
from services.database import get_conn
from services.ff_calendar import _upsert_batch
try:
with open(path, encoding="utf-8", errors="replace") as f:
html = f.read()
except Exception as e:
return {"error": f"Cannot read file: {e}"}
try:
rows, stats = parse_ff_page(html)
except ValueError as e:
return {"error": str(e)}
except Exception as e:
logger.error(f"[ff_page_parser] parse error: {e}")
return {"error": f"Parse error: {e}"}
if not rows:
return {**stats, "error": "No usable events found — check the HTML is a full FF calendar page"}
conn = get_conn()
try:
_upsert_batch(conn, rows)
conn.commit()
logger.info(f"[ff_page_parser] {stats['inserted']} rows upserted ({stats['date_from']}{stats['date_to']})")
except Exception as e:
logger.error(f"[ff_page_parser] DB error: {e}")
return {**stats, "error": f"DB error: {e}"}
finally:
conn.close()
return stats