diff --git a/.github/workflows/fastapi-react.yml b/.github/workflows/fastapi-react.yml
index 46ee8fe7..86dcd6e4 100644
--- a/.github/workflows/fastapi-react.yml
+++ b/.github/workflows/fastapi-react.yml
@@ -3,6 +3,10 @@ name: FastAPI + React parity checks
permissions:
contents: read
+concurrency:
+ group: fastapi-react-parity-${{ github.ref }}
+ cancel-in-progress: true
+
on:
pull_request:
paths:
@@ -70,3 +74,115 @@ jobs:
- name: Production-deps audit
run: npm run audit
continue-on-error: true # dev-dep advisories only; see PARITY_REPORT.md
+
+
+ visual-parity:
+ name: Visual parity evidence
+ runs-on: ubuntu-latest
+ needs: [backend, frontend]
+ timeout-minutes: 30
+ steps:
+ - uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4
+ - uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
+ with:
+ python-version: "3.12"
+ cache: pip
+ cache-dependency-path: fastapi_react/backend/requirements.txt
+ - uses: actions/setup-node@cdca7365b2dadb8aad0a33bc7601856ffabcc48e # v4.3.0
+ with:
+ node-version: "20"
+ cache: npm
+ cache-dependency-path: fastapi_react/frontend/package-lock.json
+ - name: Install backend
+ working-directory: fastapi_react/backend
+ run: |
+ python -m pip install --upgrade pip
+ pip install -r requirements.txt
+ - name: Install Streamlit reference runtime
+ run: pip install -r requirements-web.txt
+ - name: Install frontend and browser
+ working-directory: fastapi_react/frontend
+ run: |
+ npm ci
+ npx playwright install --with-deps chromium
+ - name: Start parity stack
+ env:
+ F1_REPO_ROOT: ${{ github.workspace }}
+ ENABLE_EXPENSIVE_TOOLS: "0"
+ F1_RESEARCH_MODE: "0"
+ STREAMLIT_SERVER_HEADLESS: "true"
+ run: |
+ (cd fastapi_react/backend && uvicorn app.main:app --host 127.0.0.1 --port 8000 > /tmp/f1-backend.log 2>&1 & echo $! > /tmp/f1-backend.pid)
+ (cd fastapi_react/frontend && npm run dev -- --host 127.0.0.1 --port 5173 > /tmp/f1-frontend.log 2>&1 & echo $! > /tmp/f1-frontend.pid)
+ (streamlit run raceAnalysis.py --server.address 127.0.0.1 --server.port 8501 --server.headless true > /tmp/f1-streamlit.log 2>&1 & echo $! > /tmp/f1-streamlit.pid)
+ for i in {1..90}; do
+ if curl -fsS http://127.0.0.1:8000/api/health >/dev/null && curl -fsS http://127.0.0.1:5173 >/dev/null && curl -fsS http://127.0.0.1:8501/_stcore/health >/dev/null; then
+ exit 0
+ fi
+ sleep 2
+ done
+ cat /tmp/f1-backend.log || true
+ cat /tmp/f1-frontend.log || true
+ cat /tmp/f1-streamlit.log || true
+ exit 1
+ - name: Capture React reference
+ working-directory: fastapi_react/frontend
+ env:
+ REACT_BASE_URL: http://127.0.0.1:5173
+ REACT_WAIT_MS: "4000"
+ run: npm run capture:react
+ - name: Accessibility audit
+ id: accessibility
+ continue-on-error: true
+ working-directory: fastapi_react/frontend
+ env:
+ REACT_BASE_URL: http://127.0.0.1:5173
+ run: npm run audit:a11y
+ - name: Operational benchmark
+ id: benchmark
+ continue-on-error: true
+ working-directory: fastapi_react/frontend
+ env:
+ BENCH_BASE_URL: http://127.0.0.1:5173
+ BENCH_API_URL: http://127.0.0.1:8000
+ run: npm run benchmark
+ - name: Capture local Streamlit reference
+ id: streamlit_capture
+ continue-on-error: true
+ working-directory: fastapi_react/frontend
+ env:
+ STREAMLIT_BASE_URL: http://127.0.0.1:8501
+ STREAMLIT_WAIT_MS: "5000"
+ run: npm run capture:streamlit
+ - name: Diff screenshots
+ id: visual_diff
+ continue-on-error: true
+ working-directory: fastapi_react/frontend
+ run: npm run capture:diff
+ - name: Upload visual parity evidence
+ if: always()
+ uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
+ with:
+ name: fastapi-react-parity-evidence
+ path: |
+ fastapi_react/parity_evidence/visual
+ fastapi_react/parity_evidence/accessibility.json
+ fastapi_react/parity_evidence/benchmarks.json
+ if-no-files-found: warn
+ - name: Enforce parity evidence gates
+ if: always()
+ env:
+ ACCESSIBILITY_OUTCOME: ${{ steps.accessibility.outcome }}
+ BENCHMARK_OUTCOME: ${{ steps.benchmark.outcome }}
+ STREAMLIT_CAPTURE_OUTCOME: ${{ steps.streamlit_capture.outcome }}
+ VISUAL_DIFF_OUTCOME: ${{ steps.visual_diff.outcome }}
+ run: |
+ failed=0
+ for gate in ACCESSIBILITY_OUTCOME BENCHMARK_OUTCOME STREAMLIT_CAPTURE_OUTCOME VISUAL_DIFF_OUTCOME; do
+ outcome="${!gate}"
+ echo "$gate=$outcome"
+ if [ "$outcome" != "success" ]; then
+ failed=1
+ fi
+ done
+ exit "$failed"
diff --git a/fastapi_react/backend/app/main.py b/fastapi_react/backend/app/main.py
index a9095d85..23da0f45 100644
--- a/fastapi_react/backend/app/main.py
+++ b/fastapi_react/backend/app/main.py
@@ -1,5 +1,6 @@
from __future__ import annotations
+import datetime
import os
from typing import Any
@@ -22,6 +23,7 @@
from app.services.data import (
filter_schema,
list_data_files,
+ load_main_data,
model_manifest,
precomputed,
query_main,
@@ -81,6 +83,16 @@ def brand_logo() -> FileResponse:
@app.get("/api/meta")
def meta() -> dict[str, Any]:
+ data = load_main_data()
+ years = data.get("grandPrixYear")
+ race_start_year = int(years.min()) if years is not None and years.notna().any() else 2016
+ current_year = int(years.max()) if years is not None and years.notna().any() else datetime.datetime.now().year
+ candidates = [path for path in DATA_DIR.iterdir() if path.is_file()] if DATA_DIR.exists() else []
+ latest = max(candidates, key=lambda path: path.stat().st_mtime, default=None)
+ last_updated = (
+ datetime.datetime.fromtimestamp(latest.stat().st_mtime).strftime("%Y-%m-%d %I:%M %p")
+ if latest is not None else "No data files found"
+ )
return {
"tabs": [
"Data Explorer", "Analytics", "Current Season", "Next Race",
@@ -89,6 +101,10 @@ def meta() -> dict[str, Any]:
"models": MODEL_TYPES,
"expensive_tools_enabled": ENABLE_EXPENSIVE_TOOLS,
"manual_tools": list(TOOLS),
+ "race_start_year": race_start_year,
+ "current_year": current_year,
+ "last_updated": last_updated,
+ "code_deployed_at": datetime.datetime.now(datetime.UTC).strftime("%Y-%m-%d %H:%M:%S UTC"),
}
diff --git a/fastapi_react/backend/app/schemas.py b/fastapi_react/backend/app/schemas.py
index 432bd1cd..573ff98b 100644
--- a/fastapi_react/backend/app/schemas.py
+++ b/fastapi_react/backend/app/schemas.py
@@ -15,6 +15,7 @@ class QueryRequest(BaseModel):
filters: list[FilterSpec] = Field(default_factory=list)
columns: list[str] | None = None
sort: list[str] = Field(default_factory=list)
+ ascending: list[bool] | None = None
descending: bool = False
offset: int = 0
limit: int = Field(default=200, ge=1, le=5000)
diff --git a/fastapi_react/backend/app/services/analysis.py b/fastapi_react/backend/app/services/analysis.py
index 6ef139da..f1f130fa 100644
--- a/fastapi_react/backend/app/services/analysis.py
+++ b/fastapi_react/backend/app/services/analysis.py
@@ -1,5 +1,6 @@
from __future__ import annotations
+import pickle
from functools import lru_cache
from pathlib import Path
from typing import Any
@@ -9,7 +10,14 @@
from scipy.stats import linregress
from app.config import DATA_DIR
-from app.services.data import apply_filters, load_main_data, load_race_schedule, records
+from app.services.data import (
+ apply_filters,
+ load_main_data,
+ load_race_schedule,
+ model_manifest,
+ precomputed,
+ records,
+)
def _regression(df: pd.DataFrame, x_col: str, y_col: str) -> dict[str, Any] | None:
@@ -52,8 +60,12 @@ def _regression_series(df: pd.DataFrame, x_col: str, y_col: str) -> dict[str, An
def analytics(filters: Any, max_rows: int) -> dict[str, Any]:
+ """Return every lightweight analysis block rendered by the Streamlit Analytics tab."""
df = apply_filters(load_main_data(), filters).head(max_rows).copy()
payload: dict[str, Any] = {"rows_considered": len(df), "charts": {}, "regressions": []}
+ if df.empty:
+ return payload
+
pairs = {
"active_years_vs_final": ("resultsFinalPositionNumber", "yearsActive"),
"positions_gained_over_time": ("short_date", "positionsGained"),
@@ -61,6 +73,7 @@ def analytics(filters: Any, max_rows: int) -> dict[str, Any]:
"grid_vs_final": ("resultsStartingGridPositionNumber", "resultsFinalPositionNumber"),
"avg_practice_vs_final": ("averagePracticePosition", "resultsFinalPositionNumber"),
"pit_stop_vs_final": ("averageStopTime", "resultsFinalPositionNumber"),
+ "track_turns_vs_final": ("turns", "resultsFinalPositionNumber"),
}
for name, (x, y) in pairs.items():
if x in df and y in df:
@@ -71,8 +84,14 @@ def analytics(filters: Any, max_rows: int) -> dict[str, Any]:
if result:
payload["regressions"].append(result)
regression_titles = {
- "averagePracticePosition": ("Linear Regression: Average Practice Position vs Final Position", "Average Practice Position"),
- "resultsStartingGridPositionNumber": ("Linear Regression: Starting Position vs Final Position", "Starting Position"),
+ "averagePracticePosition": (
+ "Linear Regression: Average Practice Position vs Final Position",
+ "Average Practice Position",
+ ),
+ "resultsStartingGridPositionNumber": (
+ "Linear Regression: Starting Position vs. Final Position",
+ "Starting Position",
+ ),
}
payload["regression_series"] = []
for x, (title, x_label) in regression_titles.items():
@@ -90,19 +109,22 @@ def analytics(filters: Any, max_rows: int) -> dict[str, Any]:
"driverBestRaceResult", "driverTotalChampionshipWins", "driverTotalPolePositions",
"driverTotalRaceEntries", "driverTotalRaceStarts", "driverTotalRaceWins",
"driverTotalRaceLaps", "driverTotalPodiums", "avgLapPace", "finishingTime",
+ "resultsQualificationPositionNumber", "numberOfStops",
) if c in df]
if corr_cols:
corr = df[corr_cols].apply(pd.to_numeric, errors="coerce").corr()
payload["correlation"] = {
"columns": list(corr.columns),
"rows": [
- {"feature": idx, **{col: (None if pd.isna(v) else float(v)) for col, v in row.items()}}
+ {"Feature": idx, **{col: (None if pd.isna(v) else float(v)) for col, v in row.items()}}
for idx, row in corr.iterrows()
],
}
if {"grandPrixYear", "resultsDriverName", "resultsFinalPositionNumber"}.issubset(df.columns):
- agg = {"average_final_position": ("resultsFinalPositionNumber", "mean")}
+ agg: dict[str, tuple[str, Any]] = {
+ "average_final_position": ("resultsFinalPositionNumber", "mean"),
+ }
if "resultsPodium" in df:
agg["total_podiums"] = ("resultsPodium", "sum")
driver = df.groupby(["grandPrixYear", "resultsDriverName"]).agg(**agg).reset_index()
@@ -112,34 +134,127 @@ def analytics(filters: Any, max_rows: int) -> dict[str, Any]:
constructor = (
df.groupby(["grandPrixYear", "constructorName"])
.agg(
- total_wins=("resultsFinalPositionNumber", lambda s: int((s == 1).sum())),
+ total_wins=("resultsFinalPositionNumber", lambda values: int((values == 1).sum())),
average_final_position=("resultsFinalPositionNumber", "mean"),
).reset_index()
)
if "resultsPodium" in df:
- podium = df.groupby(["grandPrixYear", "constructorName"])["resultsPodium"].sum().reset_index(name="total_podiums")
+ podium = (
+ df.groupby(["grandPrixYear", "constructorName"])["resultsPodium"]
+ .sum().reset_index(name="total_podiums")
+ )
constructor = constructor.merge(podium, on=["grandPrixYear", "constructorName"], how="left")
payload["constructor_performance"] = records(constructor)
- if {"DNF", "resultsReasonRetired"}.issubset(df.columns):
+ if {"constructorName", "resultsDriverName", "positionsGained", "resultsFinalPositionNumber"}.issubset(df.columns):
+ driver_vs_constructor = (
+ df.groupby(["constructorName", "resultsDriverName"])
+ .agg(
+ positionsGained=("positionsGained", "sum"),
+ average_final_position=("resultsFinalPositionNumber", "mean"),
+ )
+ .reset_index()
+ .sort_values("average_final_position")
+ )
+ driver_vs_constructor["average_final_position"] = driver_vs_constructor["average_final_position"].round(2)
+ payload["driver_vs_constructor"] = records(driver_vs_constructor)
+
+ dnf_rows = pd.DataFrame()
+ if "DNF" in df:
+ dnf_rows = df[pd.to_numeric(df["DNF"], errors="coerce").fillna(0).eq(1)].copy()
+ if not dnf_rows.empty and "resultsReasonRetired" in dnf_rows:
dnf = (
- df[pd.to_numeric(df["DNF"], errors="coerce").fillna(0).eq(1)]
- .groupby("resultsReasonRetired").size().reset_index(name="count")
+ dnf_rows.groupby("resultsReasonRetired").size().reset_index(name="count")
.sort_values("count", ascending=False)
)
payload["dnf_reasons"] = records(dnf)
- dnf_group_cols = {"DNF", "resultsDriverName", "driverTotalRaceEntries"}
- if dnf_group_cols.issubset(df.columns):
- dnf_by_driver = (
- df[pd.to_numeric(df["DNF"], errors="coerce").eq(1)]
- .groupby(["resultsDriverName", "driverTotalRaceEntries"])
- .size().reset_index(name="dnf_count")
+ if not dnf_rows.empty and {"resultsDriverName", "driverTotalRaceEntries"}.issubset(dnf_rows.columns):
+ grouped = (
+ dnf_rows.groupby(["resultsDriverName", "driverTotalRaceEntries"]).size()
+ .reset_index(name="dnf_count")
)
- entries = pd.to_numeric(dnf_by_driver["driverTotalRaceEntries"], errors="coerce")
- dnf_by_driver["dnf_pct"] = (dnf_by_driver["dnf_count"] / entries * 100).round(1)
- payload["dnf_by_driver"] = records(dnf_by_driver.sort_values("dnf_pct", ascending=False))
- return payload
+ entries = pd.to_numeric(grouped["driverTotalRaceEntries"], errors="coerce")
+ grouped["dnf_pct"] = (grouped["dnf_count"] / entries * 100).round(1)
+ payload["dnf_by_driver"] = records(grouped.sort_values("dnf_pct", ascending=False))
+ if "grandPrixName" in df:
+ entries = df.groupby("grandPrixName").size().reset_index(name="race_entry_count")
+ if not dnf_rows.empty:
+ dnfs = dnf_rows.groupby("grandPrixName").size().reset_index(name="dnf_count")
+ entries = entries.merge(dnfs, on="grandPrixName", how="left")
+ else:
+ entries["dnf_count"] = 0
+ entries["dnf_count"] = entries["dnf_count"].fillna(0).astype(int)
+ entries["dnf_pct"] = (entries["dnf_count"] / entries["race_entry_count"] * 100).round(1)
+ payload["dnf_by_race"] = records(entries.sort_values("dnf_pct", ascending=False))
+ if "constructorName" in df:
+ entries = df.groupby("constructorName").size().reset_index(name="constructor_entry_count")
+ if not dnf_rows.empty:
+ dnfs = dnf_rows.groupby("constructorName").size().reset_index(name="dnf_count")
+ entries = entries.merge(dnfs, on="constructorName", how="left")
+ else:
+ entries["dnf_count"] = 0
+ entries["dnf_count"] = entries["dnf_count"].fillna(0).astype(int)
+ entries["dnf_pct"] = (
+ entries["dnf_count"] / entries["constructor_entry_count"] * 100
+ ).round(1)
+ payload["dnf_by_constructor"] = records(entries.sort_values("dnf_pct", ascending=False))
+
+ if {"grandPrixYear", "resultsDriverName", "positionsGained", "resultsPodium"}.issubset(df.columns):
+ year = int(pd.to_numeric(df["grandPrixYear"], errors="coerce").max())
+ season = (
+ df[pd.to_numeric(df["grandPrixYear"], errors="coerce") == year]
+ .groupby("resultsDriverName")
+ .agg(positions_gained=("positionsGained", "sum"), total_podiums=("resultsPodium", "sum"))
+ .reset_index()
+ )
+ payload["season_year"] = year
+ payload["season_summary"] = records(season)
+
+ if {"resultsDriverName", "resultsFinalPositionNumber"}.issubset(df.columns):
+ consistency = (
+ df.groupby("resultsDriverName")
+ .agg(finishing_position_std=("resultsFinalPositionNumber", "std"))
+ .reset_index()
+ .sort_values("finishing_position_std")
+ )
+ payload["driver_consistency"] = records(consistency)
+
+ try:
+ manifest = model_manifest("XGBoost")
+ except (KeyError, OSError, ValueError):
+ manifest = None
+ payload["model_summary"] = manifest
+
+ try:
+ importance = precomputed("permutation")
+ except (KeyError, OSError, ValueError):
+ importance = None
+ payload["feature_importance"] = importance
+ try:
+ historical = precomputed("historical_validation")
+ except (KeyError, OSError, ValueError):
+ historical = None
+ holdout = (historical or {}).get("holdout", {}) if isinstance(historical, dict) else {}
+ holdout_rows = holdout.get("rows", []) if isinstance(holdout, dict) else []
+ if isinstance(holdout_rows, list) and holdout_rows:
+ holdout_frame = pd.DataFrame(holdout_rows)
+ expected = {"ActualFinalPosition", "PredictedFinalPosition", "Error"}
+ if expected.issubset(holdout_frame.columns):
+ holdout_frame = holdout_frame.sort_values("ActualFinalPosition")
+ top3 = holdout_frame[pd.to_numeric(holdout_frame["ActualFinalPosition"], errors="coerce") <= 3].copy()
+ if not top3.empty:
+ payload["top3_mae"] = float(
+ np.mean(
+ np.abs(
+ pd.to_numeric(top3["ActualFinalPosition"], errors="coerce")
+ - pd.to_numeric(top3["PredictedFinalPosition"], errors="coerce")
+ )
+ )
+ )
+ payload["top3_predictions"] = records(top3.head(100))
+ payload["first_30_predictions"] = records(holdout_frame.head(30))
+ return payload
def current_season() -> dict[str, Any]:
schedule = load_race_schedule().copy()
@@ -388,6 +503,350 @@ def find_prediction_artifact(
}
+def _legacy_prediction_rows(race_id: str, year: int | str, race_name: str) -> list[dict[str, Any]]:
+ """Return the committed Streamlit-style CSV prediction rows when an exact race artifact exists."""
+ slugs = {
+ str(race_id).strip().lower().replace("_", "-").replace(" ", "-"),
+ str(race_name).lower().replace(" grand prix", "").replace(" ", "-"),
+ }
+ candidates: list[Path] = []
+ for slug in sorted(slugs):
+ if slug:
+ candidates.extend([
+ DATA_DIR / f"predictions_{slug}_{year}.csv",
+ DATA_DIR / f"predictions_{slug.replace('-', '_')}_{year}.csv",
+ ])
+ for candidate in candidates:
+ if candidate.is_file():
+ frame = _read_optional(candidate)
+ if frame.empty:
+ continue
+
+ data = load_main_data()
+ if "resultsDriverName" in frame and {"resultsDriverName", "driverDNFCount", "driverDNFAvg"}.issubset(data.columns):
+ latest = (
+ data.sort_values("grandPrixYear")
+ .groupby("resultsDriverName", as_index=False)
+ .tail(1)[["resultsDriverName", "driverDNFCount", "driverDNFAvg"]]
+ .drop_duplicates("resultsDriverName")
+ )
+ frame = frame.merge(latest, on="resultsDriverName", how="left", suffixes=("", "_latest"))
+ if "driverDNFCount_latest" in frame:
+ frame["driverDNFCount"] = frame.get("driverDNFCount").fillna(frame["driverDNFCount_latest"]) if "driverDNFCount" in frame else frame["driverDNFCount_latest"]
+ if "driverDNFAvg_latest" in frame:
+ frame["driverDNFAvg"] = frame.get("driverDNFAvg").fillna(frame["driverDNFAvg_latest"]) if "driverDNFAvg" in frame else frame["driverDNFAvg_latest"]
+ frame["driverDNFPercentage"] = (
+ pd.to_numeric(frame.get("driverDNFAvg"), errors="coerce").fillna(0) * 100
+ ).round(3)
+ frame = frame.drop(columns=["driverDNFCount_latest", "driverDNFAvg_latest"], errors="ignore")
+ if "PredictedDNFProbabilityStd" not in frame:
+ frame["PredictedDNFProbabilityStd"] = np.nan
+
+ sort_col = "Rank" if "Rank" in frame else (
+ "PredictedFinalPosition" if "PredictedFinalPosition" in frame else None
+ )
+ if sort_col:
+ frame = frame.sort_values(sort_col)
+ return records(frame)
+ return []
+
+
+@lru_cache(maxsize=1)
+def _position_mae_by_position() -> dict[int, float]:
+ """Return the same per-position holdout MAE mapping used by the Streamlit next-race table."""
+ try:
+ historical = precomputed("historical_validation")
+ except (KeyError, OSError, ValueError):
+ historical = None
+ rows = ((historical or {}).get("holdout") or {}).get("rows", []) if isinstance(historical, dict) else []
+ if not rows:
+ return {}
+ frame = pd.DataFrame(rows)
+ if not {"ActualFinalPosition", "PredictedFinalPosition"}.issubset(frame.columns):
+ return {}
+ frame["ActualFinalPosition"] = pd.to_numeric(frame["ActualFinalPosition"], errors="coerce")
+ frame["PredictedFinalPosition"] = pd.to_numeric(frame["PredictedFinalPosition"], errors="coerce")
+ frame = frame.dropna(subset=["ActualFinalPosition", "PredictedFinalPosition"])
+ frame["absolute_error"] = (frame["ActualFinalPosition"] - frame["PredictedFinalPosition"]).abs()
+ grouped = frame.groupby("ActualFinalPosition")["absolute_error"].mean()
+ return {int(position): float(mae) for position, mae in grouped.items()}
+
+
+@lru_cache(maxsize=1)
+def _load_dnf_model() -> Any:
+ """Load the trusted, workflow-generated DNF inference artifact."""
+ path = DATA_DIR / "models" / "dnf_model.pkl"
+ if not path.is_file():
+ return None
+ with path.open("rb") as handle:
+ artifact = pickle.load(handle) # noqa: S301 - trusted model artifact committed by this repository
+ return artifact.get("model") if isinstance(artifact, dict) else artifact
+
+
+@lru_cache(maxsize=1)
+def _dnf_feature_names() -> tuple[str, ...]:
+ """Read the authoritative DNF feature order from the committed manifest."""
+ import json
+
+ path = DATA_DIR / "models" / "dnf_manifest.json"
+ if not path.is_file():
+ return ()
+ try:
+ payload = json.loads(path.read_text(encoding="utf-8"))
+ except (OSError, json.JSONDecodeError):
+ return ()
+ return tuple(str(value) for value in payload.get("feature_names", ()))
+
+
+@lru_cache(maxsize=1)
+def dnf_diagnostics() -> dict[str, float | None]:
+ """Return min/max/mean saved-model DNF probabilities over the historical analysis rows."""
+ model = _load_dnf_model()
+ feature_names = _dnf_feature_names()
+ if model is None or not feature_names:
+ return {"min": None, "max": None, "mean": None}
+ frame = load_main_data().copy()
+ for column in feature_names:
+ if column not in frame:
+ frame[column] = np.nan
+ try:
+ probabilities = model.predict_proba(frame[list(feature_names)])[:, 1]
+ except Exception:
+ return {"min": None, "max": None, "mean": None}
+ finite = np.asarray(probabilities, dtype=float)
+ finite = finite[np.isfinite(finite)]
+ if not finite.size:
+ return {"min": None, "max": None, "mean": None}
+ return {
+ "min": float(finite.min()),
+ "max": float(finite.max()),
+ "mean": float(finite.mean()),
+ }
+
+
+def build_dnf_predictions(
+ position_predictions: dict[str, Any] | None,
+ next_race: pd.Series,
+ race_name: str,
+ weather: pd.DataFrame,
+) -> list[dict[str, Any]]:
+ """Generate Streamlit-equivalent DNF rows from the committed inference artifact."""
+ model = _load_dnf_model()
+ feature_names = _dnf_feature_names()
+ if model is None or not feature_names or not isinstance(position_predictions, dict):
+ return []
+
+ by_model = position_predictions.get("predictions_by_model") or {}
+ block = by_model.get("xgboost") or (next(iter(by_model.values()), {}) if by_model else {})
+ prediction_rows = block.get("predictions") or []
+ if not prediction_rows:
+ return []
+
+ data = load_main_data().copy()
+ if "resultsDriverName" not in data:
+ return []
+ sort_column = "grandPrixYear" if "grandPrixYear" in data else None
+ if sort_column:
+ data = data.sort_values(sort_column)
+ latest = data.groupby("resultsDriverName", as_index=False).tail(1).copy()
+ latest = latest.set_index("resultsDriverName", drop=False)
+
+ schedule_map = {
+ "turns": "turns",
+ "trackRace": "trackRace",
+ "streetRace": "streetRace",
+ }
+ weather_row = weather.iloc[0] if not weather.empty else pd.Series(dtype=object)
+ rows: list[dict[str, Any]] = []
+ feature_rows: list[dict[str, Any]] = []
+ for prediction in prediction_rows:
+ driver = str(prediction.get("driverName", ""))
+ if not driver or driver not in latest.index:
+ continue
+ source = latest.loc[driver]
+ if isinstance(source, pd.DataFrame):
+ source = source.iloc[-1]
+ feature_row = {name: source.get(name, np.nan) for name in feature_names}
+ feature_row["grandPrixName"] = race_name
+ if prediction.get("constructor"):
+ feature_row["constructorName"] = prediction["constructor"]
+ feature_row["resultsDriverName"] = driver
+ for target, source_name in schedule_map.items():
+ if target in feature_row and source_name in next_race.index and pd.notna(next_race[source_name]):
+ feature_row[target] = next_race[source_name]
+ for column in ("average_temp", "average_humidity", "average_wind_speed", "total_precipitation"):
+ if column in feature_row and column in weather_row.index and pd.notna(weather_row[column]):
+ feature_row[column] = weather_row[column]
+ feature_rows.append(feature_row)
+ rows.append({
+ "constructorName": prediction.get("constructor", source.get("constructorName")),
+ "resultsDriverName": driver,
+ "driverDNFCount": source.get("driverDNFCount"),
+ "driverDNFPercentage": (
+ round(float(source.get("driverDNFAvg", 0) or 0) * 100, 3)
+ if pd.notna(source.get("driverDNFAvg"))
+ else 0.0
+ ),
+ "PredictedDNFProbabilityStd": None,
+ })
+
+ if not feature_rows:
+ return []
+ frame = pd.DataFrame(feature_rows, columns=list(feature_names))
+ try:
+ probabilities = model.predict_proba(frame)[:, 1]
+ except Exception:
+ return []
+ for row, probability in zip(rows, probabilities, strict=True):
+ row["PredictedDNFProbabilityPercentage"] = round(float(probability) * 100, 3)
+ rows.sort(key=lambda row: float(row["PredictedDNFProbabilityPercentage"]), reverse=True)
+ return rows
+
+
+@lru_cache(maxsize=1)
+def _load_safety_car_inputs() -> pd.DataFrame:
+ """Load the same historical safety-car feature frame used by Streamlit."""
+ path = DATA_DIR / "f1SafetyCarFeatures.csv"
+ if not path.is_file():
+ return pd.DataFrame()
+ return pd.read_csv(path, sep="\t", low_memory=False)
+
+
+@lru_cache(maxsize=1)
+def _load_safety_car_model() -> Any:
+ """Load the same trusted safety-car artifact search order used by Streamlit."""
+ candidates = [
+ DATA_DIR / "models" / "xgboost" / "safetycar_model.pkl",
+ DATA_DIR / "models" / "lightgbm" / "safetycar_model.pkl",
+ DATA_DIR / "models" / "catboost" / "safetycar_model.pkl",
+ DATA_DIR / "models" / "ensemble" / "safetycar_model.pkl",
+ DATA_DIR / "models" / "safetycar_model.pkl",
+ ]
+ for path in candidates:
+ if not path.is_file():
+ continue
+ try:
+ with path.open("rb") as handle:
+ artifact = pickle.load(handle) # noqa: S301 - trusted repository artifact
+ except (OSError, pickle.UnpicklingError, AttributeError, EOFError, ImportError, ValueError):
+ continue
+ if isinstance(artifact, dict):
+ model = artifact.get("model")
+ if model is not None:
+ return model
+ continue
+ if hasattr(artifact, "predict_proba"):
+ return artifact
+ return None
+
+
+def build_safety_car_predictions(
+ next_race: pd.Series,
+ race_name: str,
+ year: int,
+ weather: pd.DataFrame,
+) -> dict[str, Any]:
+ """Mirror Streamlit's historical + synthetic next-race safety-car inference."""
+ frame = _load_safety_car_inputs().copy()
+ model = _load_safety_car_model()
+ if frame.empty or model is None or "SafetyCarStatus" not in frame:
+ return {"rows": [], "mean": None, "min": None, "max": None}
+
+ manifest_path = DATA_DIR / "models" / "safetycar_manifest.json"
+ if not manifest_path.is_file():
+ return {"rows": [], "mean": None, "min": None, "max": None}
+ import json
+
+ try:
+ feature_names = json.loads(manifest_path.read_text(encoding="utf-8")).get("feature_names", [])
+ except (OSError, json.JSONDecodeError):
+ feature_names = []
+ if not feature_names:
+ return {"rows": [], "mean": None, "min": None, "max": None}
+
+ for column in feature_names:
+ if column not in frame:
+ frame[column] = np.nan
+ features = frame[feature_names].copy()
+ try:
+ probabilities = model.predict_proba(features)[:, 1]
+ except Exception:
+ return {"rows": [], "mean": None, "min": None, "max": None}
+
+ history = pd.DataFrame({
+ "grandPrixName": frame.get("grandPrixName"),
+ "grandPrixYear": frame.get("grandPrixYear"),
+ "PredictedSafetyCarProbabilityPercentage": (probabilities * 100).round(3),
+ })
+ history["Type"] = "Historical"
+
+ synthetic: dict[str, Any] = dict.fromkeys(feature_names, np.nan)
+ synthetic["grandPrixYear"] = year
+ synthetic["grandPrixName"] = race_name
+ schedule_map = {
+ "circuitId": "circuitId",
+ "grandPrixLaps": "laps",
+ "turns": "turns",
+ "streetRace": "streetRace",
+ "trackRace": "trackRace",
+ }
+ for target, source in schedule_map.items():
+ if target in synthetic and source in next_race.index and pd.notna(next_race[source]):
+ synthetic[target] = next_race[source]
+
+ if not weather.empty:
+ weather_row = weather.iloc[0]
+ for column in ("average_temp", "average_humidity", "average_wind_speed", "total_precipitation"):
+ if column in synthetic and column in weather_row.index and pd.notna(weather_row[column]):
+ synthetic[column] = weather_row[column]
+
+ same_gp = frame[frame.get("grandPrixName", pd.Series(index=frame.index, dtype=object)) == race_name]
+ for column in feature_names:
+ if not pd.isna(synthetic[column]) or column not in frame:
+ continue
+ if pd.api.types.is_numeric_dtype(frame[column]):
+ values = pd.Series(dtype=float)
+ if not same_gp.empty and "grandPrixYear" in same_gp:
+ per_race = same_gp.groupby("grandPrixYear")[column].mean(numeric_only=True).dropna()
+ if not per_race.empty:
+ values = per_race.sort_index().tail(2)
+ synthetic[column] = (
+ values.median()
+ if not values.empty
+ else pd.to_numeric(frame[column], errors="coerce").dropna().median()
+ )
+
+ synthetic_frame = pd.DataFrame([synthetic], columns=feature_names)
+ try:
+ next_probability = float(model.predict_proba(synthetic_frame)[:, 1][0])
+ except Exception:
+ next_probability = float("nan")
+
+ current = history[
+ (history["grandPrixName"].astype(str) == race_name)
+ & (pd.to_numeric(history["grandPrixYear"], errors="coerce") != year)
+ ].drop_duplicates(subset=["grandPrixYear"])
+
+ if np.isfinite(next_probability):
+ current = pd.concat([
+ current,
+ pd.DataFrame([{
+ "grandPrixName": race_name,
+ "grandPrixYear": year,
+ "PredictedSafetyCarProbabilityPercentage": round(next_probability * 100, 3),
+ "Type": "Next Race",
+ }]),
+ ], ignore_index=True)
+
+ current = current.sort_values("grandPrixYear", ascending=False)
+ percentage = history["PredictedSafetyCarProbabilityPercentage"]
+ return {
+ "rows": records(current),
+ "mean": float(percentage.mean()),
+ "min": float(percentage.min()),
+ "max": float(percentage.max()),
+ }
+
def next_race_bundle() -> dict[str, Any]:
schedule = load_race_schedule().copy()
date_col = "date" if "date" in schedule else ("short_date" if "short_date" in schedule else None)
@@ -451,17 +910,40 @@ def next_race_bundle() -> dict[str, Any]:
messages = messages[messages["grandPrixId"].astype(str) == str(race_id)]
predictions = find_prediction_artifact(str(race_id), str(year), str(race_name), row[date_col])
+ legacy_predictions = _legacy_prediction_rows(str(race_id), int(year), str(race_name))
+ dnf_predictions = legacy_predictions or build_dnf_predictions(predictions, row, str(race_name), weather)
+ safety_car_predictions = build_safety_car_predictions(row, str(race_name), int(year), weather)
pit_stops = fastest_pit_stops(str(race_id))
+ position_mae_by_position = _position_mae_by_position()
+ try:
+ manifest = model_manifest("XGBoost") or {}
+ except (KeyError, OSError, ValueError):
+ manifest = {}
+ model_mae = (manifest.get("metrics") or {}).get("mae")
+ if isinstance(predictions, dict) and predictions.get("format") == "json":
+ by_model = predictions.get("predictions_by_model") or {}
+ xgboost_block = by_model.get("xgboost") or (next(iter(by_model.values()), {}) if by_model else {})
+ model_mae = xgboost_block.get("model_mae", model_mae)
return {
"next_race": records(next_frame)[0],
"race_id": None if race_id is None else str(race_id),
"race_name": str(race_name),
"year": int(year) if pd.notna(year) else None,
- "past_results": records(past.drop_duplicates().head(1000)),
+ "past_results": records(
+ past.drop_duplicates(
+ subset=[column for column in ("resultsDriverName", "grandPrixYear") if column in past.columns]
+ ).head(1000)
+ ),
"driver_performance": records(driver_perf),
"constructor_performance": records(constructor_perf),
"weather": records(weather.head(500)),
"race_messages": records(messages.head(500)),
"fastest_pit_stops": pit_stops,
"predictions": predictions,
+ "legacy_predictions": legacy_predictions,
+ "dnf_predictions": dnf_predictions,
+ "dnf_diagnostics": dnf_diagnostics(),
+ "safety_car_predictions": safety_car_predictions,
+ "model_mae": model_mae,
+ "position_mae_by_position": position_mae_by_position,
}
diff --git a/fastapi_react/backend/app/services/data.py b/fastapi_react/backend/app/services/data.py
index 00b6587c..d31df622 100644
--- a/fastapi_react/backend/app/services/data.py
+++ b/fastapi_react/backend/app/services/data.py
@@ -49,6 +49,51 @@ def load_main_data() -> pd.DataFrame:
if not MAIN_DATA.exists():
raise FileNotFoundError(f"Missing required dataset: {MAIN_DATA}")
df = pd.read_csv(MAIN_DATA, sep="\t", low_memory=False)
+
+ # Streamlit's get_shared_dataset() recreates several canonical driver/constructor
+ # statistics from legacy merge-suffixed columns before it builds the filter
+ # sidebar. Mirror those aliases here so the API exposes the same filters and
+ # query semantics instead of silently dropping them.
+ streamlit_aliases = {
+ "bestChampionshipPosition": "bestChampionshipPosition_results_with_qualifying",
+ "bestStartingGridPosition": "bestStartingGridPosition_results_with_qualifying",
+ "bestRaceResult": "bestRaceResult_results_with_qualifying",
+ "totalChampionshipWins": "totalChampionshipWins_results_with_qualifying",
+ "totalRaceStarts": "totalRaceStarts_results_with_qualifying",
+ "totalRaceWins": "totalRaceWins_results_with_qualifying",
+ "totalRaceLaps": "totalRaceLaps_results_with_qualifying",
+ "totalPodiums": "totalPodiums_results_with_qualifying",
+ "totalPoints": "totalPoints_results_with_qualifying",
+ "totalChampionshipPoints": "totalChampionshipPoints_results_with_qualifying",
+ "totalFastestLaps": "totalFastestLaps_results_with_qualifying",
+ "totalRaceEntries": "totalRaceEntries_results_with_qualifying",
+ }
+ for canonical, legacy in streamlit_aliases.items():
+ if canonical not in df.columns and legacy in df.columns:
+ df[canonical] = df[legacy]
+
+ # The Streamlit dataset then merges the current constructor and driver
+ # standings before column_names is created. Use key-based maps rather
+ # than a dataframe merge so the API keeps one row per race/driver while
+ # exposing the identical current-standings filter fields.
+ constructor_standings_path = DATA_DIR / "constructor_standings.csv"
+ if constructor_standings_path.exists() and "constructorId_results" in df.columns:
+ standings = pd.read_csv(constructor_standings_path, sep="\t", low_memory=False)
+ if "id" in standings.columns:
+ keyed = standings.drop_duplicates("id").set_index("id")
+ for column in standings.columns:
+ if column in {"id", "name", "fullName", "countryId", "TeamName"}:
+ continue
+ df[column] = df["constructorId_results"].map(keyed[column])
+
+ driver_standings_path = DATA_DIR / "driver_standings.csv"
+ if driver_standings_path.exists() and "resultsDriverId" in df.columns:
+ standings = pd.read_csv(driver_standings_path, sep="\t", low_memory=False)
+ if "driverId" in standings.columns:
+ keyed = standings.drop_duplicates("driverId").set_index("driverId")
+ if "driverRank" in keyed.columns:
+ df["driverRank"] = df["resultsDriverId"].map(keyed["driverRank"])
+
for candidate in ("short_date", "date", "grandPrixDate"):
if candidate in df.columns:
df[candidate] = pd.to_datetime(df[candidate], errors="coerce")
@@ -104,13 +149,15 @@ def apply_filters(df: pd.DataFrame, filters: list[Any]) -> pd.DataFrame:
@lru_cache(maxsize=1)
-def streamlit_filter_rules() -> tuple[dict[str, str], frozenset[str]]:
- """Read the Streamlit filter labels and exclusions without importing its app."""
+def streamlit_filter_rules() -> tuple[dict[str, str], frozenset[str], frozenset[str]]:
+ """Read Streamlit filter labels, exclusions, and its loaded-column contract."""
source_path = REPO_ROOT / "raceAnalysis.py"
tree = ast.parse(source_path.read_text(encoding="utf-8"), filename=str(source_path))
- literal_names = {"column_rename_for_filter", "exclusionList", "suffixes_to_exclude"}
+ literal_names = {"column_rename_for_filter", "exclusionList", "suffixes_to_exclude", "selected_columns"}
values: dict[str, Any] = {}
- for node in tree.body:
+ # selected_columns lives inside load_data(); walk the tree so the API
+ # follows the same authoritative loaded-column contract as Streamlit.
+ for node in ast.walk(tree):
if not isinstance(node, ast.Assign):
continue
for target in node.targets:
@@ -124,15 +171,20 @@ def streamlit_filter_rules() -> tuple[dict[str, str], frozenset[str]]:
excluded = set(values.get("exclusionList", ()))
suffixes = values.get("suffixes_to_exclude", ())
excluded.update(column for column in load_main_data().columns if column.endswith(tuple(suffixes)))
- return labels, frozenset(excluded)
-
+ selected = set(values.get("selected_columns", ()))
+ # These friendly-name fields include columns introduced by the standings
+ # merges (for example Points and bestChampionshipPosition).
+ selected.update(labels)
+ return labels, frozenset(excluded), frozenset(selected)
def filter_schema() -> list[dict[str, Any]]:
df = load_main_data()
- labels, excluded = streamlit_filter_rules()
+ labels, excluded, selected = streamlit_filter_rules()
schema: list[dict[str, Any]] = []
+ # Streamlit explicitly sorts `column_names` before building sidebar controls.
+ # Preserve that raw-field alphabetical order; labels are applied only for display.
for column in sorted(df.columns):
- if column in excluded:
+ if column in excluded or (selected and column not in selected):
continue
series = df[column]
non_null = series.dropna()
@@ -168,7 +220,13 @@ def query_main(request: Any) -> dict[str, Any]:
if request.sort:
valid = [c for c in request.sort if c in df.columns]
if valid:
- df = df.sort_values(valid, ascending=not request.descending)
+ sort_ascending: bool | list[bool]
+ if request.ascending and len(request.ascending) == len(request.sort):
+ direction_map = dict(zip(request.sort, request.ascending, strict=True))
+ sort_ascending = [direction_map[column] for column in valid]
+ else:
+ sort_ascending = not request.descending
+ df = df.sort_values(valid, ascending=sort_ascending)
if request.columns:
valid = [c for c in request.columns if c in df.columns]
if valid:
diff --git a/fastapi_react/backend/test_api.py b/fastapi_react/backend/test_api.py
index b14ea211..da9cef69 100644
--- a/fastapi_react/backend/test_api.py
+++ b/fastapi_react/backend/test_api.py
@@ -6,6 +6,7 @@
"""
from __future__ import annotations
+import numpy as np
import pandas as pd
import pytest
from fastapi.testclient import TestClient
@@ -58,6 +59,28 @@ def test_data_explorer_schema_returns_filters() -> None:
assert body["filters"], "schema should return at least one filterable column"
+def test_data_explorer_schema_matches_streamlit_standings_filters() -> None:
+ response = client.get("/api/data-explorer/schema")
+ assert response.status_code == 200
+ filters = {item["column"]: item for item in response.json()["filters"]}
+
+ expected_labels = {
+ "Points": "Current Year Points (Driver)",
+ "bestChampionshipPosition": "Best Champ Pos.",
+ "bestRaceResult": "Best Race Result",
+ "bestStartingGridPosition": "Best Starting Grid Pos.",
+ "constructorRank": "Constructor Rank",
+ "driverRank": "Driver Rank",
+ }
+ for column, label in expected_labels.items():
+ assert column in filters
+ assert filters[column]["label"] == label
+
+ assert filters["Points"]["kind"] == "range"
+ assert filters["bestChampionshipPosition"]["kind"] == "range"
+ assert filters["constructorRank"]["kind"] == "range"
+
+
def test_data_explorer_query_unfiltered() -> None:
response = client.post("/api/data-explorer/query", json={"limit": 5})
assert response.status_code == 200
@@ -165,6 +188,53 @@ def test_next_race_endpoint() -> None:
assert "fastest_pit_stops" in body
+def test_safety_car_loader_matches_streamlit_search_order(
+ monkeypatch: pytest.MonkeyPatch,
+ tmp_path,
+) -> None:
+ import pickle
+
+ models = tmp_path / "models"
+ xgboost = models / "xgboost"
+ xgboost.mkdir(parents=True)
+ with (xgboost / "safetycar_model.pkl").open("wb") as handle:
+ pickle.dump({"model": "wrapped-safety-model"}, handle)
+
+ monkeypatch.setattr(analysis, "DATA_DIR", tmp_path)
+ analysis._load_safety_car_model.cache_clear()
+ try:
+ assert analysis._load_safety_car_model() == "wrapped-safety-model"
+ finally:
+ analysis._load_safety_car_model.cache_clear()
+
+
+def test_safety_car_predictions_include_historical_and_next_race(monkeypatch: pytest.MonkeyPatch) -> None:
+ class FakeSafetyModel:
+ def predict_proba(self, frame: pd.DataFrame) -> np.ndarray:
+ return np.tile(np.array([[0.25, 0.75]]), (len(frame), 1))
+
+ history = pd.DataFrame({
+ "grandPrixName": ["Singapore Grand Prix", "Singapore Grand Prix"],
+ "grandPrixYear": [2024, 2025],
+ "turns": [19, 19],
+ "SafetyCarStatus": [1, 0],
+ })
+ monkeypatch.setattr(analysis, "_load_safety_car_inputs", lambda: history)
+ monkeypatch.setattr(analysis, "_load_safety_car_model", lambda: FakeSafetyModel())
+
+ payload = analysis.build_safety_car_predictions(
+ pd.Series({"turns": 19}),
+ "Singapore Grand Prix",
+ 2026,
+ pd.DataFrame(),
+ )
+
+ assert payload["mean"] == pytest.approx(75.0)
+ assert payload["rows"][0]["grandPrixYear"] == 2026
+ assert payload["rows"][0]["Type"] == "Next Race"
+ assert len(payload["rows"]) == 3
+
+
def test_tire_strategy_endpoint_returns_selected_race_and_year_summary() -> None:
response = client.get("/api/analytics/tire-strategy")
assert response.status_code == 200
diff --git a/fastapi_react/frontend/src/App.jsx b/fastapi_react/frontend/src/App.jsx
index abe00fff..adffd3c7 100644
--- a/fastapi_react/frontend/src/App.jsx
+++ b/fastapi_react/frontend/src/App.jsx
@@ -1,4 +1,4 @@
-import { useEffect, useState } from 'react'
+import { useEffect, useRef, useState } from "react";
import { api } from "./api";
import DataExplorer from "./pages/DataExplorer";
import Analytics from "./pages/Analytics";
@@ -7,99 +7,102 @@ import NextRace from "./pages/NextRace";
import Models from "./pages/Models";
import RawData from "./pages/RawData";
import BettingResearch from "./pages/BettingResearch";
+import FilterSidebar from "./components/FilterSidebar";
-const pages = {
- "Data Explorer": DataExplorer,
- "Analytics": Analytics,
- "Current Season": CurrentSeason,
- "Next Race": NextRace,
- "Predictive Models": Models,
- "Raw Data": RawData,
- "Betting Research": BettingResearch,
-};
+const pages = [
+ { key: "Data Explorer", label: "📊 Data Explorer", Component: DataExplorer },
+ { key: "Analytics", label: "📈 Analytics & Visualizations", Component: Analytics },
+ { key: "Current Season", label: "🏎️ Schedule", Component: CurrentSeason },
+ { key: "Next Race", label: "🏁 Next Race", Component: NextRace },
+ { key: "Predictive Models", label: "🤖 Predictive Models", Component: Models },
+ { key: "Raw Data", label: "💾 Data & Debug", Component: RawData },
+ { key: "Betting Research", label: "📐 Betting Research", Component: BettingResearch },
+];
-const icons = {
- "Data Explorer": "\u25A6",
- "Analytics": "\u2301",
- "Current Season": "\u25F7",
- "Next Race": "\uD83C\uDFC1",
- "Predictive Models": "\u25C6",
- "Raw Data": "\u2261",
- "Betting Research": "\uD83D\uDCD0",
-};
-
-const BASE_TITLE = "F1 Analysis";
+const BASE_TITLE = "Gridlocked - Formula 1 Betting & Analytics";
export default function App() {
const [active, setActive] = useState("Data Explorer");
- const [health, setHealth] = useState(null);
- const [theme, setTheme] = useState(() => {
- try { return localStorage.getItem("f1analysis.theme") === "light" ? "light" : "dark"; }
- catch { return "dark"; }
+ const [meta, setMeta] = useState(null);
+ const [filterRevision, setFilterRevision] = useState(0);
+ const tabStripRef = useRef(null);
+ const [filtersActive, setFiltersActive] = useState(() => {
+ try { return Boolean(JSON.parse(sessionStorage.getItem("f1analysis.filters") || "null")?.applied); }
+ catch { return false; }
});
useEffect(() => {
- api.get("/api/health").then(setHealth).catch(() => {});
+ api.get("/api/meta").then(setMeta).catch(() => {});
const hash = decodeURIComponent(location.hash.replace("#/", ""));
- if (pages[hash]) setActive(hash);
- }, []);
+ if (pages.some(page => page.key === hash)) setActive(hash);
- useEffect(() => {
- document.title = `${active} \u2014 ${BASE_TITLE}`;
- }, [active]);
+ const syncFilters = () => {
+ try { setFiltersActive(Boolean(JSON.parse(sessionStorage.getItem("f1analysis.filters") || "null")?.applied)); }
+ catch { setFiltersActive(false); }
+ setFilterRevision(value => value + 1);
+ };
+ window.addEventListener("f1analysis:filters-changed", syncFilters);
+ return () => window.removeEventListener("f1analysis:filters-changed", syncFilters);
+ }, []);
useEffect(() => {
- document.documentElement.dataset.theme = theme;
- try { localStorage.setItem("f1analysis.theme", theme); } catch { /* storage is optional */ }
- }, [theme]);
+ document.title = BASE_TITLE;
+ const activeTab = tabStripRef.current?.querySelector('[aria-selected="true"]');
+ activeTab?.scrollIntoView?.({ block: "nearest", inline: "nearest" });
+ }, [active, filtersActive]);
function navigate(page) {
setActive(page);
location.hash = `/${encodeURIComponent(page)}`;
- window.scrollTo({ top: 0, behavior: "smooth" });
+ window.scrollTo({ top: 0, behavior: "auto" });
}
- const Page = pages[active];
+ const Page = pages.find(page => page.key === active)?.Component || DataExplorer;
+ const startYear = meta?.race_start_year ?? 2016;
+ const currentYear = meta?.current_year ?? new Date().getFullYear();
+
return (
-
+
@@ -59,7 +59,7 @@ export function LinePanel({ title, rows = [], x, y, xLabel = axisLabels[x] || x,
-
+
@@ -80,7 +80,28 @@ export function BarPanel({ title, rows = [], x, y, xLabel = axisLabels[x] || x,
-
+
+
+
+
+
+ );
+}
+
+/** @param {{ title: string, rows?: Array>, x: string, ys: string[], xLabel?: string, yLabel?: string }} props */
+export function MultiBarPanel({ title, rows = [], x, ys, xLabel = axisLabels[x] || x, yLabel = "Count" }) {
+ if (!rows.length) return null;
+ return (
+
+
+
+
+
+
+
+
+
+ {ys.map((key, index) => )}
@@ -108,3 +129,62 @@ export function RegressionPanel({ title, points = [], fit = [], x, y, xLabel, yL
);
}
+
+
+const SERIES_COLORS = [
+ "#0068c9", "#ff4b4b", "#00a86b", "#7d3cff", "#f0a202", "#2a9d8f",
+ "#e76f51", "#264653", "#8d99ae", "#9b5de5", "#00bbf9", "#f15bb5",
+];
+
+/** @param {{ title: string, rows?: Array>, x: string, y: string, series: string, xLabel?: string, yLabel?: string }} props */
+export function MultiLinePanel({ title, rows = [], x, y, series, xLabel = axisLabels[x] || x, yLabel = axisLabels[y] || y }) {
+ if (!rows.length) return null;
+ const seriesNames = [...new Set(rows.map(row => String(row[series] ?? "")).filter(Boolean))];
+ const xValues = [...new Set(rows.map(row => row[x]))];
+ const byX = xValues.map(xValue => {
+ /** @type {Record} */
+ const point = { [x]: xValue };
+ for (const row of rows.filter(item => item[x] === xValue)) {
+ point[String(row[series])] = row[y];
+ }
+ return point;
+ });
+ return (
+
+
+
+
+
+
+
+
+
+ {seriesNames.map((name, index) => (
+
+ ))}
+
+
+
+
+ );
+}
+
+/** @param {{ title: string, rows?: Array>, nameKey: string, valueKey: string }} props */
+export function PiePanel({ title, rows = [], nameKey, valueKey }) {
+ if (!rows.length) return null;
+ return (
+
+
+
+
+
+ {rows.map((row, index) => | )}
+
+
+
+
+
+
+
+ );
+}
diff --git a/fastapi_react/frontend/src/components/Charts.test.jsx b/fastapi_react/frontend/src/components/Charts.test.jsx
index 0df30d9f..4d8641fe 100644
--- a/fastapi_react/frontend/src/components/Charts.test.jsx
+++ b/fastapi_react/frontend/src/components/Charts.test.jsx
@@ -1,49 +1,54 @@
import { describe, it, expect, beforeEach } from 'vitest';
import { render } from '@testing-library/react';
-import { ScatterPanel, LinePanel, BarPanel } from './Charts.jsx';
+import {
+ ScatterPanel, LinePanel, BarPanel, MultiBarPanel, RegressionPanel, MultiLinePanel, PiePanel,
+} from './Charts.jsx';
beforeEach(() => {
- // ResponsiveContainer measures its parent; jsdom gives it 0x0 by default
- // which makes Recharts charts render as empty SVGs in tests. We mock
- // getBoundingClientRect on the container's parent so charts have a size.
Object.defineProperty(HTMLElement.prototype, 'getBoundingClientRect', {
configurable: true,
value: () => ({ width: 800, height: 400, top: 0, left: 0, right: 800, bottom: 400, x: 0, y: 0 }),
});
});
-describe('ScatterPanel', () => {
- it('returns null for empty rows', () => {
- const { container } = render();
+describe('chart panels', () => {
+ it.each([
+ ['scatter', ScatterPanel, { title: 'Scatter', rows: [], x: 'a', y: 'b' }],
+ ['line', LinePanel, { title: 'Line', rows: [], x: 'a', y: 'b' }],
+ ['bar', BarPanel, { title: 'Bar', rows: [], x: 'a', y: 'b' }],
+ ['multi bar', MultiBarPanel, { title: 'Multi Bar', rows: [], x: 'a', ys: ['b', 'c'] }],
+ ['regression', RegressionPanel, { title: 'Regression', points: [], fit: [], x: 'a', y: 'b', xLabel: 'A', yLabel: 'B' }],
+ ['multi line', MultiLinePanel, { title: 'Multi Line', rows: [], x: 'a', y: 'b', series: 'series' }],
+ ['pie', PiePanel, { title: 'Pie', rows: [], nameKey: 'name', valueKey: 'value' }],
+ ])('returns null for empty %s data', (_name, Component, props) => {
+ const { container } = render();
expect(container).toBeEmptyDOMElement();
});
- it('renders a card with title for non-empty rows', () => {
- render();
- expect(document.querySelector('.card')).toBeInTheDocument();
- });
-});
+ it('renders all supported chart types with accessible chart containers', () => {
+ const { rerender } = render();
+ expect(document.querySelector('[role="img"]')).toHaveAttribute('aria-label', expect.stringContaining('Scatter'));
-describe('LinePanel', () => {
- it('returns null for empty rows', () => {
- const { container } = render();
- expect(container).toBeEmptyDOMElement();
- });
+ rerender();
+ expect(document.querySelector('.card')).toBeInTheDocument();
- it('renders a chart container for non-empty rows', () => {
- render();
+ rerender();
expect(document.querySelector('.card')).toBeInTheDocument();
- });
-});
-describe('BarPanel', () => {
- it('returns null for empty rows', () => {
- const { container } = render();
- expect(container).toBeEmptyDOMElement();
- });
+ rerender();
+ expect(document.querySelector('[aria-label*="2 series"]')).toBeInTheDocument();
- it('renders a chart container for non-empty rows', () => {
- render();
- expect(document.querySelector('.card')).toBeInTheDocument();
+ rerender();
+ expect(document.querySelector('[aria-label*="Regression fit"]')).toBeInTheDocument();
+
+ rerender();
+ expect(document.querySelector('[aria-label*="2 series"]')).toBeInTheDocument();
+
+ rerender();
+ expect(document.querySelector('[aria-label*="2 categories"]')).toBeInTheDocument();
});
});
diff --git a/fastapi_react/frontend/src/components/FilterSidebar.jsx b/fastapi_react/frontend/src/components/FilterSidebar.jsx
new file mode 100644
index 00000000..0078218d
--- /dev/null
+++ b/fastapi_react/frontend/src/components/FilterSidebar.jsx
@@ -0,0 +1,138 @@
+import { useEffect, useMemo, useState } from "react";
+import { api } from "../api";
+
+function FilterControl({ spec, value, onChange }) {
+ if (spec.kind === "boolean") {
+ return (
+
+ );
+ }
+
+ if (spec.kind === "exact" && spec.options) {
+ return (
+
+ );
+ }
+
+ if (spec.kind === "range") {
+ const low = Number(Array.isArray(value) ? value[0] : spec.min);
+ const high = Number(Array.isArray(value) ? value[1] : spec.max);
+ const min = Math.trunc(Number(spec.min));
+ const max = Math.trunc(Number(spec.max));
+ const currentLow = Math.trunc(Number.isFinite(low) ? low : min);
+ const currentHigh = Math.trunc(Number.isFinite(high) ? high : max);
+ return (
+
+ );
+ }
+
+ if (spec.kind === "date_range") {
+ const current = Array.isArray(value) ? value : [spec.min, spec.max];
+ return (
+
+ );
+ }
+
+ return (
+
+ );
+}
+
+export default function FilterSidebar() {
+ const [schema, setSchema] = useState([]);
+ const [values, setValues] = useState(() => {
+ try { return JSON.parse(sessionStorage.getItem("f1analysis.filters") || "null")?.values || {}; }
+ catch { return {}; }
+ });
+ const [error, setError] = useState(null);
+
+ useEffect(() => {
+ api.get("/api/data-explorer/schema")
+ .then(response => setSchema(response.filters || []))
+ .catch(setError);
+ }, []);
+
+ const byName = useMemo(() => Object.fromEntries(schema.map(spec => [spec.column, spec])), [schema]);
+
+ function activeFilters(nextValues) {
+ return Object.entries(nextValues).flatMap(([column, value]) => {
+ if (value == null || value === "" || (Array.isArray(value) && value.some(item => item === ""))) return [];
+ const spec = byName[column];
+ if (!spec) return [];
+ let normalized = value;
+ if (spec.kind === "range") normalized = value.map(Number);
+ return [{ column, kind: spec.kind, value: normalized }];
+ });
+ }
+
+ function apply(column, value) {
+ const next = { ...values, [column]: value };
+ setValues(next);
+ const filters = activeFilters(next);
+ sessionStorage.setItem("f1analysis.filters", JSON.stringify({ applied: true, filters, values: next }));
+ window.dispatchEvent(new CustomEvent("f1analysis:filters-changed"));
+ }
+
+ return (
+
+ );
+}
diff --git a/fastapi_react/frontend/src/components/UI.jsx b/fastapi_react/frontend/src/components/UI.jsx
index aa0d94bb..419e6930 100644
--- a/fastapi_react/frontend/src/components/UI.jsx
+++ b/fastapi_react/frontend/src/components/UI.jsx
@@ -38,20 +38,21 @@ export function Status(props = {}) {
return children || null;
}
-/** @param {{ rows?: Array>, columns?: string[], maxHeight?: number, ariaLabel?: string }} props */
+/** @param {{ rows?: Array>, columns?: string[], maxHeight?: number, ariaLabel?: string, headerMap?: Record, checkboxColumns?: string[] }} props */
/* eslint-disable jsx-a11y/no-noninteractive-tabindex */
export function DataTable(props = {}) {
- const { rows = [], columns = undefined, maxHeight = 560, ariaLabel = "Data table, scrollable region" } = props;
+ const { rows = [], columns = undefined, maxHeight = 560, ariaLabel = undefined, headerMap = {}, checkboxColumns = [] } = props;
if (!rows?.length) return No rows available.
;
const cols = columns?.length ? columns : Object.keys(rows[0] || {});
+ const landmarkProps = ariaLabel ? { role: "region", "aria-label": ariaLabel } : {};
return (
-
+
- {cols.map(c => | {c} | )}
+ {cols.map(c => | {headerMap[c] || c} | )}
{rows.map((row, i) => (
- {cols.map(c => | {formatCell(row[c])} | )}
+ {cols.map(c => {checkboxColumns.includes(c) ? : formatCell(row[c])} | )}
))}
@@ -82,9 +83,9 @@ export function Metric({ label, value }) {
/** @param {{ tabs: string[], active: string, onChange: (tab: string) => void }} props */
export function Tabs({ tabs, active, onChange }) {
return (
-
+
{tabs.map((tab, index) => (
-