diff --git a/.github/workflows/fastapi-react.yml b/.github/workflows/fastapi-react.yml index 46ee8fe7..86dcd6e4 100644 --- a/.github/workflows/fastapi-react.yml +++ b/.github/workflows/fastapi-react.yml @@ -3,6 +3,10 @@ name: FastAPI + React parity checks permissions: contents: read +concurrency: + group: fastapi-react-parity-${{ github.ref }} + cancel-in-progress: true + on: pull_request: paths: @@ -70,3 +74,115 @@ jobs: - name: Production-deps audit run: npm run audit continue-on-error: true # dev-dep advisories only; see PARITY_REPORT.md + + + visual-parity: + name: Visual parity evidence + runs-on: ubuntu-latest + needs: [backend, frontend] + timeout-minutes: 30 + steps: + - uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4 + - uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6 + with: + python-version: "3.12" + cache: pip + cache-dependency-path: fastapi_react/backend/requirements.txt + - uses: actions/setup-node@cdca7365b2dadb8aad0a33bc7601856ffabcc48e # v4.3.0 + with: + node-version: "20" + cache: npm + cache-dependency-path: fastapi_react/frontend/package-lock.json + - name: Install backend + working-directory: fastapi_react/backend + run: | + python -m pip install --upgrade pip + pip install -r requirements.txt + - name: Install Streamlit reference runtime + run: pip install -r requirements-web.txt + - name: Install frontend and browser + working-directory: fastapi_react/frontend + run: | + npm ci + npx playwright install --with-deps chromium + - name: Start parity stack + env: + F1_REPO_ROOT: ${{ github.workspace }} + ENABLE_EXPENSIVE_TOOLS: "0" + F1_RESEARCH_MODE: "0" + STREAMLIT_SERVER_HEADLESS: "true" + run: | + (cd fastapi_react/backend && uvicorn app.main:app --host 127.0.0.1 --port 8000 > /tmp/f1-backend.log 2>&1 & echo $! > /tmp/f1-backend.pid) + (cd fastapi_react/frontend && npm run dev -- --host 127.0.0.1 --port 5173 > /tmp/f1-frontend.log 2>&1 & echo $! > /tmp/f1-frontend.pid) + (streamlit run raceAnalysis.py --server.address 127.0.0.1 --server.port 8501 --server.headless true > /tmp/f1-streamlit.log 2>&1 & echo $! > /tmp/f1-streamlit.pid) + for i in {1..90}; do + if curl -fsS http://127.0.0.1:8000/api/health >/dev/null && curl -fsS http://127.0.0.1:5173 >/dev/null && curl -fsS http://127.0.0.1:8501/_stcore/health >/dev/null; then + exit 0 + fi + sleep 2 + done + cat /tmp/f1-backend.log || true + cat /tmp/f1-frontend.log || true + cat /tmp/f1-streamlit.log || true + exit 1 + - name: Capture React reference + working-directory: fastapi_react/frontend + env: + REACT_BASE_URL: http://127.0.0.1:5173 + REACT_WAIT_MS: "4000" + run: npm run capture:react + - name: Accessibility audit + id: accessibility + continue-on-error: true + working-directory: fastapi_react/frontend + env: + REACT_BASE_URL: http://127.0.0.1:5173 + run: npm run audit:a11y + - name: Operational benchmark + id: benchmark + continue-on-error: true + working-directory: fastapi_react/frontend + env: + BENCH_BASE_URL: http://127.0.0.1:5173 + BENCH_API_URL: http://127.0.0.1:8000 + run: npm run benchmark + - name: Capture local Streamlit reference + id: streamlit_capture + continue-on-error: true + working-directory: fastapi_react/frontend + env: + STREAMLIT_BASE_URL: http://127.0.0.1:8501 + STREAMLIT_WAIT_MS: "5000" + run: npm run capture:streamlit + - name: Diff screenshots + id: visual_diff + continue-on-error: true + working-directory: fastapi_react/frontend + run: npm run capture:diff + - name: Upload visual parity evidence + if: always() + uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6 + with: + name: fastapi-react-parity-evidence + path: | + fastapi_react/parity_evidence/visual + fastapi_react/parity_evidence/accessibility.json + fastapi_react/parity_evidence/benchmarks.json + if-no-files-found: warn + - name: Enforce parity evidence gates + if: always() + env: + ACCESSIBILITY_OUTCOME: ${{ steps.accessibility.outcome }} + BENCHMARK_OUTCOME: ${{ steps.benchmark.outcome }} + STREAMLIT_CAPTURE_OUTCOME: ${{ steps.streamlit_capture.outcome }} + VISUAL_DIFF_OUTCOME: ${{ steps.visual_diff.outcome }} + run: | + failed=0 + for gate in ACCESSIBILITY_OUTCOME BENCHMARK_OUTCOME STREAMLIT_CAPTURE_OUTCOME VISUAL_DIFF_OUTCOME; do + outcome="${!gate}" + echo "$gate=$outcome" + if [ "$outcome" != "success" ]; then + failed=1 + fi + done + exit "$failed" diff --git a/fastapi_react/backend/app/main.py b/fastapi_react/backend/app/main.py index a9095d85..23da0f45 100644 --- a/fastapi_react/backend/app/main.py +++ b/fastapi_react/backend/app/main.py @@ -1,5 +1,6 @@ from __future__ import annotations +import datetime import os from typing import Any @@ -22,6 +23,7 @@ from app.services.data import ( filter_schema, list_data_files, + load_main_data, model_manifest, precomputed, query_main, @@ -81,6 +83,16 @@ def brand_logo() -> FileResponse: @app.get("/api/meta") def meta() -> dict[str, Any]: + data = load_main_data() + years = data.get("grandPrixYear") + race_start_year = int(years.min()) if years is not None and years.notna().any() else 2016 + current_year = int(years.max()) if years is not None and years.notna().any() else datetime.datetime.now().year + candidates = [path for path in DATA_DIR.iterdir() if path.is_file()] if DATA_DIR.exists() else [] + latest = max(candidates, key=lambda path: path.stat().st_mtime, default=None) + last_updated = ( + datetime.datetime.fromtimestamp(latest.stat().st_mtime).strftime("%Y-%m-%d %I:%M %p") + if latest is not None else "No data files found" + ) return { "tabs": [ "Data Explorer", "Analytics", "Current Season", "Next Race", @@ -89,6 +101,10 @@ def meta() -> dict[str, Any]: "models": MODEL_TYPES, "expensive_tools_enabled": ENABLE_EXPENSIVE_TOOLS, "manual_tools": list(TOOLS), + "race_start_year": race_start_year, + "current_year": current_year, + "last_updated": last_updated, + "code_deployed_at": datetime.datetime.now(datetime.UTC).strftime("%Y-%m-%d %H:%M:%S UTC"), } diff --git a/fastapi_react/backend/app/schemas.py b/fastapi_react/backend/app/schemas.py index 432bd1cd..573ff98b 100644 --- a/fastapi_react/backend/app/schemas.py +++ b/fastapi_react/backend/app/schemas.py @@ -15,6 +15,7 @@ class QueryRequest(BaseModel): filters: list[FilterSpec] = Field(default_factory=list) columns: list[str] | None = None sort: list[str] = Field(default_factory=list) + ascending: list[bool] | None = None descending: bool = False offset: int = 0 limit: int = Field(default=200, ge=1, le=5000) diff --git a/fastapi_react/backend/app/services/analysis.py b/fastapi_react/backend/app/services/analysis.py index 6ef139da..f1f130fa 100644 --- a/fastapi_react/backend/app/services/analysis.py +++ b/fastapi_react/backend/app/services/analysis.py @@ -1,5 +1,6 @@ from __future__ import annotations +import pickle from functools import lru_cache from pathlib import Path from typing import Any @@ -9,7 +10,14 @@ from scipy.stats import linregress from app.config import DATA_DIR -from app.services.data import apply_filters, load_main_data, load_race_schedule, records +from app.services.data import ( + apply_filters, + load_main_data, + load_race_schedule, + model_manifest, + precomputed, + records, +) def _regression(df: pd.DataFrame, x_col: str, y_col: str) -> dict[str, Any] | None: @@ -52,8 +60,12 @@ def _regression_series(df: pd.DataFrame, x_col: str, y_col: str) -> dict[str, An def analytics(filters: Any, max_rows: int) -> dict[str, Any]: + """Return every lightweight analysis block rendered by the Streamlit Analytics tab.""" df = apply_filters(load_main_data(), filters).head(max_rows).copy() payload: dict[str, Any] = {"rows_considered": len(df), "charts": {}, "regressions": []} + if df.empty: + return payload + pairs = { "active_years_vs_final": ("resultsFinalPositionNumber", "yearsActive"), "positions_gained_over_time": ("short_date", "positionsGained"), @@ -61,6 +73,7 @@ def analytics(filters: Any, max_rows: int) -> dict[str, Any]: "grid_vs_final": ("resultsStartingGridPositionNumber", "resultsFinalPositionNumber"), "avg_practice_vs_final": ("averagePracticePosition", "resultsFinalPositionNumber"), "pit_stop_vs_final": ("averageStopTime", "resultsFinalPositionNumber"), + "track_turns_vs_final": ("turns", "resultsFinalPositionNumber"), } for name, (x, y) in pairs.items(): if x in df and y in df: @@ -71,8 +84,14 @@ def analytics(filters: Any, max_rows: int) -> dict[str, Any]: if result: payload["regressions"].append(result) regression_titles = { - "averagePracticePosition": ("Linear Regression: Average Practice Position vs Final Position", "Average Practice Position"), - "resultsStartingGridPositionNumber": ("Linear Regression: Starting Position vs Final Position", "Starting Position"), + "averagePracticePosition": ( + "Linear Regression: Average Practice Position vs Final Position", + "Average Practice Position", + ), + "resultsStartingGridPositionNumber": ( + "Linear Regression: Starting Position vs. Final Position", + "Starting Position", + ), } payload["regression_series"] = [] for x, (title, x_label) in regression_titles.items(): @@ -90,19 +109,22 @@ def analytics(filters: Any, max_rows: int) -> dict[str, Any]: "driverBestRaceResult", "driverTotalChampionshipWins", "driverTotalPolePositions", "driverTotalRaceEntries", "driverTotalRaceStarts", "driverTotalRaceWins", "driverTotalRaceLaps", "driverTotalPodiums", "avgLapPace", "finishingTime", + "resultsQualificationPositionNumber", "numberOfStops", ) if c in df] if corr_cols: corr = df[corr_cols].apply(pd.to_numeric, errors="coerce").corr() payload["correlation"] = { "columns": list(corr.columns), "rows": [ - {"feature": idx, **{col: (None if pd.isna(v) else float(v)) for col, v in row.items()}} + {"Feature": idx, **{col: (None if pd.isna(v) else float(v)) for col, v in row.items()}} for idx, row in corr.iterrows() ], } if {"grandPrixYear", "resultsDriverName", "resultsFinalPositionNumber"}.issubset(df.columns): - agg = {"average_final_position": ("resultsFinalPositionNumber", "mean")} + agg: dict[str, tuple[str, Any]] = { + "average_final_position": ("resultsFinalPositionNumber", "mean"), + } if "resultsPodium" in df: agg["total_podiums"] = ("resultsPodium", "sum") driver = df.groupby(["grandPrixYear", "resultsDriverName"]).agg(**agg).reset_index() @@ -112,34 +134,127 @@ def analytics(filters: Any, max_rows: int) -> dict[str, Any]: constructor = ( df.groupby(["grandPrixYear", "constructorName"]) .agg( - total_wins=("resultsFinalPositionNumber", lambda s: int((s == 1).sum())), + total_wins=("resultsFinalPositionNumber", lambda values: int((values == 1).sum())), average_final_position=("resultsFinalPositionNumber", "mean"), ).reset_index() ) if "resultsPodium" in df: - podium = df.groupby(["grandPrixYear", "constructorName"])["resultsPodium"].sum().reset_index(name="total_podiums") + podium = ( + df.groupby(["grandPrixYear", "constructorName"])["resultsPodium"] + .sum().reset_index(name="total_podiums") + ) constructor = constructor.merge(podium, on=["grandPrixYear", "constructorName"], how="left") payload["constructor_performance"] = records(constructor) - if {"DNF", "resultsReasonRetired"}.issubset(df.columns): + if {"constructorName", "resultsDriverName", "positionsGained", "resultsFinalPositionNumber"}.issubset(df.columns): + driver_vs_constructor = ( + df.groupby(["constructorName", "resultsDriverName"]) + .agg( + positionsGained=("positionsGained", "sum"), + average_final_position=("resultsFinalPositionNumber", "mean"), + ) + .reset_index() + .sort_values("average_final_position") + ) + driver_vs_constructor["average_final_position"] = driver_vs_constructor["average_final_position"].round(2) + payload["driver_vs_constructor"] = records(driver_vs_constructor) + + dnf_rows = pd.DataFrame() + if "DNF" in df: + dnf_rows = df[pd.to_numeric(df["DNF"], errors="coerce").fillna(0).eq(1)].copy() + if not dnf_rows.empty and "resultsReasonRetired" in dnf_rows: dnf = ( - df[pd.to_numeric(df["DNF"], errors="coerce").fillna(0).eq(1)] - .groupby("resultsReasonRetired").size().reset_index(name="count") + dnf_rows.groupby("resultsReasonRetired").size().reset_index(name="count") .sort_values("count", ascending=False) ) payload["dnf_reasons"] = records(dnf) - dnf_group_cols = {"DNF", "resultsDriverName", "driverTotalRaceEntries"} - if dnf_group_cols.issubset(df.columns): - dnf_by_driver = ( - df[pd.to_numeric(df["DNF"], errors="coerce").eq(1)] - .groupby(["resultsDriverName", "driverTotalRaceEntries"]) - .size().reset_index(name="dnf_count") + if not dnf_rows.empty and {"resultsDriverName", "driverTotalRaceEntries"}.issubset(dnf_rows.columns): + grouped = ( + dnf_rows.groupby(["resultsDriverName", "driverTotalRaceEntries"]).size() + .reset_index(name="dnf_count") ) - entries = pd.to_numeric(dnf_by_driver["driverTotalRaceEntries"], errors="coerce") - dnf_by_driver["dnf_pct"] = (dnf_by_driver["dnf_count"] / entries * 100).round(1) - payload["dnf_by_driver"] = records(dnf_by_driver.sort_values("dnf_pct", ascending=False)) - return payload + entries = pd.to_numeric(grouped["driverTotalRaceEntries"], errors="coerce") + grouped["dnf_pct"] = (grouped["dnf_count"] / entries * 100).round(1) + payload["dnf_by_driver"] = records(grouped.sort_values("dnf_pct", ascending=False)) + if "grandPrixName" in df: + entries = df.groupby("grandPrixName").size().reset_index(name="race_entry_count") + if not dnf_rows.empty: + dnfs = dnf_rows.groupby("grandPrixName").size().reset_index(name="dnf_count") + entries = entries.merge(dnfs, on="grandPrixName", how="left") + else: + entries["dnf_count"] = 0 + entries["dnf_count"] = entries["dnf_count"].fillna(0).astype(int) + entries["dnf_pct"] = (entries["dnf_count"] / entries["race_entry_count"] * 100).round(1) + payload["dnf_by_race"] = records(entries.sort_values("dnf_pct", ascending=False)) + if "constructorName" in df: + entries = df.groupby("constructorName").size().reset_index(name="constructor_entry_count") + if not dnf_rows.empty: + dnfs = dnf_rows.groupby("constructorName").size().reset_index(name="dnf_count") + entries = entries.merge(dnfs, on="constructorName", how="left") + else: + entries["dnf_count"] = 0 + entries["dnf_count"] = entries["dnf_count"].fillna(0).astype(int) + entries["dnf_pct"] = ( + entries["dnf_count"] / entries["constructor_entry_count"] * 100 + ).round(1) + payload["dnf_by_constructor"] = records(entries.sort_values("dnf_pct", ascending=False)) + + if {"grandPrixYear", "resultsDriverName", "positionsGained", "resultsPodium"}.issubset(df.columns): + year = int(pd.to_numeric(df["grandPrixYear"], errors="coerce").max()) + season = ( + df[pd.to_numeric(df["grandPrixYear"], errors="coerce") == year] + .groupby("resultsDriverName") + .agg(positions_gained=("positionsGained", "sum"), total_podiums=("resultsPodium", "sum")) + .reset_index() + ) + payload["season_year"] = year + payload["season_summary"] = records(season) + + if {"resultsDriverName", "resultsFinalPositionNumber"}.issubset(df.columns): + consistency = ( + df.groupby("resultsDriverName") + .agg(finishing_position_std=("resultsFinalPositionNumber", "std")) + .reset_index() + .sort_values("finishing_position_std") + ) + payload["driver_consistency"] = records(consistency) + + try: + manifest = model_manifest("XGBoost") + except (KeyError, OSError, ValueError): + manifest = None + payload["model_summary"] = manifest + + try: + importance = precomputed("permutation") + except (KeyError, OSError, ValueError): + importance = None + payload["feature_importance"] = importance + try: + historical = precomputed("historical_validation") + except (KeyError, OSError, ValueError): + historical = None + holdout = (historical or {}).get("holdout", {}) if isinstance(historical, dict) else {} + holdout_rows = holdout.get("rows", []) if isinstance(holdout, dict) else [] + if isinstance(holdout_rows, list) and holdout_rows: + holdout_frame = pd.DataFrame(holdout_rows) + expected = {"ActualFinalPosition", "PredictedFinalPosition", "Error"} + if expected.issubset(holdout_frame.columns): + holdout_frame = holdout_frame.sort_values("ActualFinalPosition") + top3 = holdout_frame[pd.to_numeric(holdout_frame["ActualFinalPosition"], errors="coerce") <= 3].copy() + if not top3.empty: + payload["top3_mae"] = float( + np.mean( + np.abs( + pd.to_numeric(top3["ActualFinalPosition"], errors="coerce") + - pd.to_numeric(top3["PredictedFinalPosition"], errors="coerce") + ) + ) + ) + payload["top3_predictions"] = records(top3.head(100)) + payload["first_30_predictions"] = records(holdout_frame.head(30)) + return payload def current_season() -> dict[str, Any]: schedule = load_race_schedule().copy() @@ -388,6 +503,350 @@ def find_prediction_artifact( } +def _legacy_prediction_rows(race_id: str, year: int | str, race_name: str) -> list[dict[str, Any]]: + """Return the committed Streamlit-style CSV prediction rows when an exact race artifact exists.""" + slugs = { + str(race_id).strip().lower().replace("_", "-").replace(" ", "-"), + str(race_name).lower().replace(" grand prix", "").replace(" ", "-"), + } + candidates: list[Path] = [] + for slug in sorted(slugs): + if slug: + candidates.extend([ + DATA_DIR / f"predictions_{slug}_{year}.csv", + DATA_DIR / f"predictions_{slug.replace('-', '_')}_{year}.csv", + ]) + for candidate in candidates: + if candidate.is_file(): + frame = _read_optional(candidate) + if frame.empty: + continue + + data = load_main_data() + if "resultsDriverName" in frame and {"resultsDriverName", "driverDNFCount", "driverDNFAvg"}.issubset(data.columns): + latest = ( + data.sort_values("grandPrixYear") + .groupby("resultsDriverName", as_index=False) + .tail(1)[["resultsDriverName", "driverDNFCount", "driverDNFAvg"]] + .drop_duplicates("resultsDriverName") + ) + frame = frame.merge(latest, on="resultsDriverName", how="left", suffixes=("", "_latest")) + if "driverDNFCount_latest" in frame: + frame["driverDNFCount"] = frame.get("driverDNFCount").fillna(frame["driverDNFCount_latest"]) if "driverDNFCount" in frame else frame["driverDNFCount_latest"] + if "driverDNFAvg_latest" in frame: + frame["driverDNFAvg"] = frame.get("driverDNFAvg").fillna(frame["driverDNFAvg_latest"]) if "driverDNFAvg" in frame else frame["driverDNFAvg_latest"] + frame["driverDNFPercentage"] = ( + pd.to_numeric(frame.get("driverDNFAvg"), errors="coerce").fillna(0) * 100 + ).round(3) + frame = frame.drop(columns=["driverDNFCount_latest", "driverDNFAvg_latest"], errors="ignore") + if "PredictedDNFProbabilityStd" not in frame: + frame["PredictedDNFProbabilityStd"] = np.nan + + sort_col = "Rank" if "Rank" in frame else ( + "PredictedFinalPosition" if "PredictedFinalPosition" in frame else None + ) + if sort_col: + frame = frame.sort_values(sort_col) + return records(frame) + return [] + + +@lru_cache(maxsize=1) +def _position_mae_by_position() -> dict[int, float]: + """Return the same per-position holdout MAE mapping used by the Streamlit next-race table.""" + try: + historical = precomputed("historical_validation") + except (KeyError, OSError, ValueError): + historical = None + rows = ((historical or {}).get("holdout") or {}).get("rows", []) if isinstance(historical, dict) else [] + if not rows: + return {} + frame = pd.DataFrame(rows) + if not {"ActualFinalPosition", "PredictedFinalPosition"}.issubset(frame.columns): + return {} + frame["ActualFinalPosition"] = pd.to_numeric(frame["ActualFinalPosition"], errors="coerce") + frame["PredictedFinalPosition"] = pd.to_numeric(frame["PredictedFinalPosition"], errors="coerce") + frame = frame.dropna(subset=["ActualFinalPosition", "PredictedFinalPosition"]) + frame["absolute_error"] = (frame["ActualFinalPosition"] - frame["PredictedFinalPosition"]).abs() + grouped = frame.groupby("ActualFinalPosition")["absolute_error"].mean() + return {int(position): float(mae) for position, mae in grouped.items()} + + +@lru_cache(maxsize=1) +def _load_dnf_model() -> Any: + """Load the trusted, workflow-generated DNF inference artifact.""" + path = DATA_DIR / "models" / "dnf_model.pkl" + if not path.is_file(): + return None + with path.open("rb") as handle: + artifact = pickle.load(handle) # noqa: S301 - trusted model artifact committed by this repository + return artifact.get("model") if isinstance(artifact, dict) else artifact + + +@lru_cache(maxsize=1) +def _dnf_feature_names() -> tuple[str, ...]: + """Read the authoritative DNF feature order from the committed manifest.""" + import json + + path = DATA_DIR / "models" / "dnf_manifest.json" + if not path.is_file(): + return () + try: + payload = json.loads(path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError): + return () + return tuple(str(value) for value in payload.get("feature_names", ())) + + +@lru_cache(maxsize=1) +def dnf_diagnostics() -> dict[str, float | None]: + """Return min/max/mean saved-model DNF probabilities over the historical analysis rows.""" + model = _load_dnf_model() + feature_names = _dnf_feature_names() + if model is None or not feature_names: + return {"min": None, "max": None, "mean": None} + frame = load_main_data().copy() + for column in feature_names: + if column not in frame: + frame[column] = np.nan + try: + probabilities = model.predict_proba(frame[list(feature_names)])[:, 1] + except Exception: + return {"min": None, "max": None, "mean": None} + finite = np.asarray(probabilities, dtype=float) + finite = finite[np.isfinite(finite)] + if not finite.size: + return {"min": None, "max": None, "mean": None} + return { + "min": float(finite.min()), + "max": float(finite.max()), + "mean": float(finite.mean()), + } + + +def build_dnf_predictions( + position_predictions: dict[str, Any] | None, + next_race: pd.Series, + race_name: str, + weather: pd.DataFrame, +) -> list[dict[str, Any]]: + """Generate Streamlit-equivalent DNF rows from the committed inference artifact.""" + model = _load_dnf_model() + feature_names = _dnf_feature_names() + if model is None or not feature_names or not isinstance(position_predictions, dict): + return [] + + by_model = position_predictions.get("predictions_by_model") or {} + block = by_model.get("xgboost") or (next(iter(by_model.values()), {}) if by_model else {}) + prediction_rows = block.get("predictions") or [] + if not prediction_rows: + return [] + + data = load_main_data().copy() + if "resultsDriverName" not in data: + return [] + sort_column = "grandPrixYear" if "grandPrixYear" in data else None + if sort_column: + data = data.sort_values(sort_column) + latest = data.groupby("resultsDriverName", as_index=False).tail(1).copy() + latest = latest.set_index("resultsDriverName", drop=False) + + schedule_map = { + "turns": "turns", + "trackRace": "trackRace", + "streetRace": "streetRace", + } + weather_row = weather.iloc[0] if not weather.empty else pd.Series(dtype=object) + rows: list[dict[str, Any]] = [] + feature_rows: list[dict[str, Any]] = [] + for prediction in prediction_rows: + driver = str(prediction.get("driverName", "")) + if not driver or driver not in latest.index: + continue + source = latest.loc[driver] + if isinstance(source, pd.DataFrame): + source = source.iloc[-1] + feature_row = {name: source.get(name, np.nan) for name in feature_names} + feature_row["grandPrixName"] = race_name + if prediction.get("constructor"): + feature_row["constructorName"] = prediction["constructor"] + feature_row["resultsDriverName"] = driver + for target, source_name in schedule_map.items(): + if target in feature_row and source_name in next_race.index and pd.notna(next_race[source_name]): + feature_row[target] = next_race[source_name] + for column in ("average_temp", "average_humidity", "average_wind_speed", "total_precipitation"): + if column in feature_row and column in weather_row.index and pd.notna(weather_row[column]): + feature_row[column] = weather_row[column] + feature_rows.append(feature_row) + rows.append({ + "constructorName": prediction.get("constructor", source.get("constructorName")), + "resultsDriverName": driver, + "driverDNFCount": source.get("driverDNFCount"), + "driverDNFPercentage": ( + round(float(source.get("driverDNFAvg", 0) or 0) * 100, 3) + if pd.notna(source.get("driverDNFAvg")) + else 0.0 + ), + "PredictedDNFProbabilityStd": None, + }) + + if not feature_rows: + return [] + frame = pd.DataFrame(feature_rows, columns=list(feature_names)) + try: + probabilities = model.predict_proba(frame)[:, 1] + except Exception: + return [] + for row, probability in zip(rows, probabilities, strict=True): + row["PredictedDNFProbabilityPercentage"] = round(float(probability) * 100, 3) + rows.sort(key=lambda row: float(row["PredictedDNFProbabilityPercentage"]), reverse=True) + return rows + + +@lru_cache(maxsize=1) +def _load_safety_car_inputs() -> pd.DataFrame: + """Load the same historical safety-car feature frame used by Streamlit.""" + path = DATA_DIR / "f1SafetyCarFeatures.csv" + if not path.is_file(): + return pd.DataFrame() + return pd.read_csv(path, sep="\t", low_memory=False) + + +@lru_cache(maxsize=1) +def _load_safety_car_model() -> Any: + """Load the same trusted safety-car artifact search order used by Streamlit.""" + candidates = [ + DATA_DIR / "models" / "xgboost" / "safetycar_model.pkl", + DATA_DIR / "models" / "lightgbm" / "safetycar_model.pkl", + DATA_DIR / "models" / "catboost" / "safetycar_model.pkl", + DATA_DIR / "models" / "ensemble" / "safetycar_model.pkl", + DATA_DIR / "models" / "safetycar_model.pkl", + ] + for path in candidates: + if not path.is_file(): + continue + try: + with path.open("rb") as handle: + artifact = pickle.load(handle) # noqa: S301 - trusted repository artifact + except (OSError, pickle.UnpicklingError, AttributeError, EOFError, ImportError, ValueError): + continue + if isinstance(artifact, dict): + model = artifact.get("model") + if model is not None: + return model + continue + if hasattr(artifact, "predict_proba"): + return artifact + return None + + +def build_safety_car_predictions( + next_race: pd.Series, + race_name: str, + year: int, + weather: pd.DataFrame, +) -> dict[str, Any]: + """Mirror Streamlit's historical + synthetic next-race safety-car inference.""" + frame = _load_safety_car_inputs().copy() + model = _load_safety_car_model() + if frame.empty or model is None or "SafetyCarStatus" not in frame: + return {"rows": [], "mean": None, "min": None, "max": None} + + manifest_path = DATA_DIR / "models" / "safetycar_manifest.json" + if not manifest_path.is_file(): + return {"rows": [], "mean": None, "min": None, "max": None} + import json + + try: + feature_names = json.loads(manifest_path.read_text(encoding="utf-8")).get("feature_names", []) + except (OSError, json.JSONDecodeError): + feature_names = [] + if not feature_names: + return {"rows": [], "mean": None, "min": None, "max": None} + + for column in feature_names: + if column not in frame: + frame[column] = np.nan + features = frame[feature_names].copy() + try: + probabilities = model.predict_proba(features)[:, 1] + except Exception: + return {"rows": [], "mean": None, "min": None, "max": None} + + history = pd.DataFrame({ + "grandPrixName": frame.get("grandPrixName"), + "grandPrixYear": frame.get("grandPrixYear"), + "PredictedSafetyCarProbabilityPercentage": (probabilities * 100).round(3), + }) + history["Type"] = "Historical" + + synthetic: dict[str, Any] = dict.fromkeys(feature_names, np.nan) + synthetic["grandPrixYear"] = year + synthetic["grandPrixName"] = race_name + schedule_map = { + "circuitId": "circuitId", + "grandPrixLaps": "laps", + "turns": "turns", + "streetRace": "streetRace", + "trackRace": "trackRace", + } + for target, source in schedule_map.items(): + if target in synthetic and source in next_race.index and pd.notna(next_race[source]): + synthetic[target] = next_race[source] + + if not weather.empty: + weather_row = weather.iloc[0] + for column in ("average_temp", "average_humidity", "average_wind_speed", "total_precipitation"): + if column in synthetic and column in weather_row.index and pd.notna(weather_row[column]): + synthetic[column] = weather_row[column] + + same_gp = frame[frame.get("grandPrixName", pd.Series(index=frame.index, dtype=object)) == race_name] + for column in feature_names: + if not pd.isna(synthetic[column]) or column not in frame: + continue + if pd.api.types.is_numeric_dtype(frame[column]): + values = pd.Series(dtype=float) + if not same_gp.empty and "grandPrixYear" in same_gp: + per_race = same_gp.groupby("grandPrixYear")[column].mean(numeric_only=True).dropna() + if not per_race.empty: + values = per_race.sort_index().tail(2) + synthetic[column] = ( + values.median() + if not values.empty + else pd.to_numeric(frame[column], errors="coerce").dropna().median() + ) + + synthetic_frame = pd.DataFrame([synthetic], columns=feature_names) + try: + next_probability = float(model.predict_proba(synthetic_frame)[:, 1][0]) + except Exception: + next_probability = float("nan") + + current = history[ + (history["grandPrixName"].astype(str) == race_name) + & (pd.to_numeric(history["grandPrixYear"], errors="coerce") != year) + ].drop_duplicates(subset=["grandPrixYear"]) + + if np.isfinite(next_probability): + current = pd.concat([ + current, + pd.DataFrame([{ + "grandPrixName": race_name, + "grandPrixYear": year, + "PredictedSafetyCarProbabilityPercentage": round(next_probability * 100, 3), + "Type": "Next Race", + }]), + ], ignore_index=True) + + current = current.sort_values("grandPrixYear", ascending=False) + percentage = history["PredictedSafetyCarProbabilityPercentage"] + return { + "rows": records(current), + "mean": float(percentage.mean()), + "min": float(percentage.min()), + "max": float(percentage.max()), + } + def next_race_bundle() -> dict[str, Any]: schedule = load_race_schedule().copy() date_col = "date" if "date" in schedule else ("short_date" if "short_date" in schedule else None) @@ -451,17 +910,40 @@ def next_race_bundle() -> dict[str, Any]: messages = messages[messages["grandPrixId"].astype(str) == str(race_id)] predictions = find_prediction_artifact(str(race_id), str(year), str(race_name), row[date_col]) + legacy_predictions = _legacy_prediction_rows(str(race_id), int(year), str(race_name)) + dnf_predictions = legacy_predictions or build_dnf_predictions(predictions, row, str(race_name), weather) + safety_car_predictions = build_safety_car_predictions(row, str(race_name), int(year), weather) pit_stops = fastest_pit_stops(str(race_id)) + position_mae_by_position = _position_mae_by_position() + try: + manifest = model_manifest("XGBoost") or {} + except (KeyError, OSError, ValueError): + manifest = {} + model_mae = (manifest.get("metrics") or {}).get("mae") + if isinstance(predictions, dict) and predictions.get("format") == "json": + by_model = predictions.get("predictions_by_model") or {} + xgboost_block = by_model.get("xgboost") or (next(iter(by_model.values()), {}) if by_model else {}) + model_mae = xgboost_block.get("model_mae", model_mae) return { "next_race": records(next_frame)[0], "race_id": None if race_id is None else str(race_id), "race_name": str(race_name), "year": int(year) if pd.notna(year) else None, - "past_results": records(past.drop_duplicates().head(1000)), + "past_results": records( + past.drop_duplicates( + subset=[column for column in ("resultsDriverName", "grandPrixYear") if column in past.columns] + ).head(1000) + ), "driver_performance": records(driver_perf), "constructor_performance": records(constructor_perf), "weather": records(weather.head(500)), "race_messages": records(messages.head(500)), "fastest_pit_stops": pit_stops, "predictions": predictions, + "legacy_predictions": legacy_predictions, + "dnf_predictions": dnf_predictions, + "dnf_diagnostics": dnf_diagnostics(), + "safety_car_predictions": safety_car_predictions, + "model_mae": model_mae, + "position_mae_by_position": position_mae_by_position, } diff --git a/fastapi_react/backend/app/services/data.py b/fastapi_react/backend/app/services/data.py index 00b6587c..d31df622 100644 --- a/fastapi_react/backend/app/services/data.py +++ b/fastapi_react/backend/app/services/data.py @@ -49,6 +49,51 @@ def load_main_data() -> pd.DataFrame: if not MAIN_DATA.exists(): raise FileNotFoundError(f"Missing required dataset: {MAIN_DATA}") df = pd.read_csv(MAIN_DATA, sep="\t", low_memory=False) + + # Streamlit's get_shared_dataset() recreates several canonical driver/constructor + # statistics from legacy merge-suffixed columns before it builds the filter + # sidebar. Mirror those aliases here so the API exposes the same filters and + # query semantics instead of silently dropping them. + streamlit_aliases = { + "bestChampionshipPosition": "bestChampionshipPosition_results_with_qualifying", + "bestStartingGridPosition": "bestStartingGridPosition_results_with_qualifying", + "bestRaceResult": "bestRaceResult_results_with_qualifying", + "totalChampionshipWins": "totalChampionshipWins_results_with_qualifying", + "totalRaceStarts": "totalRaceStarts_results_with_qualifying", + "totalRaceWins": "totalRaceWins_results_with_qualifying", + "totalRaceLaps": "totalRaceLaps_results_with_qualifying", + "totalPodiums": "totalPodiums_results_with_qualifying", + "totalPoints": "totalPoints_results_with_qualifying", + "totalChampionshipPoints": "totalChampionshipPoints_results_with_qualifying", + "totalFastestLaps": "totalFastestLaps_results_with_qualifying", + "totalRaceEntries": "totalRaceEntries_results_with_qualifying", + } + for canonical, legacy in streamlit_aliases.items(): + if canonical not in df.columns and legacy in df.columns: + df[canonical] = df[legacy] + + # The Streamlit dataset then merges the current constructor and driver + # standings before column_names is created. Use key-based maps rather + # than a dataframe merge so the API keeps one row per race/driver while + # exposing the identical current-standings filter fields. + constructor_standings_path = DATA_DIR / "constructor_standings.csv" + if constructor_standings_path.exists() and "constructorId_results" in df.columns: + standings = pd.read_csv(constructor_standings_path, sep="\t", low_memory=False) + if "id" in standings.columns: + keyed = standings.drop_duplicates("id").set_index("id") + for column in standings.columns: + if column in {"id", "name", "fullName", "countryId", "TeamName"}: + continue + df[column] = df["constructorId_results"].map(keyed[column]) + + driver_standings_path = DATA_DIR / "driver_standings.csv" + if driver_standings_path.exists() and "resultsDriverId" in df.columns: + standings = pd.read_csv(driver_standings_path, sep="\t", low_memory=False) + if "driverId" in standings.columns: + keyed = standings.drop_duplicates("driverId").set_index("driverId") + if "driverRank" in keyed.columns: + df["driverRank"] = df["resultsDriverId"].map(keyed["driverRank"]) + for candidate in ("short_date", "date", "grandPrixDate"): if candidate in df.columns: df[candidate] = pd.to_datetime(df[candidate], errors="coerce") @@ -104,13 +149,15 @@ def apply_filters(df: pd.DataFrame, filters: list[Any]) -> pd.DataFrame: @lru_cache(maxsize=1) -def streamlit_filter_rules() -> tuple[dict[str, str], frozenset[str]]: - """Read the Streamlit filter labels and exclusions without importing its app.""" +def streamlit_filter_rules() -> tuple[dict[str, str], frozenset[str], frozenset[str]]: + """Read Streamlit filter labels, exclusions, and its loaded-column contract.""" source_path = REPO_ROOT / "raceAnalysis.py" tree = ast.parse(source_path.read_text(encoding="utf-8"), filename=str(source_path)) - literal_names = {"column_rename_for_filter", "exclusionList", "suffixes_to_exclude"} + literal_names = {"column_rename_for_filter", "exclusionList", "suffixes_to_exclude", "selected_columns"} values: dict[str, Any] = {} - for node in tree.body: + # selected_columns lives inside load_data(); walk the tree so the API + # follows the same authoritative loaded-column contract as Streamlit. + for node in ast.walk(tree): if not isinstance(node, ast.Assign): continue for target in node.targets: @@ -124,15 +171,20 @@ def streamlit_filter_rules() -> tuple[dict[str, str], frozenset[str]]: excluded = set(values.get("exclusionList", ())) suffixes = values.get("suffixes_to_exclude", ()) excluded.update(column for column in load_main_data().columns if column.endswith(tuple(suffixes))) - return labels, frozenset(excluded) - + selected = set(values.get("selected_columns", ())) + # These friendly-name fields include columns introduced by the standings + # merges (for example Points and bestChampionshipPosition). + selected.update(labels) + return labels, frozenset(excluded), frozenset(selected) def filter_schema() -> list[dict[str, Any]]: df = load_main_data() - labels, excluded = streamlit_filter_rules() + labels, excluded, selected = streamlit_filter_rules() schema: list[dict[str, Any]] = [] + # Streamlit explicitly sorts `column_names` before building sidebar controls. + # Preserve that raw-field alphabetical order; labels are applied only for display. for column in sorted(df.columns): - if column in excluded: + if column in excluded or (selected and column not in selected): continue series = df[column] non_null = series.dropna() @@ -168,7 +220,13 @@ def query_main(request: Any) -> dict[str, Any]: if request.sort: valid = [c for c in request.sort if c in df.columns] if valid: - df = df.sort_values(valid, ascending=not request.descending) + sort_ascending: bool | list[bool] + if request.ascending and len(request.ascending) == len(request.sort): + direction_map = dict(zip(request.sort, request.ascending, strict=True)) + sort_ascending = [direction_map[column] for column in valid] + else: + sort_ascending = not request.descending + df = df.sort_values(valid, ascending=sort_ascending) if request.columns: valid = [c for c in request.columns if c in df.columns] if valid: diff --git a/fastapi_react/backend/test_api.py b/fastapi_react/backend/test_api.py index b14ea211..da9cef69 100644 --- a/fastapi_react/backend/test_api.py +++ b/fastapi_react/backend/test_api.py @@ -6,6 +6,7 @@ """ from __future__ import annotations +import numpy as np import pandas as pd import pytest from fastapi.testclient import TestClient @@ -58,6 +59,28 @@ def test_data_explorer_schema_returns_filters() -> None: assert body["filters"], "schema should return at least one filterable column" +def test_data_explorer_schema_matches_streamlit_standings_filters() -> None: + response = client.get("/api/data-explorer/schema") + assert response.status_code == 200 + filters = {item["column"]: item for item in response.json()["filters"]} + + expected_labels = { + "Points": "Current Year Points (Driver)", + "bestChampionshipPosition": "Best Champ Pos.", + "bestRaceResult": "Best Race Result", + "bestStartingGridPosition": "Best Starting Grid Pos.", + "constructorRank": "Constructor Rank", + "driverRank": "Driver Rank", + } + for column, label in expected_labels.items(): + assert column in filters + assert filters[column]["label"] == label + + assert filters["Points"]["kind"] == "range" + assert filters["bestChampionshipPosition"]["kind"] == "range" + assert filters["constructorRank"]["kind"] == "range" + + def test_data_explorer_query_unfiltered() -> None: response = client.post("/api/data-explorer/query", json={"limit": 5}) assert response.status_code == 200 @@ -165,6 +188,53 @@ def test_next_race_endpoint() -> None: assert "fastest_pit_stops" in body +def test_safety_car_loader_matches_streamlit_search_order( + monkeypatch: pytest.MonkeyPatch, + tmp_path, +) -> None: + import pickle + + models = tmp_path / "models" + xgboost = models / "xgboost" + xgboost.mkdir(parents=True) + with (xgboost / "safetycar_model.pkl").open("wb") as handle: + pickle.dump({"model": "wrapped-safety-model"}, handle) + + monkeypatch.setattr(analysis, "DATA_DIR", tmp_path) + analysis._load_safety_car_model.cache_clear() + try: + assert analysis._load_safety_car_model() == "wrapped-safety-model" + finally: + analysis._load_safety_car_model.cache_clear() + + +def test_safety_car_predictions_include_historical_and_next_race(monkeypatch: pytest.MonkeyPatch) -> None: + class FakeSafetyModel: + def predict_proba(self, frame: pd.DataFrame) -> np.ndarray: + return np.tile(np.array([[0.25, 0.75]]), (len(frame), 1)) + + history = pd.DataFrame({ + "grandPrixName": ["Singapore Grand Prix", "Singapore Grand Prix"], + "grandPrixYear": [2024, 2025], + "turns": [19, 19], + "SafetyCarStatus": [1, 0], + }) + monkeypatch.setattr(analysis, "_load_safety_car_inputs", lambda: history) + monkeypatch.setattr(analysis, "_load_safety_car_model", lambda: FakeSafetyModel()) + + payload = analysis.build_safety_car_predictions( + pd.Series({"turns": 19}), + "Singapore Grand Prix", + 2026, + pd.DataFrame(), + ) + + assert payload["mean"] == pytest.approx(75.0) + assert payload["rows"][0]["grandPrixYear"] == 2026 + assert payload["rows"][0]["Type"] == "Next Race" + assert len(payload["rows"]) == 3 + + def test_tire_strategy_endpoint_returns_selected_race_and_year_summary() -> None: response = client.get("/api/analytics/tire-strategy") assert response.status_code == 200 diff --git a/fastapi_react/frontend/src/App.jsx b/fastapi_react/frontend/src/App.jsx index abe00fff..adffd3c7 100644 --- a/fastapi_react/frontend/src/App.jsx +++ b/fastapi_react/frontend/src/App.jsx @@ -1,4 +1,4 @@ -import { useEffect, useState } from 'react' +import { useEffect, useRef, useState } from "react"; import { api } from "./api"; import DataExplorer from "./pages/DataExplorer"; import Analytics from "./pages/Analytics"; @@ -7,99 +7,102 @@ import NextRace from "./pages/NextRace"; import Models from "./pages/Models"; import RawData from "./pages/RawData"; import BettingResearch from "./pages/BettingResearch"; +import FilterSidebar from "./components/FilterSidebar"; -const pages = { - "Data Explorer": DataExplorer, - "Analytics": Analytics, - "Current Season": CurrentSeason, - "Next Race": NextRace, - "Predictive Models": Models, - "Raw Data": RawData, - "Betting Research": BettingResearch, -}; +const pages = [ + { key: "Data Explorer", label: "📊 Data Explorer", Component: DataExplorer }, + { key: "Analytics", label: "📈 Analytics & Visualizations", Component: Analytics }, + { key: "Current Season", label: "🏎️ Schedule", Component: CurrentSeason }, + { key: "Next Race", label: "🏁 Next Race", Component: NextRace }, + { key: "Predictive Models", label: "🤖 Predictive Models", Component: Models }, + { key: "Raw Data", label: "💾 Data & Debug", Component: RawData }, + { key: "Betting Research", label: "📐 Betting Research", Component: BettingResearch }, +]; -const icons = { - "Data Explorer": "\u25A6", - "Analytics": "\u2301", - "Current Season": "\u25F7", - "Next Race": "\uD83C\uDFC1", - "Predictive Models": "\u25C6", - "Raw Data": "\u2261", - "Betting Research": "\uD83D\uDCD0", -}; - -const BASE_TITLE = "F1 Analysis"; +const BASE_TITLE = "Gridlocked - Formula 1 Betting & Analytics"; export default function App() { const [active, setActive] = useState("Data Explorer"); - const [health, setHealth] = useState(null); - const [theme, setTheme] = useState(() => { - try { return localStorage.getItem("f1analysis.theme") === "light" ? "light" : "dark"; } - catch { return "dark"; } + const [meta, setMeta] = useState(null); + const [filterRevision, setFilterRevision] = useState(0); + const tabStripRef = useRef(null); + const [filtersActive, setFiltersActive] = useState(() => { + try { return Boolean(JSON.parse(sessionStorage.getItem("f1analysis.filters") || "null")?.applied); } + catch { return false; } }); useEffect(() => { - api.get("/api/health").then(setHealth).catch(() => {}); + api.get("/api/meta").then(setMeta).catch(() => {}); const hash = decodeURIComponent(location.hash.replace("#/", "")); - if (pages[hash]) setActive(hash); - }, []); + if (pages.some(page => page.key === hash)) setActive(hash); - useEffect(() => { - document.title = `${active} \u2014 ${BASE_TITLE}`; - }, [active]); + const syncFilters = () => { + try { setFiltersActive(Boolean(JSON.parse(sessionStorage.getItem("f1analysis.filters") || "null")?.applied)); } + catch { setFiltersActive(false); } + setFilterRevision(value => value + 1); + }; + window.addEventListener("f1analysis:filters-changed", syncFilters); + return () => window.removeEventListener("f1analysis:filters-changed", syncFilters); + }, []); useEffect(() => { - document.documentElement.dataset.theme = theme; - try { localStorage.setItem("f1analysis.theme", theme); } catch { /* storage is optional */ } - }, [theme]); + document.title = BASE_TITLE; + const activeTab = tabStripRef.current?.querySelector('[aria-selected="true"]'); + activeTab?.scrollIntoView?.({ block: "nearest", inline: "nearest" }); + }, [active, filtersActive]); function navigate(page) { setActive(page); location.hash = `/${encodeURIComponent(page)}`; - window.scrollTo({ top: 0, behavior: "smooth" }); + window.scrollTo({ top: 0, behavior: "auto" }); } - const Page = pages[active]; + const Page = pages.find(page => page.key === active)?.Component || DataExplorer; + const startYear = meta?.race_start_year ?? 2016; + const currentYear = meta?.current_year ?? new Date().getFullYear(); + return ( -
+
Skip to main content -
-
- Gridlocked -
F1 Races from 2016 to {new Date().getFullYear()}
-
-
-
- -
- -
- Powered by - Betting Oracle - Sports Prediction Analytics - All content is for informational purposes only and does not constitute betting advice. Wager responsibly. -
); diff --git a/fastapi_react/frontend/src/App.test.jsx b/fastapi_react/frontend/src/App.test.jsx index 2a8586f3..a43488f1 100644 --- a/fastapi_react/frontend/src/App.test.jsx +++ b/fastapi_react/frontend/src/App.test.jsx @@ -3,7 +3,12 @@ import { beforeEach, describe, expect, it, vi } from "vitest"; import App from "./App"; vi.mock("./api", () => ({ - api: { get: vi.fn().mockResolvedValue({ status: "ok", rss_mb: 120 }) }, + api: { get: vi.fn().mockResolvedValue({ + race_start_year: 2016, + current_year: 2026, + last_updated: "2026-09-29 09:00 PM", + code_deployed_at: "2026-09-30 01:00:00 UTC", + }) }, })); vi.mock("./pages/DataExplorer", () => ({ default: () =>

Explorer page

})); vi.mock("./pages/Analytics", () => ({ default: () =>

Analytics page

})); @@ -20,26 +25,21 @@ describe("application shell", () => { document.title = ""; }); - it("renders the reference brand, section navigation, and API status", async () => { + it("renders the Streamlit reference brand, title, captions, and seven tabs", async () => { render(); expect(screen.getByRole("img", { name: "Gridlocked" })).toHaveAttribute("src", "/api/brand/logo"); - expect(screen.getByText(/F1 Races from 2016 to/)).toBeInTheDocument(); - expect(screen.getByRole("navigation", { name: "Sections" })).toBeInTheDocument(); - expect(await screen.findByText("API connected")).toBeInTheDocument(); + expect(await screen.findByText("F1 Races from 2016 to 2026")).toBeInTheDocument(); + expect(screen.getByRole("navigation", { name: "Main sections" })).toBeInTheDocument(); + expect(screen.getAllByRole("tab")).toHaveLength(7); + expect(screen.getByRole("tab", { name: "📊 Data Explorer" })).toBeInTheDocument(); + expect(screen.getByRole("tab", { name: "💾 Data & Debug" })).toBeInTheDocument(); }); - it("switches sections from the horizontal navigation and updates the title", async () => { + it("switches sections from the Streamlit-style tab row and preserves the Streamlit page title", async () => { render(); - fireEvent.click(screen.getByRole("button", { name: "Analytics" })); + fireEvent.click(screen.getByRole("tab", { name: "📈 Analytics & Visualizations" })); expect(await screen.findByRole("heading", { name: "Analytics page" })).toBeInTheDocument(); - await waitFor(() => expect(document.title).toBe("Analytics — F1 Analysis")); + await waitFor(() => expect(document.title).toBe("Gridlocked - Formula 1 Betting & Analytics")); expect(window.location.hash).toBe("#/Analytics"); }); - - it("persists the light theme toggle", async () => { - render(); - fireEvent.click(screen.getByRole("checkbox", { name: "Use light theme" })); - await waitFor(() => expect(document.documentElement).toHaveAttribute("data-theme", "light")); - expect(localStorage.getItem("f1analysis.theme")).toBe("light"); - }); -}); \ No newline at end of file +}); diff --git a/fastapi_react/frontend/src/components/Charts.jsx b/fastapi_react/frontend/src/components/Charts.jsx index c2e19ab0..af69cadd 100644 --- a/fastapi_react/frontend/src/components/Charts.jsx +++ b/fastapi_react/frontend/src/components/Charts.jsx @@ -1,6 +1,6 @@ import { ResponsiveContainer, ScatterChart, Scatter, XAxis, YAxis, CartesianGrid, Tooltip, - LineChart, Line, BarChart, Bar, Legend, ComposedChart + LineChart, Line, BarChart, Bar, Legend, ComposedChart, PieChart, Pie, Cell } from "recharts"; import { Card } from "./UI"; @@ -39,7 +39,7 @@ export function ScatterPanel({ title, rows = [], x, y, xLabel = axisLabels[x] || - +
@@ -59,7 +59,7 @@ export function LinePanel({ title, rows = [], x, y, xLabel = axisLabels[x] || x, - + @@ -80,7 +80,28 @@ export function BarPanel({ title, rows = [], x, y, xLabel = axisLabels[x] || x, - + + + + + + ); +} + +/** @param {{ title: string, rows?: Array>, x: string, ys: string[], xLabel?: string, yLabel?: string }} props */ +export function MultiBarPanel({ title, rows = [], x, ys, xLabel = axisLabels[x] || x, yLabel = "Count" }) { + if (!rows.length) return null; + return ( + +
+ + + + + + + + {ys.map((key, index) => )}
@@ -108,3 +129,62 @@ export function RegressionPanel({ title, points = [], fit = [], x, y, xLabel, yL
); } + + +const SERIES_COLORS = [ + "#0068c9", "#ff4b4b", "#00a86b", "#7d3cff", "#f0a202", "#2a9d8f", + "#e76f51", "#264653", "#8d99ae", "#9b5de5", "#00bbf9", "#f15bb5", +]; + +/** @param {{ title: string, rows?: Array>, x: string, y: string, series: string, xLabel?: string, yLabel?: string }} props */ +export function MultiLinePanel({ title, rows = [], x, y, series, xLabel = axisLabels[x] || x, yLabel = axisLabels[y] || y }) { + if (!rows.length) return null; + const seriesNames = [...new Set(rows.map(row => String(row[series] ?? "")).filter(Boolean))]; + const xValues = [...new Set(rows.map(row => row[x]))]; + const byX = xValues.map(xValue => { + /** @type {Record} */ + const point = { [x]: xValue }; + for (const row of rows.filter(item => item[x] === xValue)) { + point[String(row[series])] = row[y]; + } + return point; + }); + return ( + +
+ + + + + + + + {seriesNames.map((name, index) => ( + + ))} + + +
+
+ ); +} + +/** @param {{ title: string, rows?: Array>, nameKey: string, valueKey: string }} props */ +export function PiePanel({ title, rows = [], nameKey, valueKey }) { + if (!rows.length) return null; + return ( + +
+ + + + {rows.map((row, index) => )} + + + + + +
+
+ ); +} diff --git a/fastapi_react/frontend/src/components/Charts.test.jsx b/fastapi_react/frontend/src/components/Charts.test.jsx index 0df30d9f..4d8641fe 100644 --- a/fastapi_react/frontend/src/components/Charts.test.jsx +++ b/fastapi_react/frontend/src/components/Charts.test.jsx @@ -1,49 +1,54 @@ import { describe, it, expect, beforeEach } from 'vitest'; import { render } from '@testing-library/react'; -import { ScatterPanel, LinePanel, BarPanel } from './Charts.jsx'; +import { + ScatterPanel, LinePanel, BarPanel, MultiBarPanel, RegressionPanel, MultiLinePanel, PiePanel, +} from './Charts.jsx'; beforeEach(() => { - // ResponsiveContainer measures its parent; jsdom gives it 0x0 by default - // which makes Recharts charts render as empty SVGs in tests. We mock - // getBoundingClientRect on the container's parent so charts have a size. Object.defineProperty(HTMLElement.prototype, 'getBoundingClientRect', { configurable: true, value: () => ({ width: 800, height: 400, top: 0, left: 0, right: 800, bottom: 400, x: 0, y: 0 }), }); }); -describe('ScatterPanel', () => { - it('returns null for empty rows', () => { - const { container } = render(); +describe('chart panels', () => { + it.each([ + ['scatter', ScatterPanel, { title: 'Scatter', rows: [], x: 'a', y: 'b' }], + ['line', LinePanel, { title: 'Line', rows: [], x: 'a', y: 'b' }], + ['bar', BarPanel, { title: 'Bar', rows: [], x: 'a', y: 'b' }], + ['multi bar', MultiBarPanel, { title: 'Multi Bar', rows: [], x: 'a', ys: ['b', 'c'] }], + ['regression', RegressionPanel, { title: 'Regression', points: [], fit: [], x: 'a', y: 'b', xLabel: 'A', yLabel: 'B' }], + ['multi line', MultiLinePanel, { title: 'Multi Line', rows: [], x: 'a', y: 'b', series: 'series' }], + ['pie', PiePanel, { title: 'Pie', rows: [], nameKey: 'name', valueKey: 'value' }], + ])('returns null for empty %s data', (_name, Component, props) => { + const { container } = render(); expect(container).toBeEmptyDOMElement(); }); - it('renders a card with title for non-empty rows', () => { - render(); - expect(document.querySelector('.card')).toBeInTheDocument(); - }); -}); + it('renders all supported chart types with accessible chart containers', () => { + const { rerender } = render(); + expect(document.querySelector('[role="img"]')).toHaveAttribute('aria-label', expect.stringContaining('Scatter')); -describe('LinePanel', () => { - it('returns null for empty rows', () => { - const { container } = render(); - expect(container).toBeEmptyDOMElement(); - }); + rerender(); + expect(document.querySelector('.card')).toBeInTheDocument(); - it('renders a chart container for non-empty rows', () => { - render(); + rerender(); expect(document.querySelector('.card')).toBeInTheDocument(); - }); -}); -describe('BarPanel', () => { - it('returns null for empty rows', () => { - const { container } = render(); - expect(container).toBeEmptyDOMElement(); - }); + rerender(); + expect(document.querySelector('[aria-label*="2 series"]')).toBeInTheDocument(); - it('renders a chart container for non-empty rows', () => { - render(); - expect(document.querySelector('.card')).toBeInTheDocument(); + rerender(); + expect(document.querySelector('[aria-label*="Regression fit"]')).toBeInTheDocument(); + + rerender(); + expect(document.querySelector('[aria-label*="2 series"]')).toBeInTheDocument(); + + rerender(); + expect(document.querySelector('[aria-label*="2 categories"]')).toBeInTheDocument(); }); }); diff --git a/fastapi_react/frontend/src/components/FilterSidebar.jsx b/fastapi_react/frontend/src/components/FilterSidebar.jsx new file mode 100644 index 00000000..0078218d --- /dev/null +++ b/fastapi_react/frontend/src/components/FilterSidebar.jsx @@ -0,0 +1,138 @@ +import { useEffect, useMemo, useState } from "react"; +import { api } from "../api"; + +function FilterControl({ spec, value, onChange }) { + if (spec.kind === "boolean") { + return ( + + ); + } + + if (spec.kind === "exact" && spec.options) { + return ( + + ); + } + + if (spec.kind === "range") { + const low = Number(Array.isArray(value) ? value[0] : spec.min); + const high = Number(Array.isArray(value) ? value[1] : spec.max); + const min = Math.trunc(Number(spec.min)); + const max = Math.trunc(Number(spec.max)); + const currentLow = Math.trunc(Number.isFinite(low) ? low : min); + const currentHigh = Math.trunc(Number.isFinite(high) ? high : max); + return ( +