-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathexport_understat_csv.py
More file actions
52 lines (39 loc) · 1.68 KB
/
Copy pathexport_understat_csv.py
File metadata and controls
52 lines (39 loc) · 1.68 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
"""Convert a JSON Dataset export into a flat CSV with every top-level field."""
from __future__ import annotations
import argparse
import csv
import json
from pathlib import Path
from typing import Any
PRIORITY_FIELDS = ["recordType", "sourceUrl", "fetchedAt"]
def fieldnames_for(rows: list[dict[str, Any]]) -> list[str]:
"""Return provenance fields first, followed by the sorted union of other fields."""
available = {key for row in rows for key in row}
first = [field for field in PRIORITY_FIELDS if field in available]
return first + sorted(available.difference(first))
def export_csv(input_path: Path, output_path: Path) -> int:
rows = json.loads(input_path.read_text(encoding="utf-8"))
if not isinstance(rows, list) or not all(isinstance(row, dict) for row in rows):
raise ValueError("Expected a JSON array of Dataset row objects.")
if not rows:
raise ValueError("The JSON Dataset export is empty.")
fields = fieldnames_for(rows)
with output_path.open("w", encoding="utf-8", newline="") as handle:
writer = csv.DictWriter(handle, fieldnames=fields, extrasaction="ignore")
writer.writeheader()
writer.writerows(rows)
return len(rows)
def main() -> None:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("input", type=Path, nargs="?", default=Path("data/sample-output.json"))
parser.add_argument(
"output",
type=Path,
nargs="?",
default=Path("data/exported-understat-data.csv"),
)
args = parser.parse_args()
count = export_csv(args.input, args.output)
print(f"Wrote {count} rows to {args.output}")
if __name__ == "__main__":
main()