diff --git a/task-1/.env.example b/task-1/.env.example index c76c0be..5d3b69f 100644 --- a/task-1/.env.example +++ b/task-1/.env.example @@ -1,2 +1,2 @@ INPUT_PATH=data/messy_sales.csv -OUTPUT_PATH=output/clean_sales.csv +OUTPUT_PATH=output/clean_sales.csv \ No newline at end of file diff --git a/task-1/.pytest_cache/.gitignore b/task-1/.pytest_cache/.gitignore new file mode 100644 index 0000000..6107291 --- /dev/null +++ b/task-1/.pytest_cache/.gitignore @@ -0,0 +1,4 @@ +**/__pycache__/ +*.pyc +*.pyo +*.pyd \ No newline at end of file diff --git a/task-1/output/clean_sales.csv b/task-1/output/clean_sales.csv new file mode 100644 index 0000000..be91924 --- /dev/null +++ b/task-1/output/clean_sales.csv @@ -0,0 +1,13 @@ +transaction_id,product_name,category,price,quantity,customer_email,date,revenue,vat +1,Laptop Pro,Electronics,999.99,2,alice@example.com,2024-03-15,1999.98,420.0 +2,Wireless Mouse,Electronics,29.99,5,bob@company.com,2024-03-15,149.95,31.49 +3,Usb Cable,Electronics,4.99,10,,2024-03-16,49.9,10.48 +4,Office Chair,Furniture,349.5,1,charlie@work.org,2024-03-16,349.5,73.39 +5,Standing Desk,Furniture,599.0,1,charlie@work.org,not_a_date,599.0,125.79 +9,Webcam Hd,Electronics,54.99,1,,2024-03-18,54.99,11.55 +10,Desk Lamp,Furniture,34.99,4,grace@university.edu,2024-03-19,139.96,29.39 +11,Noise Cancelling Headphones,Electronics,199.99,1,alice@example.com,2024-03-19,199.99,42.0 +12,Cable Management Kit,Furniture,15.99,6,henry@business.com,2024-03-20,95.94,20.15 +13,Ergonomic Mouse Pad,Furniture,24.99,3,ivan@email.com,2024-03-20,74.97,15.74 +14,Laptop Stand,Furniture,45.99,2,jenny@work.org,2024-03-21,91.98,19.32 +15,Bluetooth Speaker,Unknown,39.99,1,karl@startup.io,2024-03-21,39.99,8.4 diff --git a/task-1/src/__pycache__/__init__.cpython-311.pyc b/task-1/src/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000..4db373b Binary files /dev/null and b/task-1/src/__pycache__/__init__.cpython-311.pyc differ diff --git a/task-1/src/__pycache__/config.cpython-311.pyc b/task-1/src/__pycache__/config.cpython-311.pyc new file mode 100644 index 0000000..025ff89 Binary files /dev/null and b/task-1/src/__pycache__/config.cpython-311.pyc differ diff --git a/task-1/src/__pycache__/models.cpython-311.pyc b/task-1/src/__pycache__/models.cpython-311.pyc new file mode 100644 index 0000000..e8884b0 Binary files /dev/null and b/task-1/src/__pycache__/models.cpython-311.pyc differ diff --git a/task-1/src/__pycache__/pipeline.cpython-311.pyc b/task-1/src/__pycache__/pipeline.cpython-311.pyc new file mode 100644 index 0000000..fbc21c4 Binary files /dev/null and b/task-1/src/__pycache__/pipeline.cpython-311.pyc differ diff --git a/task-1/src/__pycache__/transforms.cpython-311.pyc b/task-1/src/__pycache__/transforms.cpython-311.pyc new file mode 100644 index 0000000..233a257 Binary files /dev/null and b/task-1/src/__pycache__/transforms.cpython-311.pyc differ diff --git a/task-1/src/config.py b/task-1/src/config.py index e59f7c1..ea5d9ae 100644 --- a/task-1/src/config.py +++ b/task-1/src/config.py @@ -8,8 +8,8 @@ 2. Read INPUT_PATH and OUTPUT_PATH from os.environ. 3. Raise ValueError if either is missing — do NOT let None silently propagate. """ -import os +import os from dotenv import load_dotenv @@ -21,9 +21,10 @@ def _required(name: str) -> str: """Read an env var; fail loudly if missing.""" - # TODO 2: Read os.environ[name]; if not set, raise ValueError with a - # message that names the missing variable AND points at .env.example. - raise NotImplementedError("Implement _required: see TODO 2 in config.py") + value = os.environ.get(name) + if not value: + raise ValueError(f"Missing required environment variable: {name}") + return value # TODO 3: Replace the placeholder lines below by calling _required(...) for @@ -31,5 +32,5 @@ def _required(name: str) -> str: # module by the rest of the pipeline as a relative import # (`from .config import INPUT_PATH, ...`), since the pipeline runs as # `python -m src.pipeline`. -INPUT_PATH: str = "" # TODO: _required("INPUT_PATH") -OUTPUT_PATH: str = "" # TODO: _required("OUTPUT_PATH") +INPUT_PATH: str = _required("INPUT_PATH") +OUTPUT_PATH: str = _required("OUTPUT_PATH") diff --git a/task-1/src/models.py b/task-1/src/models.py index 865b2b1..83a5ac5 100644 --- a/task-1/src/models.py +++ b/task-1/src/models.py @@ -8,6 +8,7 @@ boundary, so the dataclass is the schema-of-record for everything that gets written to the output CSV. """ + from dataclasses import dataclass @@ -21,13 +22,26 @@ # date: str # revenue: float = 0.0 # vat: float = 0.0 -# -# TODO 2: Add __post_init__ that raises ValueError when: -# - self.price < 0 (with a message naming the bad value) -# - not self.product_name.strip() (empty / whitespace-only product name) -# Replace this stub with your dataclass: @dataclass class Transaction: - transaction_id: int # TODO: replace this stub with the full field list above + transaction_id: int + product_name: str + category: str + price: float + quantity: int + customer_email: str + date: str + revenue: float = 0.0 + vat: float = 0.0 + + # + # TODO 2: Add __post_init__ that raises ValueError when: + # - self.price < 0 (with a message naming the bad value) + # - not self.product_name.strip() (empty / whitespace-only product name) + def __post_init__(self): + if self.price < 0: + raise ValueError(f"Price must be non-negative, got {self.price}") + if not self.product_name.strip(): + raise ValueError("Product name cannot be empty or whitespace") diff --git a/task-1/src/pipeline.py b/task-1/src/pipeline.py index adce136..70f179a 100644 --- a/task-1/src/pipeline.py +++ b/task-1/src/pipeline.py @@ -15,6 +15,7 @@ Run from the task-1/ directory: python -m src.pipeline """ + import csv from dataclasses import asdict from pathlib import Path @@ -32,13 +33,18 @@ def read_csv(path: str) -> list[dict]: """Read a CSV file into a list of dicts. I/O only — no business rules.""" # TODO: implement using csv.DictReader. - raise NotImplementedError + with open(path, "r", newline="", encoding="utf-8") as f: + reader = csv.DictReader(f) + return list(reader) def write_csv(rows: list[dict], path: str) -> None: """Write a list of dicts to CSV. I/O only — no business rules.""" # TODO: implement using csv.DictWriter. - raise NotImplementedError + with open(path, "w", newline="", encoding="utf-8") as f: + writer = csv.DictWriter(f, fieldnames=rows[0].keys() if rows else []) + writer.writeheader() + writer.writerows(rows) def run() -> None: @@ -48,13 +54,17 @@ def run() -> None: data = filter_zero_quantity(data) data = calculate_revenue(data) + for row in data: + row["price"] = float(row["price"]) + row["quantity"] = int(row["quantity"]) + row["revenue"] = float(row["revenue"]) + row["vat"] = float(row["vat"]) # Materialise as Transaction instances so the dataclass __post_init__ # acts as a final guard before serialisation. # TODO: cast price / quantity / revenue / vat to the right types here # if your transforms left them as strings, then iterate over `data` # to build Transaction(**row) for each cleaned row. transactions = [Transaction(**row) for row in data] - # Output dir must exist. Use pathlib for cross-platform safety. Path(OUTPUT_PATH).parent.mkdir(parents=True, exist_ok=True) write_csv([asdict(t) for t in transactions], OUTPUT_PATH) diff --git a/task-1/src/transforms.py b/task-1/src/transforms.py index e6bdd76..b389baa 100644 --- a/task-1/src/transforms.py +++ b/task-1/src/transforms.py @@ -18,8 +18,13 @@ def remove_invalid(rows: list[dict]) -> list[dict]: Empty here means missing, "", or whitespace-only. """ - # TODO: implement. Return a new list, do not mutate `rows`. - raise NotImplementedError + return [ + row + for row in rows + if (name := row.get("product_name")) + and name.strip() + and float(row.get("price", 0)) >= 0 + ] def clean_fields(rows: list[dict]) -> list[dict]: @@ -31,8 +36,17 @@ def clean_fields(rows: list[dict]) -> list[dict]: Return a new list. Do not mutate the input rows. """ - # TODO: implement. - raise NotImplementedError + + cleaned = [] + for row in rows: + new_row = { + **row, + "product_name": row.get("product_name", "").strip().title(), + "customer_email": row.get("customer_email", "").strip().lower(), + "category": row.get("category", "").strip() or "Unknown", + } + cleaned.append(new_row) + return cleaned def calculate_revenue(rows: list[dict], vat_rate: float = 0.21) -> list[dict]: @@ -40,11 +54,18 @@ def calculate_revenue(rows: list[dict], vat_rate: float = 0.21) -> list[dict]: Round both to 2 decimal places. Coerce price/quantity from string if needed. """ - # TODO: implement. - raise NotImplementedError + calculated = [] + for row in rows: + price = float(row.get("price", 0)) + quantity = int(row.get("quantity", 0)) + revenue = round(price * quantity, 2) + vat = round(revenue * vat_rate, 2) + calculated_row = {**row, "revenue": revenue, "vat": vat} + calculated.append(calculated_row) + return calculated def filter_zero_quantity(rows: list[dict]) -> list[dict]: """Remove rows where quantity is 0.""" # TODO: implement. - raise NotImplementedError + return [row for row in rows if int(row.get("quantity", 0)) != 0] diff --git a/task-1/tests/test_transforms.py b/task-1/tests/test_transforms.py index 8d6bd80..57b49df 100644 --- a/task-1/tests/test_transforms.py +++ b/task-1/tests/test_transforms.py @@ -6,6 +6,7 @@ - test_calculate_revenue_adds_fields - test_no_mutation """ + import pytest from src.transforms import ( @@ -17,24 +18,44 @@ def test_remove_invalid_drops_empty_names(): - # TODO: feed in 3 rows (one valid, one empty product_name, one - # whitespace-only product_name); assert only the valid one survives. - raise NotImplementedError + data = [ + {"product_name": "Laptop", "price": 999.99}, + {"product_name": "", "price": 50.0}, + {"product_name": " ", "price": 25.0}, + ] + result = remove_invalid(data) + assert len(result) == 1 + assert result[0]["product_name"] == "Laptop" def test_clean_fields_normalizes_names(): - # TODO: feed a row with messy product_name and uppercase email; assert - # the output has stripped + title-cased name and lowercase email. - raise NotImplementedError + data = [ + {"product_name": " laptop "}, + {"product_name": "PHONE"}, + {"product_name": " "}, + ] + assert clean_fields(data)[0]["product_name"] == "Laptop" + assert clean_fields(data)[1]["product_name"] == "Phone" + assert clean_fields(data)[2]["product_name"] == "" def test_calculate_revenue_adds_fields(): - # TODO: feed a row with price=100, quantity=3; assert output has - # revenue=300.0 and vat=63.0 (default VAT rate is 0.21). - raise NotImplementedError + data = [ + {"product_name": "Laptop", "price": 999.99, "quantity": 3}, + ] + result = calculate_revenue(data) + assert len(result) == 1 + assert result[0]["revenue"] == 2999.97 + assert result[0]["vat"] == 629.99 def test_no_mutation(): - # TODO: feed in a list, run any transform on it, assert the original - # list is unchanged. This is the most important test in the file. - raise NotImplementedError + data = [ + {"product_name": "Laptop", "price": 999.99, "quantity": 3}, + ] + original = data.copy() + remove_invalid(data) + clean_fields(data) + calculate_revenue(data) + filter_zero_quantity(data) + assert data == original diff --git a/task-2/AI_DEBUG.md b/task-2/AI_DEBUG.md index 83cad22..b947080 100644 --- a/task-2/AI_DEBUG.md +++ b/task-2/AI_DEBUG.md @@ -11,7 +11,17 @@ Aim for 100-200 words per section. Bullet points are fine. What went wrong? Paste the traceback or the wrong-output sample. Include the file and the line you were running when it broke. ``` -(paste here) +Traceback (most recent call last): + File "", line 198, in _run_module_as_main + File "", line 88, in _run_code + File "C:\Users\Bader\Desktop\da\c55-data-week-2\task-1\src\pipeline.py", line 25, in + from .transforms import ( + File "C:\Users\Bader\Desktop\da\c55-data-week-2\task-1\src\transforms.py", line 76, in + raw_rows = _required(INPUT_PATH) + ^^^^^^^^^^^^^^^^^^^^^ + File "C:\Users\Bader\Desktop\da\c55-data-week-2\task-1\src\config.py", line 28, in _required + raise ValueError(f"Missing required environment variable: {name}") +ValueError: Missing required environment variable: data/messy_sales.csv ``` ## The Prompt @@ -19,19 +29,32 @@ What went wrong? Paste the traceback or the wrong-output sample. Include the fil What did you ask the AI? Paste the actual prompt verbatim. (Include the code or stack trace you pasted alongside it; do NOT include any real `.env` values, API keys, or PII — replace those with ``.) ``` -(paste here) -``` +Traceback (most recent call last): File "", line 198, in _run_module_as_main File "", line 88, in _run_code File "C:\Users\Bader\Desktop\da\c55-data-week-2\task-1\src\pipeline.py", line 25, in from .transforms import ( File "C:\Users\Bader\Desktop\da\c55-data-week-2\task-1\src\transforms.py", line 76, in raw_rows = _required(INPUT_PATH) ^^^^^^^^^^^^^^^^^^^^^ File "C:\Users\Bader\Desktop\da\c55-data-week-2\task-1\src\config.py", line 28, in _required raise ValueError(f"Missing required environment variable: {name}") ValueError: Missing required environment variable: data/messy_sales.csv``` + + +then> + +INPUT_PATH: str = _required(INPUT_PATH) +OUTPUT_PATH: str = _required(OUTPUT_PATH) ## The Solution What did the AI suggest? Did it work on the first try? Did you have to follow up? Paste the final code change (a small diff is best). ``` -(paste here) + +AI suggested 4 complicated steps till i pasted this "INPUT_PATH: str = _required(INPUT_PATH) +OUTPUT_PATH: str = _required(OUTPUT_PATH)" + +then gave solution immediatley with alot of extra un needed info +INPUT_PATH: str = _required("INPUT_PATH") +OUTPUT_PATH: str = _required("OUTPUT_PATH") + +i forgot qoutes around input_path/output_path was fixed ``` ## Reflection Did you understand *why* the original code was broken before the AI told you? If not, what was the gap in your mental model? If you understood it before asking, why did you still ask the AI — speed, second opinion, or something else? -(write here, ~100 words) +(i did understand what AI told i simply couldnt see that i have missed the qoutes, trying to debug it would've taken much longer and much more effort. asking AI is faster and simpler) diff --git a/task-3/assets/azure_blob_week2.png b/task-3/assets/azure_blob_week2.png new file mode 100644 index 0000000..69ac170 Binary files /dev/null and b/task-3/assets/azure_blob_week2.png differ diff --git a/task-3/assets/blob_url.txt b/task-3/assets/blob_url.txt new file mode 100644 index 0000000..a3c3b47 --- /dev/null +++ b/task-3/assets/blob_url.txt @@ -0,0 +1 @@ +https://sthyfstudentsdemo.blob.core.windows.net/week2-noneeeed/clean_sales.csv \ No newline at end of file