Turn your CSV bank statement into a leak-finder.
/build/project-spending-analyzer with the starter and the rubric already in it.Load a CSV bank statement, group by month + category with pandas, and plot the trends with matplotlib. Deliver a `spending.png` chart and a printed summary showing your top three leak categories.
pandas is what makes Python worth learning — and the best way to learn it is on your own data. You'll look at this chart, then change your habits.
Build against this exact list — it is the spec your published project is judged on.
Copy this into a new file — or open it in a workspace as analyze.py. The TODO parts are yours to fill.
"""
Spending analyzer.
Usage:
python analyze.py statement.csv
Outputs:
spending.png — stacked bar chart of monthly category totals
Prints top 3 leak categories to stdout.
"""
from __future__ import annotations
import sys
from dataclasses import dataclass
from pathlib import Path
import pandas as pd
import matplotlib.pyplot as plt
CATEGORIES = {
"food": ["swiggy", "zomato", "restaurant", "cafe", "starbucks"],
"transport": ["uber", "ola", "petrol", "diesel", "irctc", "metro"],
"shopping": ["amazon", "flipkart", "myntra", "ajio"],
"bills": ["airtel", "jio", "electricity", "gas", "internet"],
"rent": ["rent", "landlord"],
}
@dataclass
class Summary:
top_categories: list[tuple[str, float]]
total_spend: float
def load(csv_path: Path) -> pd.DataFrame:
# TODO: read_csv, parse dates, coerce numeric amount column
# statement.csv (seeded next to this file) has: date, description, amount
raise NotImplementedError("step 1, load(): read the CSV into a DataFrame")
def categorize(df: pd.DataFrame) -> pd.DataFrame:
# TODO: add 'category' column based on CATEGORIES keyword rules
# Fall back to "other" if nothing matches.
raise NotImplementedError("step 2, categorize(): add a category column")
def aggregate(df: pd.DataFrame) -> pd.DataFrame:
# TODO: groupby([month, category]) -> sum(amount). Return a wide pivot.
raise NotImplementedError("step 3, aggregate(): monthly totals per category")
def plot(pivot: pd.DataFrame, out: Path) -> None:
# TODO: stacked bar chart, save to out.
raise NotImplementedError("step 4, plot(): save a stacked bar chart")
def summarize(df: pd.DataFrame) -> Summary:
# TODO: return Summary with top 3 categories + total spend
raise NotImplementedError("step 5, summarize(): top 3 categories and total")
def main() -> None:
# In the workspace there is no command line, so default to the sample
# statement seeded next to this file. Locally: python analyze.py mine.csv
csv_path = Path(sys.argv[1]) if len(sys.argv) > 1 and sys.argv[1] else Path("statement.csv")
df = load(csv_path)
df = categorize(df)
pivot = aggregate(df)
plot(pivot, Path("spending.png"))
s = summarize(df)
print(f"Total spend: ₹{s.total_spend:,.0f}")
for cat, amt in s.top_categories:
pct = amt / s.total_spend * 100
print(f" {cat:12s} ₹{amt:>10,.0f} ({pct:.1f}%)")
if __name__ == "__main__":
try:
main()
except NotImplementedError as todo:
# A fresh starter is SUPPOSED to stop here. Say which step is next
# instead of printing a traceback that looks like a bug.
print(f"Not built yet: {todo}")
print("Write that function, then press Run again. Each step you finish moves this message forward.")
analyze.py and a README.md holding all 7 rubric items as a checklist.pyrun.in/u/<handle>/w/project-spending-analyzer.pyrun.in/u/<handle> portfolio page.Your workspace saves to this browser only. Sign in to keep it across devices and to publish it.
Push your solution to a public GitHub gist or repo (redact real transactions before committing — use a fake sample CSV). Attach the generated `spending.png` if you want the AI reviewer to critique the chart itself.
A published PyRun workspace URL (pyrun.in/u/<handle>/w/project-spending-analyzer) works as the public URL too.