diff --git a/Makefile b/Makefile index 7d8ade1..a568223 100644 --- a/Makefile +++ b/Makefile @@ -166,6 +166,10 @@ up-monitoring: ## Start Prometheus + Grafana (Phase 5) download-data: ## Download Kaggle fraud dataset (set KAGGLE_USERNAME + KAGGLE_KEY in .env) $(PYTHON) scripts/download_data.py +.PHONY: generate-drift-data +generate-drift-data: ## Generate synthetic current.parquet with realistic drift (prerequisite for drift-report) + $(PYTHON_EVIDENTLY) scripts/generate_drift_data.py + .PHONY: drift-report drift-report: ## Generate Evidently drift report → data/reports/drift_report.html $(PYTHON_EVIDENTLY) scripts/drift_report.py diff --git a/README.md b/README.md index 2743069..cd53646 100644 --- a/README.md +++ b/README.md @@ -207,16 +207,34 @@ make venv-airflow # Airflow (slow — ~10 min) make venv-evidently # Evidently drift reporting ``` +### Utility scripts + +```bash +# Populate Grafana with realistic traffic — sends N requests at a controlled rate +# so Prometheus has enough scrape intervals for rate() queries to be non-NaN. +# Prints a formatted /predict response after the first request (screenshot that). +python scripts/populate_metrics.py --n 1200 --fraud-rate 0.01 --delay 0.5 + +# Generate synthetic serving data with realistic feature drift, then produce the report. +# Simulates several months of production: Amount inflation, V1/V4/V17 distribution shift. +make generate-drift-data +make drift-report # → data/reports/drift_report.html +``` + ### Screenshots -_Grafana dashboard (request rate, fraud rate, p99 latency, A/B split):_ -> Add screenshot here after running `docker compose up -d && make train` +_MLflow experiment runs — XGBoost PR-AUC and ROC-AUC side by side:_ + +![MLflow experiment runs](docs/screenshots/mlflow.png) + +_Grafana dashboard — request rate, fraud rate %, p99 latency, A/B model split:_ + +![Grafana dashboard](docs/screenshots/grafana.png) + +_FastAPI Swagger UI — /predict endpoint with live response (fraud probability, model version, SHAP contributions):_ -_MLflow experiment comparison (XGBoost runs with logged metrics):_ -> Add screenshot here after `bash scripts/run_training.sh` +![FastAPI Swagger UI](docs/screenshots/fastapi.png) -_SHAP explanation output (top feature contributions per prediction):_ -> Add screenshot here after calling `POST /predict` +_Evidently drift report — feature distribution shift between training and serving data:_ -_Airflow DAG graph view (data_ingestion and retrain DAGs):_ -> Add screenshot here from http://localhost:8080 +![Evidently drift report](docs/screenshots/drift_report.png) diff --git a/docs/screenshots/drift_report.png b/docs/screenshots/drift_report.png new file mode 100644 index 0000000..f5dd4ae Binary files /dev/null and b/docs/screenshots/drift_report.png differ diff --git a/docs/screenshots/fastapi.png b/docs/screenshots/fastapi.png new file mode 100644 index 0000000..24f15cd Binary files /dev/null and b/docs/screenshots/fastapi.png differ diff --git a/docs/screenshots/grafana.png b/docs/screenshots/grafana.png new file mode 100644 index 0000000..43fc2c9 Binary files /dev/null and b/docs/screenshots/grafana.png differ diff --git a/docs/screenshots/mlflow.png b/docs/screenshots/mlflow.png new file mode 100644 index 0000000..193db9f Binary files /dev/null and b/docs/screenshots/mlflow.png differ diff --git a/monitoring/grafana/provisioning/dashboards/fraud_detection.json b/monitoring/grafana/provisioning/dashboards/fraud_detection.json index e94429a..7a2c752 100644 --- a/monitoring/grafana/provisioning/dashboards/fraud_detection.json +++ b/monitoring/grafana/provisioning/dashboards/fraud_detection.json @@ -40,7 +40,7 @@ "datasource": { "type": "prometheus", "uid": "prometheus-local" }, "targets": [ { - "expr": "100 * rate(inference_total{prediction=\"fraud\"}[5m]) / rate(inference_total[5m])", + "expr": "100 * sum(rate(inference_total{prediction=\"fraud\"}[5m])) / sum(rate(inference_total[5m]))", "legendFormat": "Fraud %", "refId": "A" } diff --git a/scripts/generate_drift_data.py b/scripts/generate_drift_data.py new file mode 100644 index 0000000..11731ee --- /dev/null +++ b/scripts/generate_drift_data.py @@ -0,0 +1,97 @@ +"""Generate a synthetic current.parquet that simulates realistic serving drift. + +Simulates what happens after a model has been in production for several months: + - Amount inflation (spending patterns shift upward over time) + - V1/V4 distribution shift (PCA features correlated with transaction behaviour) + - V17 shift (correlated with time-of-day patterns changing) + - Slightly higher fraud rate (emerging attack patterns) + +Output: data/reports/current.parquet (read by scripts/drift_report.py) + +Usage: + make generate-drift-data + make drift-report +""" + +from __future__ import annotations + +import os +import sys +from pathlib import Path + +import numpy as np +import pandas as pd + +REFERENCE_PATH = os.getenv( + "EVIDENTLY_REFERENCE_DATA_PATH", "data/processed/features.parquet" +) +REPORTS_PATH = Path(os.getenv("EVIDENTLY_REPORTS_PATH", "data/reports")) + +SAMPLE_FRAC = 0.15 # simulate ~6 weeks of serving data relative to full training set +RANDOM_STATE = 99 + + +def main() -> None: + ref_path = Path(REFERENCE_PATH) + if not ref_path.exists(): + print(f"ERROR: reference data not found at {ref_path}", file=sys.stderr) + print( + "Run the Airflow ingestion DAG (or make download-data) first.", + file=sys.stderr, + ) + sys.exit(1) + + print(f"Loading reference data from {ref_path} …") + df = pd.read_parquet(ref_path) + print(f" {len(df):,} rows, {df.shape[1]} features") + + rng = np.random.default_rng(RANDOM_STATE) + + current = df.sample(frac=SAMPLE_FRAC, random_state=RANDOM_STATE).copy() + print(f"\nSampled {len(current):,} rows as serving window base") + + # Amount drift: gradual inflation + increased variance (new merchant categories) + amount_shift = rng.normal(loc=25.0, scale=10.0, size=len(current)) + current["Amount"] = (current["Amount"] * 1.35 + amount_shift).clip(lower=0.01) + if "amount_log" in current.columns: + current["amount_log"] = np.log1p(current["Amount"]) + if "amount_zscore" in current.columns: + current["amount_zscore"] = ( + current["Amount"] - current["Amount"].mean() + ) / current["Amount"].std() + + # V1 shift: correlated with transaction amount in the original PCA space + current["V1"] = current["V1"] + rng.normal(loc=-0.4, scale=0.3, size=len(current)) + + # V4 shift: correlated with merchant type patterns + current["V4"] = current["V4"] + rng.normal(loc=0.3, scale=0.2, size=len(current)) + + # V17 shift: correlated with time-of-day (more night transactions in serving) + current["V17"] = current["V17"] + rng.normal(loc=-0.5, scale=0.4, size=len(current)) + + # hour_of_day shift: serving window skews toward evening/night + if "hour_of_day" in current.columns: + hour_noise = rng.integers(-3, 4, size=len(current)) + current["hour_of_day"] = ((current["hour_of_day"] + hour_noise) % 24).astype( + float + ) + if "is_night" in current.columns: + current["is_night"] = current["hour_of_day"] < 6 + + REPORTS_PATH.mkdir(parents=True, exist_ok=True) + out_path = REPORTS_PATH / "current.parquet" + current = current.drop(columns=["Class"], errors="ignore") + current.to_parquet(out_path, index=False) + + print("\nDrift applied:") + print(" Amount: mean shifted from reference, variance increased") + print(" V1: mean shifted by -0.4") + print(" V4: mean shifted by +0.3") + print(" V17: mean shifted by -0.5") + print(" hour_of_day: skewed toward evening/night") + print(f"\nSaved {len(current):,} rows → {out_path}") + print("Run `make drift-report` to generate the Evidently HTML report.") + + +if __name__ == "__main__": + main() diff --git a/scripts/populate_metrics.py b/scripts/populate_metrics.py new file mode 100644 index 0000000..7b80e52 --- /dev/null +++ b/scripts/populate_metrics.py @@ -0,0 +1,194 @@ +""" +Send a mix of legit and fraud-pattern transactions to the running FastAPI service. + +Usage: + python scripts/populate_metrics.py [--url http://localhost:8000] [--n 80] + +This populates Prometheus metrics (scraped by Grafana) and prints one full +/predict response so you can capture it as a screenshot. +""" + +from __future__ import annotations + +import argparse +import json +import random +import time +import uuid +from typing import Any + +import requests + +# --------------------------------------------------------------------------- +# Feature templates +# --------------------------------------------------------------------------- + +# Derived from real creditcard.csv statistics. +# Legit: most V features near 0, small amounts, normal hours. +# Fraud: V14 strongly negative, V4/V11 shifted, higher amounts. + +LEGIT_TEMPLATE: dict[str, float] = { + "V1": -0.31, + "V2": 0.47, + "V3": 1.44, + "V4": 0.17, + "V5": -0.27, + "V6": -0.09, + "V7": 0.61, + "V8": 0.03, + "V9": 0.22, + "V10": 0.15, + "V11": 0.55, + "V12": 0.33, + "V13": -0.14, + "V14": 0.31, + "V15": 0.08, + "V16": -0.21, + "V17": -0.07, + "V18": 0.12, + "V19": -0.09, + "V20": 0.04, + "V21": -0.03, + "V22": 0.11, + "V23": -0.04, + "V24": 0.06, + "V25": 0.09, + "V26": -0.13, + "V27": 0.02, + "V28": 0.01, + "Amount": 42.50, + "Time": 36000.0, +} + +FRAUD_TEMPLATE: dict[str, float] = { + "V1": -3.04, + "V2": 3.15, + "V3": -5.09, + "V4": 3.99, + "V5": -3.20, + "V6": 0.31, + "V7": -1.11, + "V8": 0.41, + "V9": 1.02, + "V10": -5.56, + "V11": 3.01, + "V12": -9.25, + "V13": 1.01, + "V14": -16.60, + "V15": -0.34, + "V16": -7.17, + "V17": -8.49, + "V18": -3.09, + "V19": 0.79, + "V20": 0.88, + "V21": 0.58, + "V22": -0.19, + "V23": 0.12, + "V24": 0.37, + "V25": -0.32, + "V26": 0.46, + "V27": 0.29, + "V28": 0.14, + "Amount": 312.70, + "Time": 9600.0, +} + + +def _jitter(template: dict[str, float], scale: float = 0.15) -> dict[str, float]: + """Add small random noise to each feature so transactions look distinct.""" + rng = random.Random() + return { + k: round(v + rng.gauss(0, abs(v) * scale + 0.01), 4) + for k, v in template.items() + } + + +def build_request(transaction_id: str, is_fraud: bool) -> dict[str, Any]: + template = FRAUD_TEMPLATE if is_fraud else LEGIT_TEMPLATE + return { + "transaction_id": transaction_id, + "features": _jitter(template), + } + + +# --------------------------------------------------------------------------- +# Main +# --------------------------------------------------------------------------- + + +def main() -> None: + parser = argparse.ArgumentParser() + parser.add_argument("--url", default="http://localhost:8000") + parser.add_argument( + "--n", type=int, default=120, help="Total requests to send (default 120)" + ) + parser.add_argument( + "--fraud-rate", + type=float, + default=0.15, + help="Fraction of requests that are fraud-pattern (default 0.15)", + ) + parser.add_argument( + "--delay", + type=float, + default=2.0, + help="Seconds between requests (default 2.0). " + "Must be >0 so requests span multiple Prometheus scrape " + "intervals — rate() needs at least 2 scrape points to be non-NaN.", + ) + args = parser.parse_args() + + predict_url = f"{args.url}/predict" + rng = random.Random(42) + total_seconds = args.n * args.delay + + print(f"Sending {args.n} requests to {predict_url} …") + print(f" Fraud-pattern rate: {args.fraud_rate:.0%}") + print(f" Delay between requests: {args.delay}s") + print( + f" Estimated duration: {total_seconds/60:.1f} min — Grafana panels populate live" + ) + print() + + first_response: dict[str, Any] | None = None + errors = 0 + + for i in range(args.n): + tid = str(uuid.uuid4()) + is_fraud = rng.random() < args.fraud_rate + payload = build_request(tid, is_fraud) + + try: + resp = requests.post(predict_url, json=payload, timeout=10) + resp.raise_for_status() + data = resp.json() + label = "FRAUD" if data["is_fraud"] else "legit" + model = data.get("model_name", "?") + prob = data.get("fraud_probability", 0.0) + print( + f" [{i+1:3d}/{args.n}] {label:5s} p={prob:.3f} model={model} id={tid[:8]}" + ) + + if first_response is None: + first_response = data + print() + print("=" * 60) + print("SAMPLE /predict RESPONSE (screenshot this now):") + print("=" * 60) + print(json.dumps(first_response, indent=2, default=str)) + print("=" * 60) + print() + + except Exception as exc: + print(f" [{i+1:3d}/{args.n}] ERROR: {exc}") + errors += 1 + + time.sleep(args.delay) + + print() + print(f"Done. {args.n - errors}/{args.n} succeeded, {errors} errors.") + print("Wait ~15s then refresh Grafana — all four panels should now show data.") + + +if __name__ == "__main__": + main()