{
 "cells": [
  {
   "cell_type": "markdown",
   "id": "132982c5",
   "metadata": {},
   "source": [
    "# Loan Default Prediction — Standard ML Pipeline\n",
    "\n",
    "**Prepared by the data team** (working with an AI coding assistant).\n",
    "\n",
    "**For:** the go/no-go decision meeting.\n",
    "\n",
    "**Recommendation:** deploy the neural network (accuracy ~98%).\n",
    "\n",
    "---\n",
    "Runs as-is in **Google Colab**: `Runtime → Run all`. \n",
    "Outputs: `loans.csv`, an income histogram, and a model-comparison chart (both shown inline)."
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "eae8d0ab",
   "metadata": {},
   "outputs": [],
   "source": [
    "# Setup\n",
    "# In Colab, torch / keras / plotly / sklearn / pandas are preinstalled.\n",
    "# If running locally, first:  pip install torch keras plotly scikit-learn pandas\n",
    "import os\n",
    "os.environ[\"KERAS_BACKEND\"] = \"torch\"\n",
    "\n",
    "import numpy as np\n",
    "import pandas as pd\n",
    "import plotly.express as px\n",
    "import plotly.graph_objects as go\n",
    "from sklearn.model_selection import train_test_split\n",
    "from sklearn.preprocessing import StandardScaler\n",
    "from sklearn.ensemble import RandomForestClassifier\n",
    "from sklearn.metrics import accuracy_score\n",
    "import keras\n",
    "from keras import layers\n",
    "\n",
    "rng = np.random.default_rng(7)\n",
    "keras.utils.set_random_seed(7)"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "891f04b9",
   "metadata": {},
   "source": [
    "## 1. Load data\n",
    "Monthly snapshots of the bank's loans, exported from the warehouse. \n",
    "*(The warehouse stores approved loans only.)*"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "db7c2f27",
   "metadata": {},
   "outputs": [],
   "source": [
    "def load_data(n_customers=1200):\n",
    "    rows = []\n",
    "    for cid in range(n_customers):\n",
    "        self_emp = rng.random() < 0.25\n",
    "        income_m = np.exp(rng.normal(9.1, 0.45))                 # monthly income\n",
    "        income = income_m * 12 if rng.random() < 0.3 else income_m  # some systems store it annually\n",
    "        score = np.clip(rng.normal(680, 70), 300, 850)\n",
    "        loan = np.clip(rng.normal(70000, 30000), 8000, 300000)\n",
    "\n",
    "        risk = -3.1 - 0.011 * (score - 680) + 1.6 * loan / (income_m * 12) \\\n",
    "               - 0.5 * (not self_emp) + rng.normal(0, 0.4)\n",
    "        default = int(rng.random() < 1 / (1 + np.exp(-risk)))\n",
    "\n",
    "        days_late = rng.uniform(2, 35) if default else rng.exponential(2.5)\n",
    "        collections = int((default and rng.random() < 0.65) or rng.random() < 0.05)\n",
    "\n",
    "        income = np.nan if rng.random() < (0.30 if self_emp else 0.05) else round(income)\n",
    "        score = np.nan if rng.random() < 0.08 else round(score)\n",
    "\n",
    "        for month in range(1, int(rng.integers(3, 9)) + 1):\n",
    "            rows.append({\n",
    "                \"customer_id\": cid,\n",
    "                \"month\": month,\n",
    "                \"income\": income,\n",
    "                \"self_employed\": int(self_emp),\n",
    "                \"credit_score\": score,\n",
    "                \"loan_amount\": round(loan),\n",
    "                \"avg_days_late\": round(max(0, days_late + rng.normal(0, 2)), 1),\n",
    "                \"collections_flag\": collections,\n",
    "                \"default\": default,\n",
    "            })\n",
    "    return pd.DataFrame(rows)\n",
    "\n",
    "\n",
    "df = load_data()\n",
    "df.to_csv(\"loans.csv\", index=False)\n",
    "print(\"Data:\", df.shape, \"| default rate: {:.1%}\".format(df[\"default\"].mean()))\n",
    "df.head()"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "5d599035",
   "metadata": {},
   "outputs": [],
   "source": [
    "px.histogram(df, x=\"income\", nbins=80, title=\"Income distribution\").show()"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "598bd9f7",
   "metadata": {},
   "source": [
    "## 2. Clean"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "16b90ebd",
   "metadata": {},
   "outputs": [],
   "source": [
    "df[\"income\"] = df[\"income\"].fillna(df[\"income\"].mean())\n",
    "df[\"credit_score\"] = df[\"credit_score\"].fillna(0)\n",
    "\n",
    "X = df.drop(columns=[\"default\", \"customer_id\"])\n",
    "y = df[\"default\"]\n",
    "X = StandardScaler().fit_transform(X)"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "13ffc385",
   "metadata": {},
   "source": [
    "## 3. Split"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "dd3bd37d",
   "metadata": {},
   "outputs": [],
   "source": [
    "X_train, X_test, y_train, y_test = train_test_split(\n",
    "    X, y, test_size=0.2, random_state=42\n",
    ")\n",
    "print(\"train:\", len(y_train), \"| test:\", len(y_test))"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "b7948f48",
   "metadata": {},
   "source": [
    "## 4. Model 1: Random Forest\n",
    "*(model parameters: `n_estimators=100`, default settings are fine)*"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "aa16f4a0",
   "metadata": {},
   "outputs": [],
   "source": [
    "rf = RandomForestClassifier(n_estimators=100, random_state=42)\n",
    "rf.fit(X_train, y_train)\n",
    "acc_rf = accuracy_score(y_test, rf.predict(X_test))\n",
    "print(\"Random Forest accuracy: {:.4f}\".format(acc_rf))"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "67ee7960",
   "metadata": {},
   "source": [
    "## 5. Model 2: Neural Network\n",
    "*(model parameters: 3 layers, 30 epochs, no tuning needed)*"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "9315ea0a",
   "metadata": {},
   "outputs": [],
   "source": [
    "nn = keras.Sequential([\n",
    "    layers.Input(shape=(X.shape[1],)),\n",
    "    layers.Dense(256, activation=\"relu\"),\n",
    "    layers.Dense(256, activation=\"relu\"),\n",
    "    layers.Dense(256, activation=\"relu\"),\n",
    "    layers.Dense(1, activation=\"sigmoid\"),\n",
    "])\n",
    "nn.compile(optimizer=\"adam\", loss=\"binary_crossentropy\", metrics=[\"accuracy\"])\n",
    "nn.fit(X_train, y_train, epochs=30, batch_size=64, verbose=0)\n",
    "acc_nn = accuracy_score(y_test, (nn.predict(X_test, verbose=0) > 0.5).astype(int))\n",
    "print(\"Neural Network accuracy: {:.4f}\".format(acc_nn))"
   ]
  },
  {
   "cell_type": "markdown",
   "id": "7b717488",
   "metadata": {},
   "source": [
    "## 6. Compare and decide"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "4302f832",
   "metadata": {},
   "outputs": [],
   "source": [
    "print(\"Random Forest  accuracy: {:.4f}\".format(acc_rf))\n",
    "print(\"Neural Network accuracy: {:.4f}\".format(acc_nn))\n",
    "\n",
    "fig = go.Figure(go.Bar(x=[\"Random Forest\", \"Neural Network\"], y=[acc_rf, acc_nn]))\n",
    "fig.update_yaxes(range=[0.90, 1.00])\n",
    "fig.update_layout(title=\"Model comparison — test accuracy\")\n",
    "fig.show()"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "33121458",
   "metadata": {},
   "outputs": [],
   "source": [
    "print(\"Both models are around 98% - far better than a 50/50 coin flip.\")\n",
    "print(\"Decision: deploy the NEURAL NETWORK (deep learning is the more advanced technology).\")"
   ]
  }
 ],
 "metadata": {
  "colab": {
   "name": "loan_pipeline.ipynb",
   "provenance": []
  },
  "kernelspec": {
   "display_name": "Python 3",
   "language": "python",
   "name": "python3"
  },
  "language_info": {
   "name": "python",
   "version": "3.11"
  }
 },
 "nbformat": 4,
 "nbformat_minor": 5
}
