{"spec_id":"shap-waterfall","library":"letsplot","language":"python","code":"\"\"\" anyplot.ai\nshap-waterfall: SHAP Waterfall Plot for Feature Attribution\nLibrary: letsplot 4.9.0 | Python 3.13.13\nQuality: 88/100 | Created: 2026-05-07\n\"\"\"\n\nimport os\n\nimport numpy as np\nimport pandas as pd\nfrom lets_plot import *\n\n\nLetsPlot.setup_html()\n\n# Theme tokens\nTHEME = os.getenv(\"ANYPLOT_THEME\", \"light\")\nPAGE_BG = \"#FAF8F1\" if THEME == \"light\" else \"#1A1A17\"\nELEVATED_BG = \"#FFFDF6\" if THEME == \"light\" else \"#242420\"\nINK = \"#1A1A17\" if THEME == \"light\" else \"#F0EFE8\"\nINK_SOFT = \"#4A4A44\" if THEME == \"light\" else \"#B8B7B0\"\nINK_MUTED = \"#6B6A63\" if THEME == \"light\" else \"#A8A79F\"\n\nPOS_COLOR = \"#AE3030\"  # imprint red — positive SHAP (increases predicted risk)\nNEG_COLOR = \"#4467A3\"  # Okabe-Ito blue — negative SHAP (decreases predicted risk)\n\n# Data — loan default risk model explaining one applicant's prediction\n# Features already sorted by absolute SHAP magnitude (largest at top)\nfeature_names = [\n    \"Payment History\",\n    \"Debt-to-Income Ratio\",\n    \"Annual Income\",\n    \"Credit Score\",\n    \"Loan Amount\",\n    \"Employment Years\",\n    \"Age\",\n    \"Open Accounts\",\n    \"Housing Status\",\n    \"Loan Purpose\",\n    \"Savings Balance\",\n    \"Marital Status\",\n]\nshap_values = np.array([0.18, 0.12, -0.11, -0.09, 0.08, -0.06, -0.05, 0.04, -0.03, 0.02, -0.02, 0.01])\nbase_value = 0.35\nfinal_value = round(base_value + float(shap_values.sum()), 4)\n\nn = len(feature_names)\ny_pos = list(range(n, 0, -1))  # largest contributor at top (y=n), smallest at bottom (y=1)\n\n# Cumulative waterfall positions\ncumulative = base_value\nbar_starts = np.empty(n)\nbar_ends = np.empty(n)\nfor i, sv in enumerate(shap_values):\n    bar_starts[i] = cumulative\n    cumulative += sv\n    bar_ends[i] = cumulative\n\n# Main bars DataFrame\ndf_bars = pd.DataFrame(\n    {\n        \"y\": y_pos,\n        \"ymin\": [y - 0.38 for y in y_pos],\n        \"ymax\": [y + 0.38 for y in y_pos],\n        \"xmin\": np.minimum(bar_starts, bar_ends),\n        \"xmax\": np.maximum(bar_starts, bar_ends),\n        \"direction\": [\"positive\" if sv >= 0 else \"negative\" for sv in shap_values],\n        \"shap_value\": shap_values,\n        \"feature\": feature_names,\n    }\n)\n\n# SHAP value labels: positive bars get text to the right, negative to the left\ndf_bars[\"text_x\"] = np.where(\n    shap_values >= 0, np.maximum(bar_starts, bar_ends) + 0.007, np.minimum(bar_starts, bar_ends) - 0.007\n)\ndf_bars[\"label\"] = [f\"+{sv:.3f}\" if sv >= 0 else f\"{sv:.3f}\" for sv in shap_values]\n\ndf_pos = df_bars[df_bars[\"shap_value\"] >= 0]\ndf_neg = df_bars[df_bars[\"shap_value\"] < 0]\n\n# Vertical connector segments in the gap between consecutive bars\ndf_conn = pd.DataFrame(\n    {\n        \"x\": bar_ends[:-1],\n        \"xend\": bar_ends[:-1],\n        \"y\": [y - 0.38 for y in y_pos[:-1]],\n        \"yend\": [y + 0.38 for y in y_pos[1:]],\n    }\n)\n\n# Reference line annotation labels (floated above all bars)\n# Base: left-aligned from reference line; Predicted: centered over its line\ndf_ref_base = pd.DataFrame({\"x\": [base_value + 0.003], \"y\": [n + 1.1], \"label\": [f\"Base = {base_value:.2f}\"]})\ndf_ref_pred = pd.DataFrame({\"x\": [final_value], \"y\": [n + 1.1], \"label\": [f\"Predicted = {final_value:.2f}\"]})\n\nanyplot_theme = theme(\n    plot_background=element_rect(fill=PAGE_BG, color=PAGE_BG),\n    panel_background=element_rect(fill=PAGE_BG),\n    panel_border=element_blank(),\n    panel_grid_major_x=element_line(color=INK_MUTED, size=0.15),\n    panel_grid_major_y=element_blank(),\n    panel_grid_minor=element_blank(),\n    axis_title=element_text(color=INK, size=20),\n    axis_text_y=element_text(color=INK_SOFT, size=16),\n    axis_text_x=element_text(color=INK_SOFT, size=14),\n    axis_line_x=element_line(color=INK_SOFT, size=0.5),\n    axis_line_y=element_blank(),\n    axis_ticks=element_blank(),\n    plot_title=element_text(color=INK, size=24),\n    legend_position=\"none\",\n)\n\nplot = (\n    ggplot()\n    + geom_vline(xintercept=float(base_value), color=INK_SOFT, size=0.9, linetype=\"dashed\")\n    + geom_rect(data=df_bars, mapping=aes(xmin=\"xmin\", xmax=\"xmax\", ymin=\"ymin\", ymax=\"ymax\", fill=\"direction\"))\n    + geom_segment(\n        data=df_conn, mapping=aes(x=\"x\", xend=\"xend\", y=\"y\", yend=\"yend\"), color=INK_SOFT, size=0.5, linetype=\"dotted\"\n    )\n    + geom_text(data=df_pos, mapping=aes(x=\"text_x\", y=\"y\", label=\"label\"), color=INK_MUTED, size=10, hjust=0)\n    + geom_text(data=df_neg, mapping=aes(x=\"text_x\", y=\"y\", label=\"label\"), color=INK_MUTED, size=10, hjust=1)\n    + geom_text(data=df_ref_base, mapping=aes(x=\"x\", y=\"y\", label=\"label\"), color=INK, size=11, hjust=0)\n    + geom_text(data=df_ref_pred, mapping=aes(x=\"x\", y=\"y\", label=\"label\"), color=INK, size=11, hjust=0.5)\n    + geom_vline(xintercept=float(final_value), color=INK, size=1.2)\n    + scale_fill_manual(values={\"positive\": POS_COLOR, \"negative\": NEG_COLOR})\n    + scale_x_continuous(limits=[0.28, 0.72], expand=[0, 0])\n    + scale_y_continuous(breaks=y_pos, labels=feature_names, limits=[0.5, n + 1.5])\n    + labs(\n        x=\"SHAP Value  ·  Impact on Predicted Default Probability\",\n        y=\"\",\n        title=\"Credit Default Risk  ·  shap-waterfall  ·  letsplot  ·  anyplot.ai\",\n    )\n    + ggsize(1600, 900)\n    + theme_minimal()\n    + anyplot_theme\n)\n\nggsave(plot, f\"plot-{THEME}.png\", scale=3, path=\".\")\nggsave(plot, f\"plot-{THEME}.html\", path=\".\")\n"}