﻿\documentclass[10pt]{article}
\usepackage[margin=1in]{geometry}
\usepackage{amsmath,amssymb}
\usepackage{algorithm}
\usepackage{algpseudocode}
\usepackage{booktabs}
\usepackage{array}
\usepackage{graphicx}
\usepackage{multirow}
\usepackage{makecell}
\renewcommand{\thetable}{S\arabic{table}}
\renewcommand{\thefigure}{S\arabic{figure}}

\newcommand{\xb}{\mathbf{x}}
\newcommand{\phivector}{\boldsymbol{\phi}}
\title{Supplementary Information: Methods Details}
\author{Chengye Yan}
\date{}

\begin{document}
\maketitle

\section*{Supplementary Methods}

\subsection*{S1. Lending Club Data Construction and Leakage Control}

The Lending Club cohort used in the main analysis was constructed from public LendingClub loan-level data accessed through the Kaggle Lending Club Loan Data mirror. The analysed cohort was restricted to accepted loans originated in Q4 2018 and to records with matured 36-month outcome information in the data snapshot used for analysis. After preprocessing, the Lending Club analytic sample contained 39,847 records and 26 application-time variables.

The binary outcome was constructed from loan performance status. The variable \texttt{loan\_status} was used only for label construction and was excluded from all model inputs. Positive default outcomes corresponded to default or charge-off status categories; performing or fully paid loans were treated as non-default outcomes. Records outside the Q4 2018 origination window or without the required outcome status for the 36-month target definition were excluded from the analytic cohort.

Feature screening followed a leakage-control rule: retained predictors had to be available at application time or loan origination. Identifiers, fields with more than 50\% missingness, and variables created after origination were removed before model fitting. The excluded post-origination field families included payment history, recovery, collection, hardship, settlement, servicing, and status/date variables. Representative excluded fields included \texttt{loan\_status}, \texttt{last\_pymnt\_*}, \texttt{last\_credit\_pull\_*}, \texttt{recoveries}, \texttt{collection\_*}, \texttt{total\_rec\_*}, \texttt{total\_pymnt\_*}, \texttt{out\_prncp\_*}, \texttt{next\_pymnt\_*}, \texttt{hardship\_*}, \texttt{settlement\_*}, and \texttt{debt\_settlement\_flag}. Missing numerical variables were imputed with training-set medians and missing categorical variables with training-set modes, followed by encoding and standardization within each training split.

These rules define the data-construction protocol used in the manuscript. Exact independent reconstruction of this cohort requires the executable preprocessing script, source-file identifier or snapshot date, complete 26-variable analytic feature list, and full leakage-exclusion list.

\subsection*{S2. Model Architecture Details}

\begin{table}[htbp]
\centering
\caption{Layer-level architecture specifications used in the main experiments.}
\begin{tabular}{@{}p{0.22\linewidth}p{0.68\linewidth}@{}}
\toprule
\textbf{Model} & \textbf{Specification} \\
\midrule
MLP & Three fully connected hidden layers with dimensions 256, 128, and 64; ReLU activation after each hidden layer; dropout $p=0.3$; sigmoid output unit for binary classification. \\
ResNet-1D & Three residual Conv1D blocks with channel widths 64, 128, and 256. Each block contains two Conv1D layers (kernel size 3, padding 1), BatchNorm, ReLU activation, and a skip connection. Blocks 2 and 3 use stride-2 downsampling in the first convolution. Adaptive average pooling and a 64-unit fully connected head precede the sigmoid output. \\
TabTransformer & Categorical features are embedded into 32-dimensional vectors and processed with 4-head self-attention blocks using LayerNorm and feed-forward layers of dimension 128. Numerical features are LayerNorm-normalized and projected to 128 dimensions. The categorical and numerical streams are concatenated and passed through a 128-to-64 MLP head with ReLU and sigmoid output. \\
\bottomrule
\end{tabular}
\end{table}

\subsection*{S3. Standard Attack Equations}

FGSM generates adversarial examples using a single gradient step:
\begin{equation}
\xb' = \xb + \varepsilon \cdot \mathrm{sign}\left(\nabla_{\xb} J(\xb, y; \boldsymbol{\theta})\right).
\end{equation}

PGD iterates the same first-order update with projection onto the $L_\infty$ ball:
\begin{equation}
\xb^{(t+1)} =
\Pi_{\mathcal{B}_{\varepsilon}(\xb)}
\left(\xb^{(t)} + \alpha \cdot
\mathrm{sign}\left(\nabla_{\xb}J(\xb^{(t)}, y; \boldsymbol{\theta})\right)\right).
\end{equation}

The CW attack was implemented as an $L_2$ optimization problem:
\begin{equation}
\min_{\boldsymbol{\delta}} \; \|\boldsymbol{\delta}\|_2 + c \cdot f(\xb + \boldsymbol{\delta}),
\end{equation}
where $c$ was selected through a 9-step binary search and the confidence parameter was evaluated over $\kappa \in \{0,5,10\}$.

\subsection*{S4. E-PGD Proxy Diagnostic}

The SHAP-gate-aware E-PGD diagnostic used a bi-objective projected gradient direction:
\begin{equation}
\mathbf{g}_{\mathrm{E\mbox{-}PGD}} =
\nabla_{\xb}\left[
\mathcal{L}_{\mathrm{cls}}(\xb^{(t)}, y)
- \lambda \sum_{i=1}^{3}
d_M\left(\phivector_i(\xb^{(t)}), \bar{\phivector}_i^{\mathrm{train}}\right)
\right],
\end{equation}
where $\lambda=0.05$, $\mathcal{L}_{\mathrm{cls}}$ is the classification loss, and $d_M$ is the Mahalanobis distance to the model-specific training SHAP centroid. The distance term was implemented as a differentiable attribution-distance surrogate rather than exact KernelSHAP differentiation. The perturbation update followed the PGD projection operator:
\begin{equation}
\xb^{(t+1)} =
\Pi_{\mathcal{B}_{\varepsilon}(\xb)}
\left(\xb^{(t)} + \alpha \cdot \mathrm{sign}(\mathbf{g}_{\mathrm{E\mbox{-}PGD}})\right).
\end{equation}
The diagnostic used $\varepsilon \in \{0.01,0.05,0.10,0.20\}$, $T=40$, $\alpha=\varepsilon/10$, and one random restart. It was used only as a routing-evasion sensitivity diagnostic and not as a full adaptive robustness evaluation.
KernelSHAP is not part of the differentiable computation graph; the attribution-distance term uses a differentiable surrogate (see Supplementary Algorithm~S3 for full pseudocode).

\subsection*{S5. Supplementary Algorithm S1: PGD Attack}

\begin{algorithm}[htbp]
\centering
\caption{Projected Gradient Descent Adversarial Attack}
\begin{algorithmic}[1]
\Require{Original sample $\xb$, true label $y$, model $f_{\boldsymbol{\theta}}$, loss $J$, perturbation budget $\varepsilon$, step size $\alpha$, iterations $T$, random restarts $R$}
\Ensure{Adversarial example $\xb_{\mathrm{adv}}$}
\State $\xb_{\mathrm{best}} \leftarrow \xb$, $\mathrm{loss}_{\mathrm{best}} \leftarrow -\infty$
\For{$r = 1$ \textbf{to} $R$}
    \State $\xb^{(0)} \leftarrow \xb + \mathbf{u}$, where $\mathbf{u} \sim \mathcal{U}(-\varepsilon,\varepsilon)$
    \For{$t = 0$ \textbf{to} $T-1$}
        \State $\mathbf{g} \leftarrow \nabla_{\xb} J(\xb^{(t)}, y; \boldsymbol{\theta})$
        \State $\xb^{(t+1)} \leftarrow \xb^{(t)} + \alpha \cdot \mathrm{sign}(\mathbf{g})$
        \State $\xb^{(t+1)} \leftarrow \Pi_{\mathcal{B}_{\varepsilon}(\xb)}(\xb^{(t+1)})$
        \State $\xb^{(t+1)} \leftarrow \mathrm{clip}(\xb^{(t+1)}, \mathbf{x}_{\min}, \mathbf{x}_{\max})$
    \EndFor
    \If{$J(\xb^{(T)}, y; \boldsymbol{\theta}) > \mathrm{loss}_{\mathrm{best}}$}
        \State $\xb_{\mathrm{best}} \leftarrow \xb^{(T)}$
        \State $\mathrm{loss}_{\mathrm{best}} \leftarrow J(\xb^{(T)}, y; \boldsymbol{\theta})$
    \EndIf
\EndFor
\State \Return $\xb_{\mathrm{best}}$
\end{algorithmic}
\end{algorithm}

\subsection*{S6. Supplementary Algorithm S2: Ensemble Adversarial Training}

\begin{algorithm}[htbp]
\centering
\caption{Ensemble Adversarial Training with SHAP-Based Routing Preparation}
\begin{algorithmic}[1]
\Require{Training data $\mathcal{D}_{\mathrm{train}}$, validation data $\mathcal{D}_{\mathrm{val}}$, architectures $\{\mathcal{A}_i\}_{i=1}^{K}$, PGD parameters $(\varepsilon,T,\alpha)$, SHAP background set $\mathcal{D}_{\mathrm{bg}}$}
\Ensure{Trained ensemble $\{f_{\boldsymbol{\theta}_i}\}_{i=1}^{K}$ and SHAP centroids $\{\bar{\phivector}_i\}_{i=1}^{K}$}
\For{$i = 1$ \textbf{to} $K$}
    \State Initialize $f_{\boldsymbol{\theta}_i}$ from architecture $\mathcal{A}_i$
    \For{epoch $=1$ \textbf{to} $N_{\mathrm{epochs}}$}
        \For{each mini-batch $(\xb_b,y_b)$}
            \State Generate $\xb_b^{\mathrm{adv}}$ using PGD
            \State Compute $J(\xb_b^{\mathrm{adv}},y_b;\boldsymbol{\theta}_i)$ and update $\boldsymbol{\theta}_i$
        \EndFor
        \State Evaluate clean AUC on $\mathcal{D}_{\mathrm{val}}$
        \If{validation AUC does not improve for 15 epochs}
            \State \textbf{break}
        \EndIf
    \EndFor
    \State Compute KernelSHAP vectors for $\mathcal{D}_{\mathrm{bg}}$
    \State Store centroid $\bar{\phivector}_i = |\mathcal{D}_{\mathrm{bg}}|^{-1}\sum_{\xb \in \mathcal{D}_{\mathrm{bg}}}\phivector_i(\xb)$
\EndFor
\State \Return $\{f_{\boldsymbol{\theta}_i}\}_{i=1}^{K}$ and $\{\bar{\phivector}_i\}_{i=1}^{K}$
\end{algorithmic}
\end{algorithm}

\subsection*{S7. Supplementary Algorithm S3: E-PGD Routing-Evasion Proxy Diagnostic}

\begin{algorithm}[htbp]
\centering
\caption{E-PGD Routing-Evasion Proxy Diagnostic}
\begin{algorithmic}[1]
\Require{Sample $\xb$, true label $y$, ensemble $\{f_{\boldsymbol{\theta}_i}\}_{i=1}^{K}$,
         SHAP centroids $\{\bar{\phivector}_i\}$,
         covariance matrices $\{\mathbf{S}_i\}$,
         budget $\varepsilon$, step size $\alpha$, iterations $T$,
         restarts $R{=}1$, surrogate weight $\lambda_{\mathrm{s}}{=}0.05$}
\Ensure{Adversarial example $\xb_{\mathrm{adv}}$}
\State $\xb_{\mathrm{best}} \leftarrow \xb$,\quad $\mathrm{loss}_{\mathrm{best}} \leftarrow -\infty$
\For{$r = 1$ \textbf{to} $R$}
    \State $\xb^{(0)} \leftarrow \xb + \mathbf{u}$,\quad $\mathbf{u} \sim \mathcal{U}(-\varepsilon, \varepsilon)$
    \For{$t = 0$ \textbf{to} $T-1$}
        \State $\mathbf{g}_{\mathrm{cls}} \leftarrow
               \nabla_{\xb}\,\mathcal{L}_{\mathrm{cls}}\!\left(\xb^{(t)}, y;\,
               \{f_{\boldsymbol{\theta}_i}\}\right)$
               \Comment{Gradient to maximize classification loss}
        \State $\mathbf{g}_{\mathrm{attr}} \leftarrow
               \nabla_{\xb}\sum_{i=1}^{K}
               d_M\!\left(\hat{\phivector}_i(\xb^{(t)}),\,\bar{\phivector}_i\right)$
               \Comment{$\hat{\phivector}_i$: differentiable surrogate; KernelSHAP not differentiated}
        \State $\mathbf{g} \leftarrow \mathbf{g}_{\mathrm{cls}}
               - \lambda_{\mathrm{s}}\,\mathbf{g}_{\mathrm{attr}}$
               \Comment{Negative sign: minimize attribution distance (keep SHAP close to centroid)}
        \State $\xb^{(t+1)} \leftarrow \xb^{(t)} + \alpha \cdot \mathrm{sign}(\mathbf{g})$
        \State $\xb^{(t+1)} \leftarrow
               \Pi_{\mathcal{B}_{\varepsilon}(\xb)}\!\left(\xb^{(t+1)}\right)$
    \EndFor
    \If{$\mathcal{L}_{\mathrm{cls}}\!\left(\xb^{(T)}, y\right) > \mathrm{loss}_{\mathrm{best}}$}
        \State $\xb_{\mathrm{best}} \leftarrow \xb^{(T)}$,\quad
               $\mathrm{loss}_{\mathrm{best}} \leftarrow
               \mathcal{L}_{\mathrm{cls}}\!\left(\xb^{(T)}, y\right)$
    \EndIf
\EndFor
\State \Return $\xb_{\mathrm{best}}$
\end{algorithmic}
\end{algorithm}

\noindent\textbf{Optimization boundary.}
KernelSHAP is a model-agnostic estimator based on coalition
sampling and is not part of the differentiable computation graph.
The attribution-distance term $\mathbf{g}_{\mathrm{attr}}$ uses a
differentiable surrogate that passes gradients through constituent
model outputs but not through the SHAP sampling procedure.
Consequently, E-PGD is a routing-evasion sensitivity diagnostic
rather than a fully adaptive attack evaluation against the complete
routed ensemble.

\section*{Supplementary Results}

\subsection*{Supplementary Tables}

\input{tables clean no highlight/Table_02_Evaluation_Metrics.tex}

\input{tables clean no highlight/Table_03_Degradation.tex}

\input{tables clean no highlight/Table_04_Transferability_Matrix.tex}

\input{tables clean no highlight/Table_07_SHAP_Stability.tex}

\input{tables clean no highlight/Table_S6_Efficiency_Baseline.tex}

\input{tables clean no highlight/Table_S7_Routing_Sensitivity.tex}

\input{tables clean no highlight/Table_S8_MultiRestart_Sensitivity.tex}

\input{tables clean no highlight/Table_S9_Confusion_Matrix.tex}

\input{tables clean no highlight/Table_S10_Failure_Profile.tex}

\input{tables clean no highlight/Table_S11_Chronological.tex}

\paragraph{Chronological sensitivity analysis.}
For Lending Club, we additionally evaluated a chronological sensitivity split using
the loan issue date for splitting only. The held-out December 2018 cohort contained
13,146 loans; earlier Q4 2018 records were used for training and validation. Under
10-restart PGD, the No Defense baseline declined from clean AUC 0.7533 to attacked
AUC 0.5642, whereas the Proposed configuration retained attacked AUC 0.7185 and
restored 286 of the 333 no-defense PGD-flipped default cases (DSR = 0.8589;
Supplementary Table~S11).

\subsection*{Supplementary Figures}

\begin{figure}[htbp]
\centering
\includegraphics[width=\linewidth]{../02_separate_figures/Figure_03_CW_CrossAttack.png}
\caption{Vulnerability under CW attack and cross-attack comparison. (a) AUC as a function of CW confidence parameter $\kappa \in \{0, 5, 10\}$ for three architectures on the German Credit dataset. (b) Same analysis for the Lending Club dataset. The CW attack induces a more gradual degradation pattern than FGSM/PGD under the selected settings. (c) Grouped bar chart comparing maximum AUC degradation ($\Delta$AUC) across FGSM, PGD, and CW for each architecture and dataset. Error bars represent $\pm$SD over 5 seeds.}
\label{sfig:cw_crossattack}
\end{figure}

\begin{figure}[htbp]
\centering
\includegraphics[width=\linewidth]{../02_separate_figures/Figure_06_Ablation_Waterfall.png}
\caption{Ablation study: marginal contribution of each defense framework component. Horizontal waterfall chart showing the cumulative AUC improvement on the German Credit dataset under PGD attack ($\varepsilon = 0.1$). Starting from the No Defense baseline (AUC = 0.556), successive components add adversarial training, heterogeneous ensemble branches, and SHAP-based soft voting routing. Error bars represent $\pm$SD across 5 random seeds.}
\label{sfig:ablation_waterfall}
\end{figure}

\begin{figure}[htbp]
\centering
\includegraphics[width=\linewidth]{../02_separate_figures/Figure_07_Robustness_Accuracy.png}
\caption{Clean-versus-defended AUC tradeoff under PGD adversarial defense. The scatter plot compares clean-data AUC with post-defense AUC under PGD attack at $\varepsilon = 0.1$ for representative defense configurations on German Credit and Lending Club. The proposed framework occupies a favorable position among the selected configurations.}
\label{sfig:robustness_accuracy}
\end{figure}

\end{document}

