API Examples

One page per function, each executed on every build with the output it actually produced. Selecting comes first: compile_selected() chooses the target, the tree count and the depth, and reports once. The rest start from the same setup — the public UCI credit-card default panel from load_credit_default(), a seeded split, a ceiling model and a compiled artifact — so pages can be read in any order. Credit is where CompileML started; the mechanism is not specific to it.

"""Shared state for every worked example.

Each example page shows a short piece of code and the output it actually
produced. They all need the same thing first: a dataset, a teacher, a
compiled artifact. Rebuilding that twenty times would make the pipeline slow
enough that nobody would run it, so it is built once and cached.

The cache is keyed on the pinned library commit. Bump the pin and it
rebuilds; nothing here can quietly describe an artifact from an older build.

This module is shown on the site as the setup every example assumes.
"""

from __future__ import annotations

import json
import pathlib
import pickle

import numpy as np
from sklearn.ensemble import HistGradientBoostingClassifier
from sklearn.model_selection import train_test_split

from compileml.artifact import build_artifact, save_artifact
from compileml.bands import monotone_quantile_bands
from compileml.compile import train_whitebox
from compileml.datasets import load_credit_default
from compileml.runtime import load_artifact

PIPELINE = pathlib.Path(__file__).resolve().parent.parent
CACHE = PIPELINE / ".cache"
PIN = json.loads((PIPELINE / "pin.json").read_text(encoding="utf-8"))
STAMP = PIN["compileml_commit"][:12]

SEED = 42

# One entry per feature. CompileML measures reason coverage and warns when it
# is incomplete, because generic feature names are not suitable text for an
# adverse-action notice — the wording is the institution's policy, not the
# library's. These are written for the example, not for a real lender.
REASONS = {
    "LIMIT_BAL": {
        "code": "LOW_CREDIT_LIMIT",
        "negative": "The approved credit limit on this account is low.",
        "positive": "The approved credit limit on this account is high.",
    },
    **{
        f"PAY_{i}": {
            "code": f"REPAYMENT_STATUS_M{i}",
            "negative": f"Repayment was behind schedule {i} month(s) before assessment.",
            "positive": f"Repayment was on or ahead of schedule {i} month(s) before assessment.",
        }
        for i in range(1, 7)
    },
    **{
        f"BILL_AMT{i}": {
            "code": f"STATEMENT_BALANCE_M{i}",
            "negative": f"The statement balance {i} month(s) before assessment was high.",
            "positive": f"The statement balance {i} month(s) before assessment was low.",
        }
        for i in range(1, 7)
    },
    **{
        f"PAY_AMT{i}": {
            "code": f"AMOUNT_PAID_M{i}",
            "negative": f"The amount paid {i} month(s) before assessment was small.",
            "positive": f"The amount paid {i} month(s) before assessment was large.",
        }
        for i in range(1, 7)
    },
}


def _build() -> dict:
    X, y, feature_names = load_credit_default()
    X_train, X_test, y_train, y_test = train_test_split(
        X, y, test_size=0.25, stratify=y, random_state=SEED
    )

    teacher = HistGradientBoostingClassifier(random_state=0).fit(X_train, y_train)
    teacher_scores = teacher.predict_proba(X_train)[:, 1]
    teacher_test_scores = teacher.predict_proba(X_test)[:, 1]

    whitebox, fidelity = train_whitebox(X_train, teacher_scores)
    latent = whitebox.predict(X_train).clip(0, 1)

    bands = monotone_quantile_bands(latent, y_train, n_bands=10)

    artifact = build_artifact(
        whitebox,
        feature_names,
        baseline=np.median(X_train, axis=0),
        band_edges=bands,
        calibration_latent=latent,
        calibration_y=y_train,
        reasons=REASONS,
    )

    CACHE.mkdir(parents=True, exist_ok=True)
    save_artifact(artifact, str(CACHE / f"decision-{STAMP}.json"))

    return {
        "X_train": X_train,
        "X_test": X_test,
        "y_train": y_train,
        "y_test": y_test,
        "feature_names": feature_names,
        "teacher_scores": teacher_scores,
        "teacher_test_scores": teacher_test_scores,
        "whitebox": whitebox,
        "fidelity": fidelity,
        "latent": latent,
        "bands": bands,
    }


def _load() -> dict:
    blob = CACHE / f"setup-{STAMP}.pkl"
    if blob.is_file():
        with blob.open("rb") as handle:
            return pickle.load(handle)

    state = _build()
    CACHE.mkdir(parents=True, exist_ok=True)
    with blob.open("wb") as handle:
        pickle.dump(state, handle)
    return state


_state = _load()

X_train = _state["X_train"]
X_test = _state["X_test"]
y_train = _state["y_train"]
y_test = _state["y_test"]
feature_names = _state["feature_names"]
teacher_scores = _state["teacher_scores"]
teacher_test_scores = _state["teacher_test_scores"]
whitebox = _state["whitebox"]
fidelity = _state["fidelity"]
latent = _state["latent"]
bands = _state["bands"]

ARTIFACT_PATH = str(CACHE / f"decision-{STAMP}.json")
artifact = load_artifact(ARTIFACT_PATH)

#: One applicant, used wherever an example needs a single row.
row = X_test[0].tolist()

Selecting

Compiling

Bands

Deciding

Scorecards

Tuning

Validation & governance

Export

Plots

Monitoring