Resolving overlaps in 20 dimensions

Reproducing the TRACED ablation on HDFR-B with the released spdal code

The TRACED paper separates its prediction strategy into two components and tests each one on its own (an ablation study):

Variant Exterior rule (trend) Coincident rule (D4) method in spdal
TRACED ✓ ✓ "overlap-outside"
VHLB-TRA ✓ "outside"
VHLB-D4 ✓ "overlap"
VHLB-Simple "none"

All four variants share the same learned neurons; they differ only in how predict treats ambiguous points. This page reruns the experiment on HDFR-B, a 20-dimensional, two-class stream where the class regions overlap. To keep the page light it uses the first 10,000 of the 100,000 samples, so the paper’s Table 3 values (full stream) are shown for reference only.

Protocol, following the paper and the repository’s experiments/benchmark.py: the data are shuffled; the first 20% (4 of 20 chunks) is used to tune hyperparameters by prequential macro-F1; the remaining 80% (16 chunks, sizes varying by ±25%) is processed test-then-train. Features pass through an incrementally fitted StandardScaler. The evaluation is repeated with 10 seeds. Tuning is done in two steps as in the script: first r, γ and δ, then α, β and m, over the values in Table 2 of the paper. Table 2 does not list r, so the script’s values {1, 2.5, 5} are used.

Tune, then evaluate the four variants
import itertools, warnings
import numpy as np
from joblib import Parallel, delayed
from sklearn.datasets import make_classification
from sklearn.metrics import f1_score
from sklearn.preprocessing import StandardScaler
from spdal import TRACED

warnings.filterwarnings("ignore")

def get_chunk_idx(x, num_chunk=10, size_range=0.5, seed=0):   # from experiments/utils
    total, normal = len(x), len(x) / num_chunk
    step = normal * size_range / 2
    rng = np.random.default_rng(seed)
    idx = [int(rng.choice(np.arange(np.round(normal * i - step), np.round(normal * i + step) + 1)))
           for i in range(1, num_chunk)]
    return [0] + idx + [total]

# HDFR-B is load_data("Synthetic3") in the repository
X, y = make_classification(n_samples=100_000, random_state=125, n_clusters_per_class=1)
INIT_SEED, N_SAMPLES, N_CHUNKS = 3, 10_000, 20                # get_num_chunks_v6: 20 for 10k samples
perm = np.random.default_rng(INIT_SEED).permutation(len(X))
X, y = X[perm][:N_SAMPLES], y[perm][:N_SAMPLES]                # a 10k subset of the shuffled stream
n_tune = int(0.2 * len(X))
X_tune, y_tune, X_eval_all, y_eval_all = X[:n_tune], y[:n_tune], X[n_tune:], y[n_tune:]
n_tune_chunks = max(int(N_CHUNKS * 0.2), 2)
CLASSES = np.unique(y)

# ---- tuning: prequential macro-F1 on the tuning split -------------------------------------
tune_cuts = get_chunk_idx(X_tune, n_tune_chunks, seed=0)[1:]

def tune_score(params):
    scaler, model = StandardScaler(), TRACED(**params)
    first = slice(0, tune_cuts[0])
    scaler.partial_fit(X_tune[first])
    model.fit(scaler.transform(X_tune[first]), y_tune[first])
    scores = []
    for a, b in zip(tune_cuts[:-1], tune_cuts[1:]):
        Xi, yi = X_tune[a:b], y_tune[a:b]
        scores.append(f1_score(yi, model.predict(scaler.transform(Xi)), average="macro"))
        scaler.partial_fit(Xi)
        model.partial_fit(scaler.transform(Xi), yi, classes=CLASSES)
    return float(np.mean(scores))

def grid_search(grid):
    combos = [dict(zip(grid, values)) for values in itertools.product(*grid.values())]
    scores = Parallel(n_jobs=-1)(delayed(tune_score)(c) for c in combos)
    return combos[int(np.argmax(scores))], max(scores), len(combos)

TABLE2 = {"alpha": [0, 0.2, 0.4, 0.6, 0.8, 1], "beta": [0, 0.005, 0.01, 0.05, 0.1, 0.2],
          "reduce_dims": [0, 1, 2, 3], "r": [1, 2.5, 5],
          "width_parameter": [0, 0.2, 0.4, 0.6, 0.8, 1], "delta": [1, 2, 4]}
step1, _, n1 = grid_search({"method": ["overlap-outside"], **{k: TABLE2[k] for k in ("r", "width_parameter", "delta")}})
best, best_score, n2 = grid_search({**{k: [v] for k, v in step1.items()},
                                    **{k: TABLE2[k] for k in ("alpha", "beta", "reduce_dims")}})

# ---- evaluation: one model per seed, four prediction variants ------------------------------
VARIANTS = {"TRACED": "overlap-outside", "VHLB-TRA": "outside", "VHLB-D4": "overlap", "VHLB-Simple": "none"}
SEEDS = [INIT_SEED ** (i + 1) for i in range(10)]
n_eval_chunks = N_CHUNKS - n_tune_chunks

def evaluate(seed):
    p = np.random.default_rng(seed).permutation(len(X_eval_all))
    Xe, ye = X_eval_all[p], y_eval_all[p]
    cuts = get_chunk_idx(Xe, n_eval_chunks, seed=seed)
    scaler, model = StandardScaler(), TRACED(**best)
    y_true, y_pred = [], {v: [] for v in VARIANTS}
    curve = {v: [] for v in VARIANTS}
    handled = {"test": 0, "coincident": 0, "exterior": 0}   # points each rule acts on (full TRACED)
    for i in range(len(cuts) - 2):
        Xa, ya = Xe[cuts[i]:cuts[i + 1]], ye[cuts[i]:cuts[i + 1]]
        Xb, yb = Xe[cuts[i + 1]:cuts[i + 2]], ye[cuts[i + 1]:cuts[i + 2]]
        scaler.partial_fit(Xa)
        model.partial_fit(scaler.transform(Xa), ya, classes=CLASSES)
        y_true.append(yb)
        model.count_overlap = model.count_outside = 0
        for name, method in VARIANTS.items():
            model.method = method
            y_pred[name].append(model.predict(scaler.transform(Xb)))
            if name == "TRACED":
                handled["test"] += len(Xb)
                handled["coincident"] += int(model.count_overlap)
                handled["exterior"] += int(model.count_outside)
            curve[name].append(float(np.mean(np.concatenate(y_pred[name]) == np.concatenate(y_true))))
    yt = np.concatenate(y_true)
    return handled, {name: {"acc": float(np.mean(np.concatenate(pr) == yt)),
                   "f1": float(f1_score(yt, np.concatenate(pr), average="macro")),
                   "curve": curve[name]}
            for name, pr in y_pred.items()}

outputs = Parallel(n_jobs=-1)(delayed(evaluate)(s) for s in SEEDS)
results = [res for _, res in outputs]
handled = {k: sum(h[k] for h, _ in outputs) for k in ("test", "coincident", "exterior")}

summary = [{"variant": v,
            "acc_mean": 100 * np.mean([r[v]["acc"] for r in results]), "acc_sd": 100 * np.std([r[v]["acc"] for r in results]),
            "f1_mean": 100 * np.mean([r[v]["f1"] for r in results]), "f1_sd": 100 * np.std([r[v]["f1"] for r in results])}
           for v in VARIANTS]
PAPER = {"TRACED": (96.35, 0.19, 96.35, 0.18), "VHLB-TRA": (95.69, 0.61, 95.69, 0.58),
         "VHLB-D4": (96.35, 0.19, 96.34, 0.18), "VHLB-Simple": (95.68, 0.61, 95.68, 0.58)}
curves = [{"variant": v, "chunk": i + 1, "ca": 100 * float(np.mean([r[v]["curve"][i] for r in results]))}
          for v in VARIANTS for i in range(len(results[0][v]["curve"]))]

ojs_define(hdfr_summary=summary, hdfr_curves=curves,
           hdfr_paper=[{"variant": v, "acc_mean": a, "acc_sd": asd, "f1_mean": f, "f1_sd": fsd}
                       for v, (a, asd, f, fsd) in PAPER.items()],
           hdfr_best={k: (v.item() if isinstance(v, np.generic) else v) for k, v in best.items()},
           hdfr_tuning={"score": best_score, "n_configs": n1 + n2},
           hdfr_handled=handled)

Results

Reading the result

The absolute numbers are lower than in the paper, most likely because this page learns from a tenth of the data.