{ "model": "DistilBERT (uncased) fine-tuned, last 2 blocks + linear head", "task": "binary-text-classification", "dataset": "Enron-Spam (SetFit/enron_spam), cleaned, stratified 70/15/15 split", "split": "test", "primary_metric": { "name": "f1", "value": 0.9937743190661479 }, "metrics": { "accuracy": 0.9936633663366337, "precision": 0.9930015552099534, "recall": 0.9945482866043613, "f1": 0.9937743190661479, "roc_auc": 0.9997530506249356, "average_precision": 0.9997640777977358 }, "threshold": 0.5, "data": { "n_train": 23565, "n_val": 5050, "n_test": 5050, "spam_share_test": 0.5085, "n_classes": 2 }, "params": 66363649, "train_time_s": null, "device": "cuda", "smoke": false, "source": "converted from the original training run (first version of this project, train.py on a CUDA GPU with AMP, checkpoint of 2026-07-13, best epoch 5); re-evaluated 2026-09-25 on CPU with model.Predictor", "versions": { "torch": "2.14.0+cpu", "transformers": "5.17.0", "tokenizers": "0.23.2", "huggingface_hub": "1.32.0", "safetensors": "0.8.0", "sklearn": "1.9.1", "numpy": "2.5.3" }, "trained_at": "2026-07-13", "confusion_matrix": { "labels": [ "ham", "spam" ], "rows": "true", "cols": "predicted", "counts": [ [ 2464, 18 ], [ 14, 2554 ] ] }, "comparison": { "DistilBERT fine-tuned (deployed)": { "accuracy": 0.9936633663366337, "precision": 0.9930015552099534, "recall": 0.9945482866043613, "f1": 0.9937743190661479, "roc_auc": 0.9997530506249356, "average_precision": 0.9997640777977358, "params": 66363649, "errors": 32 }, "TF-IDF + logistic regression": { "accuracy": 0.9912871287128713, "precision": 0.9880123743232792, "recall": 0.9949376947040498, "f1": 0.9914629414047342, "roc_auc": 0.9992293422297865, "average_precision": 0.999197846216229, "features": 50000, "train_time_s": 25.1, "errors": 44 } }, "history": { "train_loss": [ 0.1758, 0.0276, 0.0151, 0.0089, 0.0056 ], "val_loss": [ 0.0292, 0.0196, 0.0171, 0.0177, 0.0166 ], "val_acc": [ 0.9901, 0.9931, 0.9941, 0.9941, 0.9943 ], "val_f1": [ 0.9903, 0.9932, 0.9942, 0.9942, 0.9944 ], "best_epoch": 5, "source": "log of the original training run (CUDA + AMP)" }, "trainable_params": 14176513, "eval_time_s_cpu": 1005.0, "legacy_reported": { "accuracy": 0.9937, "precision": 0.993, "recall": 0.9945, "f1": 0.9938, "confusion": [ [ 2464, 18 ], [ 14, 2554 ] ] }, "evaluated_at": "2026-09-25", "eval_note": "CPU time for the 5,050 test emails on a shared 20-core desktop (6 threads); an earlier run under heavier load took 2452 s" }