Spaces:
Sleeping
Sleeping
File size: 5,788 Bytes
6a1cba7 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 | from __future__ import annotations
import argparse
import json
import sys
from collections import Counter
from datetime import date
from pathlib import Path
from typing import Any
ROOT = Path(__file__).resolve().parents[1]
if str(ROOT) not in sys.path:
sys.path.insert(0, str(ROOT))
from stem_ai import __version__
from stem_ai.scanner import audit_repository
DEFAULT_BASELINE = ROOT / "audits/benchmark-v1.5/local-10-v1.5.4-stage3-3tier-impact"
DEFAULT_OUT = ROOT / "audits/benchmark-v1.5/local-10-v1.5.6-ca-fp-impact"
def main() -> None:
parser = argparse.ArgumentParser(description="Run local-10 and compare v1.5.6 CA false-positive impact.")
parser.add_argument("--baseline", type=Path, default=DEFAULT_BASELINE)
parser.add_argument("--out", type=Path, default=DEFAULT_OUT)
args = parser.parse_args()
args.out.mkdir(parents=True, exist_ok=True)
manifest = json.loads((args.baseline / "benchmark_manifest.json").read_text(encoding="utf-8"))
baseline_records = _load_jsonl(args.baseline / "benchmark_results.jsonl")
baseline_by_repo = {r["repo"]: r for r in baseline_records}
records = []
for repo in manifest["repos"]:
result = audit_repository(Path(repo["local_path"]))
records.append(_record(repo, baseline_by_repo[repo["repo"]], result))
_write_json(args.out / "benchmark_manifest.json", _manifest(manifest, records))
_write_jsonl(args.out / "benchmark_results.jsonl", records)
(args.out / "comparison_summary.md").write_text(_summary(records), encoding="utf-8")
print(f"records={len(records)}")
print(args.out)
def _load_jsonl(path: Path) -> list[dict[str, Any]]:
return [json.loads(line) for line in path.read_text(encoding="utf-8").splitlines() if line.strip()]
def _record(repo: dict[str, Any], baseline: dict[str, Any], result: dict[str, Any]) -> dict[str, Any]:
old_score = int(baseline["v1_5_4_score"])
new_score = int(result["score"]["final_score"])
old_tier = baseline["v1_5_4_tier"]
new_tier = result["score"]["formal_tier"]
return {
"repo": repo["repo"],
"local_name": repo["local_name"],
"local_path": repo["local_path"],
"commit": repo["commit"],
"baseline_version": baseline["stem_ai_version"],
"stem_ai_version": __version__,
"baseline_score": old_score,
"v1_5_6_score": new_score,
"score_delta": new_score - old_score,
"baseline_tier": old_tier,
"v1_5_6_tier": new_tier,
"tier_delta": _tier_index(new_tier) - _tier_index(old_tier),
"baseline_ca_severity": baseline.get("ca_severity"),
"v1_5_6_ca_severity": result["classification"]["ca_severity"],
"baseline_score_cap": baseline.get("score_cap"),
"v1_5_6_score_cap": result["classification"]["score_cap"],
"v1_5_6_has_boundary": result["classification"]["has_explicit_clinical_boundary"],
"v1_5_6_t0_hard_floor": result["classification"]["t0_hard_floor"],
}
def _tier_index(tier: str) -> int:
for i, prefix in enumerate(["T0", "T1", "T2", "T3", "T4"]):
if tier.startswith(prefix):
return i
return -1
def _manifest(base_manifest: dict[str, Any], records: list[dict[str, Any]]) -> dict[str, Any]:
return {
"schema_version": "stem-bio-ai-benchmark-v1.5-local10-ca-fp-impact",
"generated_at": date.today().isoformat(),
"stem_ai_version": __version__,
"benchmark_type": "local10_ca_false_positive_impact",
"baseline_comparison": str(DEFAULT_BASELINE.relative_to(ROOT)).replace("\\", "/"),
"repo_count": len(records),
"repos": base_manifest["repos"],
}
def _summary(records: list[dict[str, Any]]) -> str:
changed = [r for r in records if r["score_delta"] or r["tier_delta"]]
tier_changes = [r for r in records if r["tier_delta"]]
score_deltas = [r["score_delta"] for r in records]
tier_counts = Counter(r["v1_5_6_tier"] for r in records)
lines = [
"# Local-10 v1.5.6 CA False-Positive Impact",
"",
f"- Records: {len(records)}",
f"- Score/tier changes vs v1.5.4 baseline: {len(changed)}",
f"- Tier changes vs v1.5.4 baseline: {len(tier_changes)}",
f"- Score delta range: {min(score_deltas)} to {max(score_deltas)}",
f"- Mean score delta: {sum(score_deltas) / len(score_deltas):.1f}",
f"- v1.5.6 tier distribution: {dict(tier_counts)}",
"",
"## Interpretation",
"",
"- ClawBio gains score within T2 because explicit non-medical-device and no-diagnosis boundary language is now recognized.",
"- BioClaw moves from T0 to T1 because workspace triage is no longer treated as patient triage.",
"- No repository moves into T3/T4 from this precision patch.",
"",
"| Repo | Score | Tier | CA | Cap | Boundary | T0 floor |",
"|---|---:|---|---|---|---|---|",
]
for r in records:
score = f"{r['baseline_score']} -> {r['v1_5_6_score']} ({r['score_delta']:+d})"
tier = f"{r['baseline_tier']} -> {r['v1_5_6_tier']}"
ca = f"{r['baseline_ca_severity']} -> {r['v1_5_6_ca_severity']}"
cap = f"{r['baseline_score_cap']} -> {r['v1_5_6_score_cap']}"
lines.append(
f"| {r['local_name']} | {score} | {tier} | {ca} | {cap} | "
f"{r['v1_5_6_has_boundary']} | {r['v1_5_6_t0_hard_floor']} |"
)
return "\n".join(lines) + "\n"
def _write_json(path: Path, data: dict[str, Any]) -> None:
path.write_text(json.dumps(data, indent=2), encoding="utf-8")
def _write_jsonl(path: Path, rows: list[dict[str, Any]]) -> None:
path.write_text("".join(json.dumps(row) + "\n" for row in rows), encoding="utf-8")
if __name__ == "__main__":
main()
|