Text Classification
Transformers
Safetensors
English
modernbert
episodic-ingestion-compiler
grouped-softmax-ranker
field-event-ranker
mixed-mode-training
semantic-reasoning-labels
v2-labels
text-embeddings-inference
Instructions to use Avifenesh/episodic-ingestion-modernbert-field-event-ranker-mixed-v2-perf-h4 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use Avifenesh/episodic-ingestion-modernbert-field-event-ranker-mixed-v2-perf-h4 with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-classification", model="Avifenesh/episodic-ingestion-modernbert-field-event-ranker-mixed-v2-perf-h4")# Load model directly from transformers import AutoTokenizer, AutoModelForSequenceClassification tokenizer = AutoTokenizer.from_pretrained("Avifenesh/episodic-ingestion-modernbert-field-event-ranker-mixed-v2-perf-h4") model = AutoModelForSequenceClassification.from_pretrained("Avifenesh/episodic-ingestion-modernbert-field-event-ranker-mixed-v2-perf-h4", device_map="auto") - Notebooks
- Google Colab
- Kaggle
| { | |
| "device": "cuda", | |
| "eval": "data/tmp/mixed_mode_eval.v2.field_decision.jsonl", | |
| "eval_candidate_pairs": 21986, | |
| "eval_groups": 1900, | |
| "eval_rows": 220, | |
| "last_loss": 0.08536958694458008, | |
| "max_steps": 160, | |
| "model": "answerdotai/ModernBERT-base", | |
| "output_dir": "models/modernbert-field-event-ranker-mixed-v2-perf-H4", | |
| "peak_allocated_gib": 4.076, | |
| "peak_reserved_gib": 4.17, | |
| "positive_train_pairs": 12854, | |
| "skip_save_model": false, | |
| "source_rank": { | |
| "mean_expected_source_rank": 3.0918580375782883, | |
| "mean_reciprocal_rank": 0.5255961949014605, | |
| "median_expected_source_rank": 2, | |
| "missing_prediction_pairs": 0, | |
| "per_field": { | |
| "action_causality": { | |
| "mean_rank": 3.00709219858156, | |
| "mean_reciprocal_rank": 0.48304626815265134, | |
| "missing_prediction_pairs": 0, | |
| "ranked_pairs": 141, | |
| "source_pairs": 141, | |
| "top_k_hits": { | |
| "1": 34, | |
| "2": 61, | |
| "3": 92, | |
| "5": 127 | |
| }, | |
| "top_k_recall": { | |
| "1": 0.24113475177304963, | |
| "2": 0.4326241134751773, | |
| "3": 0.6524822695035462, | |
| "5": 0.900709219858156 | |
| } | |
| }, | |
| "assistant_claims_to_verify": { | |
| "mean_rank": 2.0, | |
| "mean_reciprocal_rank": 0.5, | |
| "missing_prediction_pairs": 0, | |
| "ranked_pairs": 1, | |
| "source_pairs": 1, | |
| "top_k_hits": { | |
| "1": 0, | |
| "2": 1, | |
| "3": 1, | |
| "5": 1 | |
| }, | |
| "top_k_recall": { | |
| "1": 0.0, | |
| "2": 1.0, | |
| "3": 1.0, | |
| "5": 1.0 | |
| } | |
| }, | |
| "attempt_outcome_pairs": { | |
| "mean_rank": 2.3773584905660377, | |
| "mean_reciprocal_rank": 0.5625786163522012, | |
| "missing_prediction_pairs": 0, | |
| "ranked_pairs": 53, | |
| "source_pairs": 53, | |
| "top_k_hits": { | |
| "1": 16, | |
| "2": 31, | |
| "3": 43, | |
| "5": 52 | |
| }, | |
| "top_k_recall": { | |
| "1": 0.3018867924528302, | |
| "2": 0.5849056603773585, | |
| "3": 0.8113207547169812, | |
| "5": 0.9811320754716981 | |
| } | |
| }, | |
| "attempted_actions": { | |
| "mean_rank": 3.756501182033097, | |
| "mean_reciprocal_rank": 0.4687886557994194, | |
| "missing_prediction_pairs": 0, | |
| "ranked_pairs": 846, | |
| "source_pairs": 846, | |
| "top_k_hits": { | |
| "1": 196, | |
| "2": 383, | |
| "3": 531, | |
| "5": 713 | |
| }, | |
| "top_k_recall": { | |
| "1": 0.23167848699763594, | |
| "2": 0.45271867612293143, | |
| "3": 0.6276595744680851, | |
| "5": 0.8427895981087471 | |
| } | |
| }, | |
| "customer_identity": { | |
| "mean_rank": 1.5454545454545454, | |
| "mean_reciprocal_rank": 0.8181818181818182, | |
| "missing_prediction_pairs": 0, | |
| "ranked_pairs": 11, | |
| "source_pairs": 11, | |
| "top_k_hits": { | |
| "1": 8, | |
| "2": 8, | |
| "3": 11, | |
| "5": 11 | |
| }, | |
| "top_k_recall": { | |
| "1": 0.7272727272727273, | |
| "2": 0.7272727272727273, | |
| "3": 1.0, | |
| "5": 1.0 | |
| } | |
| }, | |
| "discarded_options": { | |
| "mean_rank": 3.0, | |
| "mean_reciprocal_rank": 0.375, | |
| "missing_prediction_pairs": 0, | |
| "ranked_pairs": 2, | |
| "source_pairs": 2, | |
| "top_k_hits": { | |
| "1": 0, | |
| "2": 1, | |
| "3": 1, | |
| "5": 2 | |
| }, | |
| "top_k_recall": { | |
| "1": 0.0, | |
| "2": 0.5, | |
| "3": 0.5, | |
| "5": 1.0 | |
| } | |
| }, | |
| "explicit_decisions": { | |
| "mean_rank": 1.0, | |
| "mean_reciprocal_rank": 1.0, | |
| "missing_prediction_pairs": 0, | |
| "ranked_pairs": 1, | |
| "source_pairs": 1, | |
| "top_k_hits": { | |
| "1": 1, | |
| "2": 1, | |
| "3": 1, | |
| "5": 1 | |
| }, | |
| "top_k_recall": { | |
| "1": 1.0, | |
| "2": 1.0, | |
| "3": 1.0, | |
| "5": 1.0 | |
| } | |
| }, | |
| "failed_attempts": { | |
| "mean_rank": 2.5, | |
| "mean_reciprocal_rank": 0.6150793650793651, | |
| "missing_prediction_pairs": 0, | |
| "ranked_pairs": 36, | |
| "source_pairs": 36, | |
| "top_k_hits": { | |
| "1": 15, | |
| "2": 22, | |
| "3": 28, | |
| "5": 34 | |
| }, | |
| "top_k_recall": { | |
| "1": 0.4166666666666667, | |
| "2": 0.6111111111111112, | |
| "3": 0.7777777777777778, | |
| "5": 0.9444444444444444 | |
| } | |
| }, | |
| "initiating_command": { | |
| "mean_rank": 2.5728643216080402, | |
| "mean_reciprocal_rank": 0.667125852929873, | |
| "missing_prediction_pairs": 0, | |
| "ranked_pairs": 199, | |
| "source_pairs": 199, | |
| "top_k_hits": { | |
| "1": 99, | |
| "2": 139, | |
| "3": 160, | |
| "5": 178 | |
| }, | |
| "top_k_recall": { | |
| "1": 0.49748743718592964, | |
| "2": 0.6984924623115578, | |
| "3": 0.8040201005025126, | |
| "5": 0.8944723618090452 | |
| } | |
| }, | |
| "invalidation_hints": { | |
| "mean_rank": 1.5, | |
| "mean_reciprocal_rank": 0.75, | |
| "missing_prediction_pairs": 0, | |
| "ranked_pairs": 2, | |
| "source_pairs": 2, | |
| "top_k_hits": { | |
| "1": 1, | |
| "2": 2, | |
| "3": 2, | |
| "5": 2 | |
| }, | |
| "top_k_recall": { | |
| "1": 0.5, | |
| "2": 1.0, | |
| "3": 1.0, | |
| "5": 1.0 | |
| } | |
| }, | |
| "next_actions": { | |
| "mean_rank": 2.875, | |
| "mean_reciprocal_rank": 0.5649440836940836, | |
| "missing_prediction_pairs": 0, | |
| "ranked_pairs": 32, | |
| "source_pairs": 32, | |
| "top_k_hits": { | |
| "1": 11, | |
| "2": 19, | |
| "3": 23, | |
| "5": 29 | |
| }, | |
| "top_k_recall": { | |
| "1": 0.34375, | |
| "2": 0.59375, | |
| "3": 0.71875, | |
| "5": 0.90625 | |
| } | |
| }, | |
| "non_promotable_context": { | |
| "mean_rank": 3.0, | |
| "mean_reciprocal_rank": 0.3333333333333333, | |
| "missing_prediction_pairs": 0, | |
| "ranked_pairs": 2, | |
| "source_pairs": 2, | |
| "top_k_hits": { | |
| "1": 0, | |
| "2": 0, | |
| "3": 2, | |
| "5": 2 | |
| }, | |
| "top_k_recall": { | |
| "1": 0.0, | |
| "2": 0.0, | |
| "3": 1.0, | |
| "5": 1.0 | |
| } | |
| }, | |
| "observed_outcomes": { | |
| "mean_rank": 3.0272189349112426, | |
| "mean_reciprocal_rank": 0.4855734009580153, | |
| "missing_prediction_pairs": 0, | |
| "ranked_pairs": 845, | |
| "source_pairs": 845, | |
| "top_k_hits": { | |
| "1": 199, | |
| "2": 393, | |
| "3": 557, | |
| "5": 754 | |
| }, | |
| "top_k_recall": { | |
| "1": 0.23550295857988165, | |
| "2": 0.4650887573964497, | |
| "3": 0.659171597633136, | |
| "5": 0.8923076923076924 | |
| } | |
| }, | |
| "outcome_of_latest_attempt": { | |
| "mean_rank": 1.6685082872928176, | |
| "mean_reciprocal_rank": 0.7629439621152327, | |
| "missing_prediction_pairs": 0, | |
| "ranked_pairs": 181, | |
| "source_pairs": 181, | |
| "top_k_hits": { | |
| "1": 108, | |
| "2": 149, | |
| "3": 171, | |
| "5": 179 | |
| }, | |
| "top_k_recall": { | |
| "1": 0.5966850828729282, | |
| "2": 0.8232044198895028, | |
| "3": 0.9447513812154696, | |
| "5": 0.988950276243094 | |
| } | |
| }, | |
| "payment_or_warranty_detail": { | |
| "mean_rank": 3.5, | |
| "mean_reciprocal_rank": 0.35, | |
| "missing_prediction_pairs": 0, | |
| "ranked_pairs": 2, | |
| "source_pairs": 2, | |
| "top_k_hits": { | |
| "1": 0, | |
| "2": 1, | |
| "3": 1, | |
| "5": 2 | |
| }, | |
| "top_k_recall": { | |
| "1": 0.0, | |
| "2": 0.5, | |
| "3": 0.5, | |
| "5": 1.0 | |
| } | |
| }, | |
| "product_name": { | |
| "mean_rank": 1.0, | |
| "mean_reciprocal_rank": 1.0, | |
| "missing_prediction_pairs": 0, | |
| "ranked_pairs": 4, | |
| "source_pairs": 4, | |
| "top_k_hits": { | |
| "1": 4, | |
| "2": 4, | |
| "3": 4, | |
| "5": 4 | |
| }, | |
| "top_k_recall": { | |
| "1": 1.0, | |
| "2": 1.0, | |
| "3": 1.0, | |
| "5": 1.0 | |
| } | |
| }, | |
| "recent_error": { | |
| "mean_rank": 1.8620689655172413, | |
| "mean_reciprocal_rank": 0.7425287356321838, | |
| "missing_prediction_pairs": 0, | |
| "ranked_pairs": 29, | |
| "source_pairs": 29, | |
| "top_k_hits": { | |
| "1": 17, | |
| "2": 23, | |
| "3": 25, | |
| "5": 28 | |
| }, | |
| "top_k_recall": { | |
| "1": 0.5862068965517241, | |
| "2": 0.7931034482758621, | |
| "3": 0.8620689655172413, | |
| "5": 0.9655172413793104 | |
| } | |
| }, | |
| "resolved_context": { | |
| "mean_rank": 4.0, | |
| "mean_reciprocal_rank": 0.25, | |
| "missing_prediction_pairs": 0, | |
| "ranked_pairs": 1, | |
| "source_pairs": 1, | |
| "top_k_hits": { | |
| "1": 0, | |
| "2": 0, | |
| "3": 0, | |
| "5": 1 | |
| }, | |
| "top_k_recall": { | |
| "1": 0.0, | |
| "2": 0.0, | |
| "3": 0.0, | |
| "5": 1.0 | |
| } | |
| }, | |
| "touched_files": { | |
| "mean_rank": 3.0, | |
| "mean_reciprocal_rank": 0.3333333333333333, | |
| "missing_prediction_pairs": 0, | |
| "ranked_pairs": 1, | |
| "source_pairs": 1, | |
| "top_k_hits": { | |
| "1": 0, | |
| "2": 0, | |
| "3": 1, | |
| "5": 1 | |
| }, | |
| "top_k_recall": { | |
| "1": 0.0, | |
| "2": 0.0, | |
| "3": 1.0, | |
| "5": 1.0 | |
| } | |
| }, | |
| "transaction_reference": { | |
| "mean_rank": 2.6, | |
| "mean_reciprocal_rank": 0.4666666666666666, | |
| "missing_prediction_pairs": 0, | |
| "ranked_pairs": 5, | |
| "source_pairs": 5, | |
| "top_k_hits": { | |
| "1": 1, | |
| "2": 1, | |
| "3": 5, | |
| "5": 5 | |
| }, | |
| "top_k_recall": { | |
| "1": 0.2, | |
| "2": 0.2, | |
| "3": 1.0, | |
| "5": 1.0 | |
| } | |
| }, | |
| "unsupported_hypotheses": { | |
| "mean_rank": 3.0, | |
| "mean_reciprocal_rank": 0.3333333333333333, | |
| "missing_prediction_pairs": 0, | |
| "ranked_pairs": 1, | |
| "source_pairs": 1, | |
| "top_k_hits": { | |
| "1": 0, | |
| "2": 0, | |
| "3": 1, | |
| "5": 1 | |
| }, | |
| "top_k_recall": { | |
| "1": 0.0, | |
| "2": 0.0, | |
| "3": 1.0, | |
| "5": 1.0 | |
| } | |
| } | |
| }, | |
| "rows": 220, | |
| "selected_fields": 997, | |
| "source_pairs": 2395, | |
| "top_k_hits": { | |
| "1": 710, | |
| "2": 1239, | |
| "3": 1660, | |
| "5": 2127 | |
| }, | |
| "top_k_source_pair_recall": { | |
| "1": 0.2964509394572025, | |
| "2": 0.5173277661795407, | |
| "3": 0.6931106471816284, | |
| "5": 0.8881002087682672 | |
| }, | |
| "worst_expected_source_ranks": [ | |
| { | |
| "expected_event": { | |
| "command": "python - <<'PY'\nimport json\nfrom pathlib import Path\nfor model in ['qwen3.6-35b-a3b-nvfp4','qwen3.6-27b-text-nvfp4-mtp']:\n p=Path('/home/avifenesh/projects/aviary/models')/model\n print('MODEL', p)\n for f in ['config.json','quantization_config.json','quantize_config.json','generation_config.json']:\n q=p/f\n if q.exists():\n data=json.loads(q.read_text())\n print(f, {k:data.get(k) for k in ['model_type','architectures','quantization_config','kv_lora_rank','num_hidden_layers','num_attention_heads','num_key_value_heads','max_position_embeddings','sliding_window','use_mtp','num_nextn_predict_layers'] if k in data})\nPY", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "python - <<'PY'\nimport json\nfrom pathlib import Path\nfor model in ['qwen3.6-35b-a3b-nvfp4','qwen3.6-27b-text-nvfp4-mtp']:\n p=Path('/home/avifenesh/projects/aviary/models')/model...", | |
| "tool_name": "exec_command" | |
| }, | |
| "expected_probability": 0.28537195920944214, | |
| "expected_source_event_id": "t6", | |
| "field": "attempted_actions", | |
| "rank": 20, | |
| "reciprocal_rank": 0.05, | |
| "row_id": "pta-codex-rollout-2026-04-30T19-43-19-019ddf45-fdda-7fb2-8764-8162ba11dc24::w14", | |
| "top_candidates": [ | |
| { | |
| "event": { | |
| "command": "{\"session_id\": 92479, \"chars\": \"\\u0003\", \"yield_time_ms\": 1000, \"max_output_tokens\": 20000}", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "{\"session_id\": 92479, \"chars\": \"\\u0003\", \"yield_time_ms\": 1000, \"max_output_tokens\": 20000}", | |
| "tool_name": "write_stdin" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t3", | |
| "support_probability": 0.9996079802513123 | |
| }, | |
| { | |
| "event": { | |
| "command": "{\"session_id\": 1703, \"chars\": \"\", \"yield_time_ms\": 1000, \"max_output_tokens\": 30000}", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "{\"session_id\": 1703, \"chars\": \"\", \"yield_time_ms\": 1000, \"max_output_tokens\": 30000}", | |
| "tool_name": "write_stdin" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t12", | |
| "support_probability": 0.9992678761482239 | |
| }, | |
| { | |
| "event": { | |
| "command": "{\"session_id\": 1703, \"chars\": \"\", \"yield_time_ms\": 1000, \"max_output_tokens\": 6000}", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "{\"session_id\": 1703, \"chars\": \"\", \"yield_time_ms\": 1000, \"max_output_tokens\": 6000}", | |
| "tool_name": "write_stdin" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t17", | |
| "support_probability": 0.9992678761482239 | |
| }, | |
| { | |
| "event": { | |
| "command": "nvidia-smi --query-gpu=memory.used,memory.free,utilization.gpu --format=csv,noheader,nounits && free -h && ss -ltnp sport = :8000", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "nvidia-smi --query-gpu=memory.used,memory.free,utilization.gpu --format=csv,noheader,nounits && free -h && ss -ltnp sport = :8000", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t13", | |
| "support_probability": 0.9985895752906799 | |
| }, | |
| { | |
| "event": { | |
| "command": "nvidia-smi --query-gpu=name,memory.used,memory.free,utilization.gpu --format=csv,noheader,nounits && free -h && ss -ltnp 'sport = :8000' || true", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "nvidia-smi --query-gpu=name,memory.used,memory.free,utilization.gpu --format=csv,noheader,nounits && free -h && ss -ltnp 'sport = :8000' || true", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t18", | |
| "support_probability": 0.998245358467102 | |
| } | |
| ] | |
| }, | |
| { | |
| "expected_event": { | |
| "command": "python3 - <<'PY'\nfrom pathlib import Path\nimport yaml\nroot=Path('.')\ncanonical=set()\nfor p in Path('data/evals').glob('*/scorecards.yml'):\n d=yaml.safe_load(p.read_text()) or {}\n for c in d.get('scorecards') or []:\n if isinstance(c,dict) and c.get('result_path'):\n canonical.add(str(Path(c['result_path']).resolve()))\nprint('canonical result paths',len(canonical))\nnotcanon=[]; incanon=[]\nfor p in Path('data/runs').rglob('judge-scorecard-*.yml'):\n d=yaml.safe_load(p.read_text()) or {}\n rp=d.get('result_path')\n if not rp:\n # common default result.md in same dir\n cand=p.parent/'result.md'\n if cand.exists(): rp=str(cand)\n key=str(Path(rp).resolve()) if rp else ''\n (incanon if key in canonical else notcanon).append((p,key))\nprint('judge ", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "python3 - <<'PY'\nfrom pathlib import Path\nimport yaml\nroot=Path('.')\ncanonical=set()\nfor p in Path('data/evals').glob('*/scorecards.yml'):\n d=yaml.safe_load(p.read_text()) or...", | |
| "tool_name": "exec_command" | |
| }, | |
| "expected_probability": 0.4863315224647522, | |
| "expected_source_event_id": "t17", | |
| "field": "attempted_actions", | |
| "rank": 20, | |
| "reciprocal_rank": 0.05, | |
| "row_id": "pta-codex-rollout-2026-04-30T19-43-19-019ddf45-fdda-7fb2-8764-8162ba11dc24::w48", | |
| "top_candidates": [ | |
| { | |
| "event": { | |
| "command": "find data/runs -name 'codex-scorecards-draft.yml' -print | wc -l && find data/runs -name 'codex-scorecards-draft.yml' -print", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "find data/runs -name 'codex-scorecards-draft.yml' -print | wc -l && find data/runs -name 'codex-scorecards-draft.yml' -print", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t4", | |
| "support_probability": 0.999783456325531 | |
| }, | |
| { | |
| "event": { | |
| "command": "sed -n '1,240p' data/evals/judge-review-queues/20260504-first10.txt", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "sed -n '1,240p' data/evals/judge-review-queues/20260504-first10.txt", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t12", | |
| "support_probability": 0.9996646642684937 | |
| }, | |
| { | |
| "event": { | |
| "command": "find data/evals/judge-review-queues -type f -maxdepth 4 -print 2>/dev/null | sort && printf '\\nreview markers\\n' && rg -n \"review_status|avi_corrected|pending|draft|accepted|rejected\" data/runs data/evals/judge-review-queues -g 'codex-scorecards-draft.yml' -g '*.yml'", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "find data/evals/judge-review-queues -type f -maxdepth 4 -print 2>/dev/null | sort && printf '\\nreview markers\\n' && rg -n \"review_status|avi_corrected|pending|draft|accepted|rej...", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t5", | |
| "support_probability": 0.998245358467102 | |
| }, | |
| { | |
| "event": { | |
| "command": "python3 - <<'PY'\nfrom pathlib import Path\nimport yaml\nfor p in sorted(Path('data/runs').rglob('codex-scorecards-draft.yml')):\n d=yaml.safe_load(p.read_text()) or {}\n cards=d.get('scorecards') or []\n print(f'{p}\\tfile_status={d.get(\"review_status\")}\\tcards={len(cards)}\\tcard_statuses={sorted({str(c.get(\"review_status\", d.get(\"review_status\", \"\"))) for c in cards if isinstance(c,dict)})}')\nPY", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "python3 - <<'PY'\nfrom pathlib import Path\nimport yaml\nfor p in sorted(Path('data/runs').rglob('codex-scorecards-draft.yml')):\n d=yaml.safe_load(p.read_text()) or {}\n cards...", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t11", | |
| "support_probability": 0.9921841025352478 | |
| }, | |
| { | |
| "event": { | |
| "command": null, | |
| "is_context_tool_echo": false, | |
| "role": "tool_result", | |
| "success": true, | |
| "text": "Chunk ID: c00386\nWall time: 0.0000 seconds\nProcess exited with code 0\nOriginal token count: 499\nOutput:\ncanonical result paths 46\njudge draft files already canonical by result_p...", | |
| "tool_name": null | |
| }, | |
| "label": 0, | |
| "source_event_id": "r19", | |
| "support_probability": 0.9759464263916016 | |
| } | |
| ] | |
| }, | |
| { | |
| "expected_event": { | |
| "command": "*** Begin Patch\n*** Update File: /home/avifenesh/projects/aviary/data/evals/judge-review-status.yml\n@@\n - result_path: /home/avifenesh/projects/aviary/data/runs/session_summarize_for_avi/20260504-memory-gym-variety/qwen36-27b-autoround-64k/round-13-agent-sh-multirepo-maintenance/result.md\n- status: pending_avi_judge_review\n- note: Recent judge-draft batch; keep waiting for Avi unless verbal correction already exists.\n+ status: reviewed_with_known_owner_correction\n+ reviewed_at: '2026-05-04'\n+ note: Avi correction is score 2; the judges overranked a mistaken, failed, not-helpful session as positive.\n - result_path: /home/avifenesh/projects/aviary/data/runs/session_summarize_for_avi/20260504-memory-gym-variety/qwen36-27b-autoround-64k/round-14-agnix-next-task-workflow-policy/result.md\n- status: pending_avi_judge_review\n- note: Recent judge-draft batch; keep waiting for Avi unless verbal correction already exists.\n+ status: reviewed_with_known_owner_correction\n+ reviewed_at: '2026-05-04'\n+ note: Avi correction is score 3; not here, not there.\n@@\n - result_path: /home/avifenesh/projects/aviary/data/runs/session_summarize_for_avi/20260504-memory-gym-variety/qwen36-27b-autoround-64k/round-16-session-pairs-signal-sample/result.md\n- status: pending_avi_judge_review\n- note: Recent judge-draft batch; keep waiting for Avi unless verbal correction already exists.\n+ status: reviewed_excluded\n+ reviewed_at: '2026-05-04'\n+ note: Avi correction is exclude if leaked/contaminated; do not train on this example.\n*** End Patch\n", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "apply_patch: *** Begin Patch\n*** Update File: /home/avifenesh/projects/aviary/data/evals/judge-review-status.yml\n@@\n - result_path: /home/avifenesh/projects/aviary/data/runs/ses...", | |
| "tool_name": "apply_patch" | |
| }, | |
| "expected_probability": 0.5248817801475525, | |
| "expected_source_event_id": "t9", | |
| "field": "attempted_actions", | |
| "rank": 20, | |
| "reciprocal_rank": 0.05, | |
| "row_id": "pta-codex-rollout-2026-04-30T19-43-19-019ddf45-fdda-7fb2-8764-8162ba11dc24::w53", | |
| "top_candidates": [ | |
| { | |
| "event": { | |
| "command": "python3 scripts/judge-review-queue.py --write data/evals/judge-review-queues/current-pending.txt", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "python3 scripts/judge-review-queue.py --write data/evals/judge-review-queues/current-pending.txt", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t12", | |
| "support_probability": 0.9999039173126221 | |
| }, | |
| { | |
| "event": { | |
| "command": "python3 -m py_compile scripts/judge-review-queue.py scripts/review-judge-drafts.py scripts/build-scorecard-sft-dataset.py && cargo test eval_score --quiet", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "python3 -m py_compile scripts/judge-review-queue.py scripts/review-judge-drafts.py scripts/build-scorecard-sft-dataset.py && cargo test eval_score --quiet", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t14", | |
| "support_probability": 0.9998910427093506 | |
| }, | |
| { | |
| "event": { | |
| "command": "nl -ba data/evals/session_summarize_for_avi/scorecards.yml | sed -n '1020,1235p'", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "nl -ba data/evals/session_summarize_for_avi/scorecards.yml | sed -n '1020,1235p'", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t3", | |
| "support_probability": 0.9996540546417236 | |
| }, | |
| { | |
| "event": { | |
| "command": "nl -ba data/evals/session_summarize_for_avi/scorecards.yml | sed -n '1230,1265p' && nl -ba data/evals/judge-review-status.yml | sed -n '70,140p'", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "nl -ba data/evals/session_summarize_for_avi/scorecards.yml | sed -n '1230,1265p' && nl -ba data/evals/judge-review-status.yml | sed -n '70,140p'", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t5", | |
| "support_probability": 0.998793363571167 | |
| }, | |
| { | |
| "event": { | |
| "command": "python3 scripts/build-scorecard-sft-dataset.py --lane session_summarize_for_avi --out data/training/session_summarize_for_avi/scorecard_sft_avi_verified.jsonl && python3 scripts/build-scorecard-sft-dataset.py --lane session_summarize_for_avi --out data/training/session_summarize_for_avi/scorecard_sft_with_drafts.jsonl --include-drafts", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "python3 scripts/build-scorecard-sft-dataset.py --lane session_summarize_for_avi --out data/training/session_summarize_for_avi/scorecard_sft_avi_verified.jsonl && python3 scripts...", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t13", | |
| "support_probability": 0.9893474578857422 | |
| } | |
| ] | |
| }, | |
| { | |
| "expected_event": { | |
| "command": "*** Begin Patch\n*** Add File: /home/avifenesh/projects/episodic-ingestion-compiler/data/examples/seed.jsonl\n+{\"id\":\"seed-0001-snake-tdd-validation-gap\",\"input\":{\"source_kind\":\"agent_trace\",\"source_uri\":\"pi://local/session/snake-tdd\",\"raw\":{\"source_events\":[{\"id\":\"u1\",\"role\":\"user\",\"timestamp\":\"2026-05-02T15:40:00Z\",\"text\":\"build a snake game that run on my machine locally without using browser, write the tests upfront\"},{\"id\":\"a1\",\"role\":\"assistant\",\"timestamp\":\"2026-05-02T15:41:00Z\",\"text\":\"I will write tests first, then implement the Snake logic and run cargo test.\"},{\"id\":\"t1\",\"role\":\"tool_call\",\"timestamp\":\"2026-05-02T15:42:00Z\",\"tool_id\":\"write_file\",\"file_paths\":[\"snake_game/test_snake.py\"],\"text\":\"create test suite for movement, food, wall collision, self collision\"},{\"id\":\"t2\",\"role\":\"tool_call\",\"timestamp\":\"2026-05-02T15:43:00Z\",\"tool_id\":\"run_command\",\"command_id\":\"cmd-cargo-test-1\",\"text\":\"cargo test\"},{\"id\":\"r2\",\"role\":\"tool_result\",\"timestamp\":\"2026-05-02T15:43:20Z\",\"tool_id\":\"run_command\",\"command_id\":\"cmd-cargo-test-1\",\"text\":\"tests passed\"},{\"id\":\"a2\",\"role\":\"assistant\",\"timestamp\":\"2026-05-02T15:44:00Z\",\"text\":\"The implementation is complete and tests pass.\"},{\"id\":\"u2\",\"role\":\"user\",\"timestamp\":\"2026-05-02T15:45:00Z\",\"text\":\"did you check also that the game work or just the tests?\"},{\"id\":\"r3\",\"role\":\"tool_result\",\"timestamp\":\"2026-05-02T15:46:00Z\",\"tool_id\":\"run_command\",\"command_id\":\"cmd-run-game-1\",\"text\":\"runtime smoke failed: game window starts but input does not begin movement\"}]}},\"expected\":{\"episode_hint\":\"snake-local-game-tdd\",\"objective_hint\":\"Build a local non-browser Snake game with tests written before implementation and verify both tests and playable runtime behavior.\",\"events\":[{\"event_id\":\"e1\",\"event_type\":\"user_request\",\"summary\":\"User requested a local non-browser Snake game and explicitly required tests upfront.\",\"provenance\":{\"source_event_ids\":[\"u1\"],\"timestamps\":[\"2026-05-02T15:40:00Z\"]},\"confidence\":0.98},{\"event_id\":\"e2\",\"event_type\":\"plan\",\"summary\":\"Assistant planned to write tests first, implement logic, and run cargo test.\",\"provenance\":{\"source_event_ids\":[\"a1\"],\"timestamps\":[\"2026-05-02T15:41:00Z\"]},\"confidence\":0.9},{\"event_id\":\"e3\",\"event_type\":\"file_change\",\"summary\":\"A test suite was created for movement, food, wall collision, and self collision.\",\"provenance\":{\"source_event_ids\":[\"t1\"],\"timestamps\":[\"2026-05-02T15:42:00Z\"],\"tool_ids\":[\"write_file\"],\"file_paths\":[\"snake_game/test_snake.py\"]},\"confidence\":0.92},{\"event_id\":\"e4\",\"event_type\":\"validation\",\"summary\":\"The unit test command passed.\",\"provenance\":{\"source_event_ids\":[\"t2\",\"r2\"],\"timestamps\":[\"2026-05-02T15:43:00Z\",\"2026-05-02T15:43:20Z\"],\"tool_ids\":[\"run_command\"],\"command_ids\":[\"cmd-cargo-test-1\"]},\"confidence\":0.95},{\"event_id\":\"e5\",\"event_type\":\"blocker\",\"summary\":\"A later runtime smoke showed the game did not actually start movement correctly despite passing tests.\",\"provenance\":{\"source_event_ids\":[\"u2\",\"r3\"],\"timestamps\":[\"2026-05-02T15:45:00Z\",\"2026-05-02T15:46:00Z\"],\"tool_ids\":[\"run_command\"],\"command_ids\":[\"cmd-run-game-1\"]},\"confidence\":0.88}],\"candidate_claims\":[{\"claim_id\":\"c1\",\"kind\":\"decided\",\"text\":\"The task acceptance requires runtime/playability validation, not only unit tests.\",\"support_event_ids\":[\"e1\",\"e5\"],\"evidence\":[{\"source_event_id\":\"u1\",\"quote\":\"build a snake game that run on my machine locally without using browser\"},{\"source_event_id\":\"u2\",\"quote\":\"did you check also that the game work or just the tests?\"}],\"truth_scope\":\"session_context\",\"promotion_hint\":\"strong_candidate\",\"confidence\":0.94},{\"claim_id\":\"c2\",\"kind\":\"validation_result\",\"text\":\"Unit tests passed, but the first runtime smoke failed to confirm playable behavior.\",\"support_event_ids\":[\"e4\",\"e5\"],\"evidence\":[{\"source_event_id\":\"r2\",\"quote\":\"tests passed\"},{\"source_event_id\":\"r3\",\"quote\":\"runtime smoke failed\"}],\"truth_scope\":\"historical\",\"promotion_hint\":\"candidate\",\"confidence\":0.9},{\"claim_id\":\"c3\",\"kind\":\"failed_because\",\"text\":\"The implementation was prematurely treated as complete after tests passed without sufficient runtime smoke validation.\",\"support_event_ids\":[\"e4\",\"e5\"],\"evidence\":[{\"source_event_id\":\"a2\",\"quote\":\"The implementation is complete and tests pass.\"},{\"source_event_id\":\"u2\",\"quote\":\"did you check also that the game work or just the tests?\"}],\"truth_scope\":\"session_context\",\"promotion_hint\":\"candidate\",\"confidence\":0.82}],\"decisions\":[{\"decision_id\":\"d1\",\"text\":\"Future similar tasks should track a runtime smoke check as a separate acceptance gate from unit tests.\",\"support_event_ids\":[\"e1\",\"e5\"],\"confidence\":0.86}],\"attempts\":[{\"attempt_id\":\"a1\",\"action\":\"Create tests before implementing Snake behavior.\",\"support_event_ids\":[\"e2\",\"e3\"],\"status\":\"succeeded\"},{\"attempt_id\":\"a2\",\"action\":\"Verify implementation with unit tests.\",\"support_event_ids\":[\"e4\"],\"status\":\"succeeded\"},{\"attempt_id\":\"a3\",\"action\":\"Verify playable runtime behavior.\",\"support_event_ids\":[\"e5\"],\"status\":\"failed\"}],\"outcomes\":[{\"outcome_id\":\"o1\",\"text\":\"Unit tests passed.\",\"support_event_ids\":[\"e4\"],\"status\":\"success\"},{\"outcome_id\":\"o2\",\"text\":\"Runtime behavior was not proven by the first test pass and required further fixing.\",\"support_event_ids\":[\"e5\"],\"status\":\"failure\"}],\"open_questions\":[{\"question_id\":\"q1\",\"text\":\"Which runtime input/rendering path prevented the game from starting movement?\",\"support_event_ids\":[\"e5\"]}],\"next_actions\":[{\"action_id\":\"n1\",\"text\":\"Add or run a local runtime smoke that verifies movement starts and input is handled.\",\"support_event_ids\":[\"e5\"],\"owner_hint\":\"agent\"}],\"warnings\":[{\"code\":\"source_gap\",\"message\":\"The trace does not include the final fixed code or successful runtime smoke, so do not promote final game correctness.\",\"source_event_ids\":[\"r3\"]}]}}\n+{\"id\":\"seed-0002-agnix-tool-release-triage\",\"input\":{\"source_kind\":\"repo_tool_trace\",\"source_uri\":\"github://agent-sh/agnix/issues/868\",\"raw\":{\"source_events\":[{\"id\":\"u1\",\"role\":\"user\",\"timestamp\":\"2026-05-05T21:47:55Z\",\"text\":\"check 868\"},{\"id\":\"t1\",\"role\":\"tool_call\",\"timestamp\":\"2026-05-05T21:47:58Z\",\"tool_id\":\"github\",\"text\":\"format issue 868 in agent-sh/agnix\"},{\"id\":\"r1\",\"role\":\"tool_result\",\"timestamp\":\"2026-05-05T21:48:00Z\",\"tool_id\":\"github\",\"text\":\"Issue #868: amp (Sourcegraph) gpt-5.5. Action required: review release notes for agnix validation rules, update config if validated fields changed, update research tracking, run workflow with update_baselines.\"},{\"id\":\"a1\",\"role\":\"assistant\",\"timestamp\":\"2026-05-05T21:48:08Z\",\"text\":\"Summary: GPT-5.5 is a model update for Amp deep mode; probably no agnix schema change unless config surfaces changed.\"},{\"id\":\"u2\",\"role\":\"user\",\"timestamp\":\"2026-05-05T21:50:00Z\",\"text\":\"so you get -1 on instruction follwing and he get +1\"},{\"id\":\"t2\",\"role\":\"tool_call\",\"timestamp\":\"2026-05-05T22:01:30Z\",\"tool_id\":\"edit\",\"file_paths\":[\".github/tool-release-baselines.json\"],\"text\":\"add changes_of_interest for amp config surfaces and relevant/irrelevant changes\"}]}},\"expected\":{\"episode_hint\":\"agnix-amp-release-triage\",\"objective_hint\":\"Triage auto-opened agnix issue #868 and identify whether Amp GPT-5.5 changes affect validation rules or tracking metadata.\",\"events\":[{\"event_id\":\"e1\",\"event_type\":\"user_request\",\"summary\":\"User asked to inspect issue #868.\",\"provenance\":{\"source_event_ids\":[\"u1\"],\"timestamps\":[\"2026-05-05T21:47:55Z\"]},\"confidence\":0.98},{\"event_id\":\"e2\",\"event_type\":\"tool_call\",\"summary\":\"GitHub issue #868 was fetched for agent-sh/agnix.\",\"provenance\":{\"source_event_ids\":[\"t1\",\"r1\"],\"timestamps\":[\"2026-05-05T21:47:58Z\",\"2026-05-05T21:48:00Z\"],\"tool_ids\":[\"github\"]},\"confidence\":0.96},{\"event_id\":\"e3\",\"event_type\":\"observation\",\"summary\":\"The issue describes an Amp model update and lists required triage actions around agnix validation rules and baselines.\",\"provenance\":{\"source_event_ids\":[\"r1\"],\"timestamps\":[\"2026-05-05T21:48:00Z\"],\"tool_ids\":[\"github\"]},\"confidence\":0.95},{\"event_id\":\"e4\",\"event_type\":\"observation\",\"summary\":\"Assistant summarized the issue as likely not requiring schema changes unless config surfaces changed.\",\"provenance\":{\"source_event_ids\":[\"a1\"],\"timestamps\":[\"2026-05-05T21:48:08Z\"]},\"confidence\":0.82},{\"event_id\":\"e5\",\"event_type\":\"file_change\",\"summary\":\"The baseline config was edited to add Amp changes_of_interest metadata.\",\"provenance\":{\"source_event_ids\":[\"t2\"],\"timestamps\":[\"2026-05-05T22:01:30Z\"],\"tool_ids\":[\"edit\"],\"file_paths\":[\".github/tool-release-baselines.json\"]},\"confidence\":0.93},{\"event_id\":\"e6\",\"event_type\":\"observation\",\"summary\":\"User noted an instruction-following penalty for the orchestrating assistant, while crediting another agent for following the intended fetch scope.\",\"provenance\":{\"source_event_ids\":[\"u2\"],\"timestamps\":[\"2026-05-05T21:50:00Z\"]},\"confidence\":0.86}],\"candidate_claims\":[{\"claim_id\":\"c1\",\"kind\":\"current_state\",\"text\":\"Issue #868 concerned Amp changing its tracked model baseline to GPT-5.5, not necessarily an agnix validation schema change.\",\"support_event_ids\":[\"e3\",\"e4\"],\"evidence\":[{\"source_event_id\":\"r1\",\"quote\":\"Issue #868: amp (Sourcegraph) gpt-5.5\"},{\"source_event_id\":\"a1\",\"quote\":\"probably no agnix schema change unless config surfaces changed\"}],\"truth_scope\":\"session_context\",\"promotion_hint\":\"candidate\",\"confidence\":0.84},{\"claim_id\":\"c2\",\"kind\":\"attempted\",\"text\":\"The agent updated .github/tool-release-baselines.json with changes_of_interest metadata for Amp.\",\"support_event_ids\":[\"e5\"],\"evidence\":[{\"source_event_id\":\"t2\",\"quote\":\"add changes_of_interest for amp config surfaces\"}],\"truth_scope\":\"historical\",\"promotion_hint\":\"strong_candidate\",\"confidence\":0.93},{\"claim_id\":\"c3\",\"kind\":\"has_next_action\",\"text\":\"The deterministic system should still verify whether the release baseline workflow and issue closure happened; the trace only shows the config edit.\",\"support_event_ids\":[\"e3\",\"e5\"],\"evidence\":[{\"source_event_id\":\"r1\",\"quote\":\"run workflow with update_baselines\"},{\"source_event_id\":\"t2\",\"quote\":\"add changes_of_interest\"}],\"truth_scope\":\"session_context\",\"promotion_hint\":\"candidate\",\"confidence\":0.8}],\"decisions\":[{\"decision_id\":\"d1\",\"text\":\"Treat model-version-only Amp releases as low schema risk unless config surfaces or validation inputs changed.\",\"support_event_ids\":[\"e3\",\"e4\",\"e5\"],\"confidence\":0.82}],\"attempts\":[{\"attempt_id\":\"a1\",\"action\":\"Fetch and inspect GitHub issue #868.\",\"support_event_ids\":[\"e1\",\"e2\"],\"status\":\"succeeded\"},{\"attempt_id\":\"a2\",\"action\":\"Add triage metadata for relevant Amp change types.\",\"support_event_ids\":[\"e5\"],\"status\":\"succeeded\"}],\"outcomes\":[{\"outcome_id\":\"o1\",\"text\":\"The trace contains a config edit but not final workflow/PR status.\",\"support_event_ids\":[\"e5\"],\"status\":\"partial\"}],\"open_questions\":[{\"question_id\":\"q1\",\"text\":\"Was the update_baselines workflow run and was issue #868 closed or linked to a PR?\",\"support_event_ids\":[\"e3\"]}],\"next_actions\":[{\"action_id\":\"n1\",\"text\":\"Check the PR or workflow result before promoting the triage as complete.\",\"support_event_ids\":[\"e3\",\"e5\"],\"owner_hint\":\"agent\"}],\"warnings\":[{\"code\":\"source_gap\",\"message\":\"The trace does not include the final CI/workflow result or issue closure state.\",\"source_event_ids\":[\"r1\",\"t2\"]}]}}\n+{\"id\":\"seed-0003-qwen-training-fastpath-decision\",\"input\":{\"source_kind\":\"training_trace\",\"source_uri\":\"file://aviary/qwen35-fastpath-experiment\",\"raw\":{\"source_events\":[{\"id\":\"u1\",\"role\":\"user\",\"timestamp\":\"2026-05-06T23:00:00Z\",\"text\":\"we trained 27b on this machine, how 9b oom on 2048?\"},{\"id\":\"a1\",\"role\":\"assistant\",\"timestamp\":\"2026-05-06T23:02:00Z\",\"text\":\"The 9B run is using a different path than the 27B smoke: full modules, different activation memory, and possibly non-fast fallback.\"},{\"id\":\"t1\",\"role\":\"tool_call\",\"timestamp\":\"2026-05-06T23:10:00Z\",\"tool_id\":\"pip\",\"command_id\":\"cmd-install-fla\",\"text\":\"install flash-linear-attention, fla-core, causal-conv1d with CUDA 13 headers workaround\"},{\"id\":\"r1\",\"role\":\"tool_result\",\"timestamp\":\"2026-05-06T23:20:00Z\",\"tool_id\":\"python\",\"command_id\":\"cmd-check-fastpath\",\"text\":\"flash-linear-attention==0.5.0, fla-core==0.5.0, causal-conv1d==1.6.1, is_fast_path_available True\"},{\"id\":\"t2\",\"role\":\"tool_call\",\"timestamp\":\"2026-05-06T23:30:00Z\",\"tool_id\":\"train\",\"command_id\":\"cmd-train-v06\",\"text\":\"train qwen35-9b context ops v0.6 invariants fla full 1792 60step\"},{\"id\":\"r2\",\"role\":\"tool_result\",\"timestamp\":\"2026-05-06T23:38:00Z\",\"tool_id\":\"train\",\"command_id\":\"cmd-train-v06\",\"text\":\"train_loss 0.2719536894311508 runtime 408.7794s peak_allocated 22.1617 GiB peak_reserved 22.2383 GiB\"},{\"id\":\"r3\",\"role\":\"tool_result\",\"timestamp\":\"2026-05-06T23:45:00Z\",\"tool_id\":\"eval\",\"command_id\":\"cmd-eval-v06\",\"text\":\"challenge raw-domain-valid 10/11 repaired-valid 11/11; remaining raw failure emitted resolution ask instead of ask_user\"}]}},\"expected\":{\"episode_hint\":\"qwen35-9b-fla-training-setup\",\"objective_hint\":\"Diagnose why a Qwen3.5 9B LoRA run OOMed and establish a stable fast-path training setup for episodic context operations.\",\"events\":[{\"event_id\":\"e1\",\"event_type\":\"user_request\",\"summary\":\"User questioned why the smaller 9B model OOMed when a 27B smoke had trained locally.\",\"provenance\":{\"source_event_ids\":[\"u1\"],\"timestamps\":[\"2026-05-06T23:00:00Z\"]},\"confidence\":0.98},{\"event_id\":\"e2\",\"event_type\":\"observation\",\"summary\":\"Assistant identified that memory behavior differed because the 9B run used different module coverage, activation path, and possibly non-fast fallback.\",\"provenance\":{\"source_event_ids\":[\"a1\"],\"timestamps\":[\"2026-05-06T23:02:00Z\"]},\"confidence\":0.82},{\"event_id\":\"e3\",\"event_type\":\"attempt\",\"summary\":\"The fast-path packages were installed using a CUDA 13 headers workaround.\",\"provenance\":{\"source_event_ids\":[\"t1\"],\"timestamps\":[\"2026-05-06T23:10:00Z\"],\"tool_ids\":[\"pip\"],\"command_ids\":[\"cmd-install-fla\"]},\"confidence\":0.9},{\"event_id\":\"e4\",\"event_type\":\"validation\",\"summary\":\"The Qwen3.5 fast path was verified as available with FLA 0.5.0 and causal-conv1d 1.6.1.\",\"provenance\":{\"source_event_ids\":[\"r1\"],\"timestamps\":[\"2026-05-06T23:20:00Z\"],\"tool_ids\":[\"python\"],\"command_ids\":[\"cmd-check-fastpath\"]},\"confidence\":0.98},{\"event_id\":\"e5\",\"event_type\":\"validation\",\"summary\":\"Qwen3.5 9B v0.6 trained for 60 steps at 1792 context with peak reserved memory around 22.24 GiB.\",\"provenance\":{\"source_event_ids\":[\"t2\",\"r2\"],\"timestamps\":[\"2026-05-06T23:30:00Z\",\"2026-05-06T23:38:00Z\"],\"tool_ids\":[\"train\"],\"command_ids\":[\"cmd-train-v06\"]},\"confidence\":0.97},{\"event_id\":\"e6\",\"event_type\":\"outcome\",\"summary\":\"Challenge eval reached 10/11 raw-domain-valid and 11/11 after deterministic repair, with one enum drift failure.\",\"provenance\":{\"source_event_ids\":[\"r3\"],\"timestamps\":[\"2026-05-06T23:45:00Z\"],\"tool_ids\":[\"eval\"],\"command_ids\":[\"cmd-eval-v06\"]},\"confidence\":0.97}],\"candidate_claims\":[{\"claim_id\":\"c1\",\"kind\":\"worked_because\",\"text\":\"The Qwen3.5 9B training path became viable after the FLA/causal-conv fast path was installed and verified.\",\"support_event_ids\":[\"e3\",\"e4\",\"e5\"],\"evidence\":[{\"source_event_id\":\"r1\",\"quote\":\"is_fast_path_available True\"},{\"source_event_id\":\"r2\",\"quote\":\"peak_reserved 22.2383 GiB\"}],\"truth_scope\":\"historical\",\"promotion_hint\":\"strong_candidate\",\"confidence\":0.92},{\"claim_id\":\"c2\",\"kind\":\"validation_result\",\"text\":\"The v0.6 adapter still had one raw enum drift issue: resolution ask instead of ask_user.\",\"support_event_ids\":[\"e6\"],\"evidence\":[{\"source_event_id\":\"r3\",\"quote\":\"remaining raw failure emitted resolution ask instead of ask_user\"}],\"truth_scope\":\"historical\",\"promotion_hint\":\"strong_candidate\",\"confidence\":0.97},{\"claim_id\":\"c3\",\"kind\":\"has_next_action\",\"text\":\"The next training material should include enum/invariant correction examples rather than more generic rows.\",\"support_event_ids\":[\"e6\"],\"evidence\":[{\"source_event_id\":\"r3\",\"quote\":\"remaining raw failure emitted resolution ask instead of ask_user\"}],\"truth_scope\":\"session_context\",\"promotion_hint\":\"candidate\",\"confidence\":0.86}],\"decisions\":[{\"decision_id\":\"d1\",\"text\":\"Use the verified FLA full path as the stable Qwen3.5 9B training setup, while keeping deterministic repair in the eval loop.\",\"support_event_ids\":[\"e4\",\"e5\",\"e6\"],\"confidence\":0.91}],\"attempts\":[{\"attempt_id\":\"a1\",\"action\":\"Install and verify Qwen3.5 fast-path dependencies.\",\"support_event_ids\":[\"e3\",\"e4\"],\"status\":\"succeeded\"},{\"attempt_id\":\"a2\",\"action\":\"Train Qwen3.5 9B context-ops v0.6 for 60 steps.\",\"support_event_ids\":[\"e5\"],\"status\":\"succeeded\"},{\"attempt_id\":\"a3\",\"action\":\"Evaluate v0.6 on the challenge split.\",\"support_event_ids\":[\"e6\"],\"status\":\"succeeded\"}],\"outcomes\":[{\"outcome_id\":\"o1\",\"text\":\"Training fit within 24 GiB class VRAM with about 22.24 GiB peak reserved.\",\"support_event_ids\":[\"e5\"],\"status\":\"success\"},{\"outcome_id\":\"o2\",\"text\":\"Eval was valid after repair but not perfectly raw-valid due to enum drift.\",\"support_event_ids\":[\"e6\"],\"status\":\"partial\"}],\"open_questions\":[{\"question_id\":\"q1\",\"text\":\"Does the same fast path stay stable for larger context or broader adapter coverage?\",\"support_event_ids\":[\"e5\"]}],\"next_actions\":[{\"action_id\":\"n1\",\"text\":\"Add correction rows for enum/invariant failures and train a short v0.7 comparison from the base model.\",\"support_event_ids\":[\"e6\"],\"owner_hint\":\"agent\"}],\"warnings\":[{\"code\":\"policy_owned_by_deterministic_system\",\"message\":\"The model can propose normalized training-state claims, but the verifier should own final acceptance of adapter quality.\",\"source_event_ids\":[\"r3\"]}]}}\n*** End Patch\n", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "apply_patch: *** Begin Patch\n*** Add File: /home/avifenesh/projects/episodic-ingestion-compiler/data/examples/seed.jsonl\n+{\"id\":\"seed-0001-snake-tdd-validation-gap\",\"input\":{\"so...", | |
| "tool_name": "apply_patch" | |
| }, | |
| "expected_probability": 0.4651445150375366, | |
| "expected_source_event_id": "t19", | |
| "field": "attempted_actions", | |
| "rank": 20, | |
| "reciprocal_rank": 0.05, | |
| "row_id": "pta-codex-rollout-2026-04-30T19-43-19-019ddf45-fdda-7fb2-8764-8162ba11dc24::w88", | |
| "top_candidates": [ | |
| { | |
| "event": { | |
| "command": "test -e /home/avifenesh/projects/episodic-ingestion-compiler; echo $?", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "test -e /home/avifenesh/projects/episodic-ingestion-compiler; echo $?", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t6", | |
| "support_probability": 0.9998511075973511 | |
| }, | |
| { | |
| "event": { | |
| "command": "mkdir -p /home/avifenesh/projects/episodic-ingestion-compiler/{src/episodic_ingestion_compiler,scripts,schemas,data/examples,data/sft,docs,tests}", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "mkdir -p /home/avifenesh/projects/episodic-ingestion-compiler/{src/episodic_ingestion_compiler,scripts,schemas,data/examples,data/sft,docs,tests}", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t12", | |
| "support_probability": 0.9995827078819275 | |
| }, | |
| { | |
| "event": { | |
| "command": "python3 - <<'PY'\ntry:\n import jsonschema\n print('jsonschema ok')\nexcept Exception as e:\n print(f'jsonschema missing: {e}')\nPY", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "python3 - <<'PY'\ntry:\n import jsonschema\n print('jsonschema ok')\nexcept Exception as e:\n print(f'jsonschema missing: {e}')\nPY", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t13", | |
| "support_probability": 0.9994966983795166 | |
| }, | |
| { | |
| "event": { | |
| "command": "{\"plan\": [{\"step\": \"Create clean repo scaffold\", \"status\": \"in_progress\"}, {\"step\": \"Define ingestion schema and contract\", \"status\": \"pending\"}, {\"step\": \"Add seed training examples and validators\", \"status\": \"pending\"}, {\"step\": \"Run validation and commit\", \"status\": \"pending\"}, {\"step\": \"Report next training path\", \"status\": \"pending\"}]}", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "{\"plan\": [{\"step\": \"Create clean repo scaffold\", \"status\": \"in_progress\"}, {\"step\": \"Define ingestion schema and contract\", \"status\": \"pending\"}, {\"step\": \"Add seed training exa...", | |
| "tool_name": "update_plan" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t3", | |
| "support_probability": 0.99935382604599 | |
| }, | |
| { | |
| "event": { | |
| "command": "git status --short", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "git status --short", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t7", | |
| "support_probability": 0.99935382604599 | |
| } | |
| ] | |
| }, | |
| { | |
| "expected_event": { | |
| "command": "python3 << 'EOF'\nimport json, re, sys\n\nwith open('/tmp/valkey_prs_batch4.json') as f:\n raw = f.read()\n\n# Split by PR markers\nprs = re.split(r'=== PR #(\\d+) ===\\n', raw)[1:]\n# pairs: [num, content, num, content, ...]\n\nfor i in range(0, len(prs), 2):\n num = prs[i]\n content = prs[i+1]\n # Split content at --- inline comments ---\n parts = content.split('--- inline comments ---\\n', 1)\n pr_json_text = parts[0].strip()\n inline_text = parts[1].strip() if len(parts) > 1 else ''\n \n try:\n pr = json.loads(pr_json_text)\n except Exception as e:\n print(f\"PR {num} json err: {e}\")\n continue\n \n print(f\"\\n===== PR #{num} =====\")\n print(f\"Title: {pr.get('title')}\")\n print(f\"Base: {pr.get('baseRefName')}\")\n print(f\"Author: {pr.get('author', {}).g", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "python3 << 'EOF'\nimport json, re, sys\n\nwith open('/tmp/valkey_prs_batch4.json') as f:\n raw = f.read()\n\n# Split by PR markers\nprs = re.split(r'=== PR #(\\d+) ===\\n', raw)[1:]\n#...", | |
| "tool_name": "Bash" | |
| }, | |
| "expected_probability": 0.3988746404647827, | |
| "expected_source_event_id": "t7", | |
| "field": "attempted_actions", | |
| "rank": 20, | |
| "reciprocal_rank": 0.05, | |
| "row_id": "pta-claude_code-agent-a55fe1edceb764d9e::w0", | |
| "top_candidates": [ | |
| { | |
| "event": { | |
| "command": "cd /home/avifenesh/projects/valkey && git log --all --grep=\"#3469\" --oneline | head -5 && echo \"---\" && git log --all --oneline | grep -i \"3469\" | head -5 && echo \"---forkless branch---\" && git branch -a | grep -i forkless | head", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "cd /home/avifenesh/projects/valkey && git log --all --grep=\"#3469\" --oneline | head -5 && echo \"---\" && git log --all --oneline | grep -i \"3469\" | head -5 && echo \"---forkless b...", | |
| "tool_name": "Bash" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t14", | |
| "support_probability": 0.999595582485199 | |
| }, | |
| { | |
| "event": { | |
| "command": "cd /home/avifenesh/projects/valkey && git fetch origin +refs/pull/3469/head:refs/remotes/origin/pr-3469 2>&1 | tail -3 && git log --format='%H %s%n%b' -1 origin/pr-3469 2>&1 | head -40", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "cd /home/avifenesh/projects/valkey && git fetch origin +refs/pull/3469/head:refs/remotes/origin/pr-3469 2>&1 | tail -3 && git log --format='%H %s%n%b' -1 origin/pr-3469 2>&1 | h...", | |
| "tool_name": "Bash" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t16", | |
| "support_probability": 0.999555766582489 | |
| }, | |
| { | |
| "event": { | |
| "command": "ls -la /tmp/valkey_prs_batch4.json && head -100 /tmp/valkey_prs_batch4.json", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "ls -la /tmp/valkey_prs_batch4.json && head -100 /tmp/valkey_prs_batch4.json", | |
| "tool_name": "Bash" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t5", | |
| "support_probability": 0.9994115829467773 | |
| }, | |
| { | |
| "event": { | |
| "command": "Read /home/avifenesh/.claude/projects/-home-avifenesh-projects-valkey-skills/ff8a0875-8828-483a-a804-8fd20d09aecb/tool-results/bq6pp40yy.txt", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "Read /home/avifenesh/.claude/projects/-home-avifenesh-projects-valkey-skills/ff8a0875-8828-483a-a804-8fd20d09aecb/tool-results/bq6pp40yy.txt", | |
| "tool_name": "Read" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t9", | |
| "support_probability": 0.9987157583236694 | |
| }, | |
| { | |
| "event": { | |
| "command": "for pr in 3498 3495 3472 3471 3469 3464 3463 3462 3461 3460; do\n echo \"=== PR #$pr ===\"\n gh pr view $pr --repo valkey-io/valkey --json title,body,reviews,comments,labels,baseRefName,mergeCommit,author 2>&1\n echo \"--- inline comments ---\"\n gh api repos/valkey-io/valkey/pulls/$pr/comments --paginate 2>&1\n echo\ndone > /tmp/valkey_prs_batch4.json 2>&1\nwc -l /tmp/valkey_prs_batch4.json", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "for pr in 3498 3495 3472 3471 3469 3464 3463 3462 3461 3460; do\n echo \"=== PR #$pr ===\"\n gh pr view $pr --repo valkey-io/valkey --json title,body,reviews,comments,labels,baseR...", | |
| "tool_name": "Bash" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t3", | |
| "support_probability": 0.9980732202529907 | |
| } | |
| ] | |
| }, | |
| { | |
| "expected_event": { | |
| "command": "for sha in 5157255b82002a79fd212de9d9ba0fe08812849f 6b384e9a5b6cd4e4f8198f5df22a74feb1b70e0f 6c329dfe2c02d09b276fd5617acdb31e6169ebc1 8caa29298c51b4bb498418515801dd0720ebce62 5deebd0f131826124b3ab6231fafb5f9b188523b 6b85ca4fffff51de5f8cfcc6ce239900a48924d0 1e2ca5ec5dd6f97215e7acbcee14e89ea56a9e02 6414720504f689a3fc699338bf22565bb70c3829 9f0106755b177f8eea3d62eed6f28f20fb54d459 a3a839972ff10a68b1a7069d70d8ce1fffb8f02a; do echo \"=== $sha ===\"; git -C /home/avifenesh/projects/valkey show --format='%s%n%b%n---END---' -s $sha 2>&1 | head -30; done", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "for sha in 5157255b82002a79fd212de9d9ba0fe08812849f 6b384e9a5b6cd4e4f8198f5df22a74feb1b70e0f 6c329dfe2c02d09b276fd5617acdb31e6169ebc1 8caa29298c51b4bb498418515801dd0720ebce62 5d...", | |
| "tool_name": "Bash" | |
| }, | |
| "expected_probability": 0.709019124507904, | |
| "expected_source_event_id": "t12", | |
| "field": "attempted_actions", | |
| "rank": 20, | |
| "reciprocal_rank": 0.05, | |
| "row_id": "pta-claude_code-agent-a5a805ef49f3ca63c::w0", | |
| "top_candidates": [ | |
| { | |
| "event": { | |
| "command": "cd /tmp/pr-survey && for pr in 3343 3342 3340 3338 3336 3333 3331 3329 3325 3324; do gh api repos/valkey-io/valkey/pulls/$pr/comments --paginate > inline-$pr.json 2>&1 & done; wait; ls -la inline-*", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "cd /tmp/pr-survey && for pr in 3343 3342 3340 3338 3336 3333 3331 3329 3325 3324; do gh api repos/valkey-io/valkey/pulls/$pr/comments --paginate > inline-$pr.json 2>&1 & done; w...", | |
| "tool_name": "Bash" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t4", | |
| "support_probability": 0.9994471669197083 | |
| }, | |
| { | |
| "event": { | |
| "command": "cd /tmp/pr-survey && for pr in 3343 3342 3340 3338 3336 3333 3331 3329 3325 3324; do echo \"=== PR $pr inline-comments ===\"; jq -r '.[] | \"@\\(.user.login) on \\(.path):\\(.line // .original_line): \\((.body // \"\")[:250])\"' inline-$pr.json 2>/dev/null | head -30; done 2>&1 | head -300", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "cd /tmp/pr-survey && for pr in 3343 3342 3340 3338 3336 3333 3331 3329 3325 3324; do echo \"=== PR $pr inline-comments ===\"; jq -r '.[] | \"@\\(.user.login) on \\(.path):\\(.line //...", | |
| "tool_name": "Bash" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t11", | |
| "support_probability": 0.9993928670883179 | |
| }, | |
| { | |
| "event": { | |
| "command": "cd /tmp/pr-survey && for pr in 3343 3342 3340 3338 3336 3333 3331 3329 3325 3324; do echo \"=== PR $pr reviews+comments ===\"; jq -r '.reviews[] | \"[review] [\\(.state)] @\\(.author.login): \\((.body // \"\")[:300])\"' pr-$pr.json; jq -r '.comments[] | \"[comment] @\\(.author.login): \\((.body // \"\")[:300])\"' pr-$pr.json; done 2>&1 | head -500", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "cd /tmp/pr-survey && for pr in 3343 3342 3340 3338 3336 3333 3331 3329 3325 3324; do echo \"=== PR $pr reviews+comments ===\"; jq -r '.reviews[] | \"[review] [\\(.state)] @\\(.author...", | |
| "tool_name": "Bash" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t9", | |
| "support_probability": 0.9992678761482239 | |
| }, | |
| { | |
| "event": { | |
| "command": "cd /tmp/pr-survey && for pr in 3343 3342 3340 3338 3336 3333 3331 3329 3325 3324; do echo \"=== PR $pr ===\"; jq -r '{title, author: .author.login, baseRefName, labels: [.labels[].name], mergeCommit: .mergeCommit.oid, body: (.body|.[:1500])}' pr-$pr.json; done 2>&1 | head -400", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "cd /tmp/pr-survey && for pr in 3343 3342 3340 3338 3336 3333 3331 3329 3325 3324; do echo \"=== PR $pr ===\"; jq -r '{title, author: .author.login, baseRefName, labels: [.labels[]...", | |
| "tool_name": "Bash" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t7", | |
| "support_probability": 0.9988664388656616 | |
| }, | |
| { | |
| "event": { | |
| "command": "cd /tmp && mkdir -p pr-survey && cd pr-survey && for pr in 3343 3342 3340 3338 3336 3333 3331 3329 3325 3324; do gh pr view $pr --repo valkey-io/valkey --json title,body,reviews,comments,labels,baseRefName,mergeCommit,author > pr-$pr.json 2>&1 & done; wait; ls -la", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "cd /tmp && mkdir -p pr-survey && cd pr-survey && for pr in 3343 3342 3340 3338 3336 3333 3331 3329 3325 3324; do gh pr view $pr --repo valkey-io/valkey --json title,body,reviews...", | |
| "tool_name": "Bash" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t3", | |
| "support_probability": 0.9987550973892212 | |
| } | |
| ] | |
| }, | |
| { | |
| "expected_event": { | |
| "command": "ROOT=/home/avifenesh/projects/aviary; RUN_LOG=\"$ROOT/data/runs/qwen35-moe-8k-triton-$(date +%Y%m%d-%H%M%S).log\"; export PATH=\"$ROOT/.venv/bin:$PATH\" PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS=0; echo \"log=$RUN_LOG\"; sudo -n systemd-run --scope -p MemoryMax=32G -p MemorySwapMax=0 \"$ROOT/.venv/bin/vllm\" serve \"$ROOT/models/qwen3.6-35b-a3b-nvfp4\" --host 127.0.0.1 --port 8000 --served-model-name aviary-local-qwen --trust-remote-code --language-model-only --skip-mm-profiling --chat-template \"$ROOT/configs/qwen-codex-chat-template.jinja\" --reasoning-parser qwen3 --enable-auto-tool-choice --tool-call-parser qwen3_coder --max-model-len 8192 --kv-cache-dtype fp8 --enable-prefix-caching --gpu-memory-utilization 0.92 --max-num-seqs 1 --max-num-batched-to", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "ROOT=/home/avifenesh/projects/aviary; RUN_LOG=\"$ROOT/data/runs/qwen35-moe-8k-triton-$(date +%Y%m%d-%H%M%S).log\"; export PATH=\"$ROOT/.venv/bin:$PATH\" PYTORCH_CUDA_ALLOC_CONF=expa...", | |
| "tool_name": "exec_command" | |
| }, | |
| "expected_probability": 0.4012185037136078, | |
| "expected_source_event_id": "t10", | |
| "field": "attempted_actions", | |
| "rank": 19, | |
| "reciprocal_rank": 0.05263157894736842, | |
| "row_id": "pta-codex-rollout-2026-04-30T19-43-19-019ddf45-fdda-7fb2-8764-8162ba11dc24::w14", | |
| "top_candidates": [ | |
| { | |
| "event": { | |
| "command": "{\"session_id\": 92479, \"chars\": \"\\u0003\", \"yield_time_ms\": 1000, \"max_output_tokens\": 20000}", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "{\"session_id\": 92479, \"chars\": \"\\u0003\", \"yield_time_ms\": 1000, \"max_output_tokens\": 20000}", | |
| "tool_name": "write_stdin" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t3", | |
| "support_probability": 0.9996079802513123 | |
| }, | |
| { | |
| "event": { | |
| "command": "{\"session_id\": 1703, \"chars\": \"\", \"yield_time_ms\": 1000, \"max_output_tokens\": 30000}", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "{\"session_id\": 1703, \"chars\": \"\", \"yield_time_ms\": 1000, \"max_output_tokens\": 30000}", | |
| "tool_name": "write_stdin" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t12", | |
| "support_probability": 0.9992678761482239 | |
| }, | |
| { | |
| "event": { | |
| "command": "{\"session_id\": 1703, \"chars\": \"\", \"yield_time_ms\": 1000, \"max_output_tokens\": 6000}", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "{\"session_id\": 1703, \"chars\": \"\", \"yield_time_ms\": 1000, \"max_output_tokens\": 6000}", | |
| "tool_name": "write_stdin" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t17", | |
| "support_probability": 0.9992678761482239 | |
| }, | |
| { | |
| "event": { | |
| "command": "nvidia-smi --query-gpu=memory.used,memory.free,utilization.gpu --format=csv,noheader,nounits && free -h && ss -ltnp sport = :8000", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "nvidia-smi --query-gpu=memory.used,memory.free,utilization.gpu --format=csv,noheader,nounits && free -h && ss -ltnp sport = :8000", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t13", | |
| "support_probability": 0.9985895752906799 | |
| }, | |
| { | |
| "event": { | |
| "command": "nvidia-smi --query-gpu=name,memory.used,memory.free,utilization.gpu --format=csv,noheader,nounits && free -h && ss -ltnp 'sport = :8000' || true", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "nvidia-smi --query-gpu=name,memory.used,memory.free,utilization.gpu --format=csv,noheader,nounits && free -h && ss -ltnp 'sport = :8000' || true", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t18", | |
| "support_probability": 0.998245358467102 | |
| } | |
| ] | |
| }, | |
| { | |
| "expected_event": { | |
| "command": "*** Begin Patch\n*** Update File: /home/avifenesh/projects/aviary/scripts/build-scorecard-sft-dataset.py\n@@\n-By default this exports only canonical Avi scorecards. Use\n---include-corrected-drafts to add draft files that Avi corrected, and\n---include-drafts only for research/debug datasets.\n+By default this exports only canonical Avi scorecards. Use\n+--include-corrected-drafts to add draft files that Avi corrected, and\n+--include-drafts only for research/debug datasets. Individual cards can set\n+exclude_from_sft: true when they are useful as reviewed eval evidence but\n+unsafe or unhelpful as training targets.\n@@\n def target_yaml(card: dict[str, Any]) -> str:\n@@\n return yaml.safe_dump(target, sort_keys=False, allow_unicode=False).strip() + \"\\n\"\n+\n+\n+def is_excluded_from_sft(card: dict[str, Any]) -> bool:\n+ if bool(card.get(\"exclude_from_sft\")):\n+ return True\n+ eligibility = str(card.get(\"training_eligibility\", \"\")).strip().lower()\n+ return eligibility in {\"exclude\", \"excluded\", \"do_not_train\", \"not_for_training\"}\n@@\n records = source_records(args.lane, args.include_corrected_drafts, args.include_drafts)\n for source_file, card, source_kind in records:\n stats[f\"source:{source_kind}\"] += 1\n+ if is_excluded_from_sft(card):\n+ stats[f\"excluded_from_sft:{source_kind}\"] += 1\n+ skipped.append(\n+ {\n+ \"source_file\": str(source_file),\n+ \"run_id\": str(card.get(\"run_id\")),\n+ \"case_id\": str(card.get(\"case_id\")),\n+ \"reason\": \"card marked exclude_from_sft\",\n+ }\n+ )\n+ continue\n result_path = result_path_for_card(args.lane, card)\n*** End Patch\n", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "apply_patch: *** Begin Patch\n*** Update File: /home/avifenesh/projects/aviary/scripts/build-scorecard-sft-dataset.py\n@@\n-By default this exports only canonical Avi scorecards. U...", | |
| "tool_name": "apply_patch" | |
| }, | |
| "expected_probability": 0.6141626238822937, | |
| "expected_source_event_id": "t3", | |
| "field": "attempted_actions", | |
| "rank": 19, | |
| "reciprocal_rank": 0.05263157894736842, | |
| "row_id": "pta-codex-rollout-2026-04-30T19-43-19-019ddf45-fdda-7fb2-8764-8162ba11dc24::w32", | |
| "top_candidates": [ | |
| { | |
| "event": { | |
| "command": "wc -l data/training/session_summarize_for_avi/scorecard_sft_avi_verified.jsonl data/training/session_summarize_for_avi/scorecard_sft_with_drafts.jsonl", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "wc -l data/training/session_summarize_for_avi/scorecard_sft_avi_verified.jsonl data/training/session_summarize_for_avi/scorecard_sft_with_drafts.jsonl", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t13", | |
| "support_probability": 0.9992903470993042 | |
| }, | |
| { | |
| "event": { | |
| "command": "python3 -m py_compile scripts/build-scorecard-sft-dataset.py && scripts/build-scorecard-sft-dataset.py --out data/training/session_summarize_for_avi/scorecard_sft_avi_verified.jsonl --include-corrected-drafts", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "python3 -m py_compile scripts/build-scorecard-sft-dataset.py && scripts/build-scorecard-sft-dataset.py --out data/training/session_summarize_for_avi/scorecard_sft_avi_verified.j...", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t9", | |
| "support_probability": 0.9989350438117981 | |
| }, | |
| { | |
| "event": { | |
| "command": "sed -n '1,120p' data/training/session_summarize_for_avi/scorecard_sft_avi_verified.jsonl.manifest.yml", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "sed -n '1,120p' data/training/session_summarize_for_avi/scorecard_sft_avi_verified.jsonl.manifest.yml", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t14", | |
| "support_probability": 0.9987550973892212 | |
| }, | |
| { | |
| "event": { | |
| "command": "scripts/build-scorecard-sft-dataset.py --out data/training/session_summarize_for_avi/scorecard_sft_with_drafts.jsonl --include-drafts", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "scripts/build-scorecard-sft-dataset.py --out data/training/session_summarize_for_avi/scorecard_sft_with_drafts.jsonl --include-drafts", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t10", | |
| "support_probability": 0.9986329674720764 | |
| }, | |
| { | |
| "event": { | |
| "command": "rg -n \"round-04-workbench|round-05-one-bash|round-06-qwen9\" data/training/session_summarize_for_avi/scorecard_sft_avi_verified.jsonl data/training/session_summarize_for_avi/scorecard_sft_avi_verified.jsonl.manifest.yml", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "rg -n \"round-04-workbench|round-05-one-bash|round-06-qwen9\" data/training/session_summarize_for_avi/scorecard_sft_avi_verified.jsonl data/training/session_summarize_for_avi/scor...", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t15", | |
| "support_probability": 0.9969245791435242 | |
| } | |
| ] | |
| }, | |
| { | |
| "expected_event": { | |
| "command": "*** Begin Patch\n*** Update File: /home/avifenesh/projects/aviary/data/runs/web_research_for_avi/20260503-github-web-research/codex-scorecards-draft.yml\n@@\n-review_status: needs_avi_review\n+review_status: avi_corrected_partial\n+reviewed_by: avi\n+review_note: >-\n+ Avi rated the main Aviary web_fetch result as 2-3. This is not a zero because\n+ it found the most-starred agent-sh repo, but it is still below usable because\n+ it missed the real open-work queue and over-presented incomplete research.\n@@\n - run_id: \"20260503-github-web-research/aviary-chat-once-webfetch\"\n model_label: qwen35-9b\n harness: aviary-chat-once\n case_id: round-01-github-open-issues-agent-sh-stars\n- decision: draft_negative_mixed_webfetch_result\n+ review_status: avi_corrected\n+ reviewed_by: avi\n+ decision: verified_low_mixed_webfetch_result_after_correction\n+ avi_score_range: \"2-3\"\n scores:\n- overall_usefulness: 2\n+ overall_usefulness: 2\n@@\n do_not_train_as_success: 1\n+ avi_feedback: >-\n+ I would give it 2-3.\n summary: >-\n Mixed but mostly negative web-research sample. The model used Aviary's\n web_search/web_fetch tools and correctly found `agent-sh/agentsys` as the\n most-starred repo with 777 stars. It failed the more important work-queue\n part: it did not inspect `avifenesh/*` properly, missed open\n `agent-sh/agnix` issues/PRs involving Avi, relied on shallow GitHub pages,\n- and presented incomplete findings too confidently.\n+ and presented incomplete findings too confidently. Avi corrected this to\n+ a 2-3 range; encode conservatively as 2 while preserving the range.\n*** End Patch\n", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "apply_patch: *** Begin Patch\n*** Update File: /home/avifenesh/projects/aviary/data/runs/web_research_for_avi/20260503-github-web-research/codex-scorecards-draft.yml\n@@\n-review_s...", | |
| "tool_name": "apply_patch" | |
| }, | |
| "expected_probability": 0.7799928784370422, | |
| "expected_source_event_id": "t7", | |
| "field": "attempted_actions", | |
| "rank": 19, | |
| "reciprocal_rank": 0.05263157894736842, | |
| "row_id": "pta-codex-rollout-2026-04-30T19-43-19-019ddf45-fdda-7fb2-8764-8162ba11dc24::w34", | |
| "top_candidates": [ | |
| { | |
| "event": { | |
| "command": "git diff -- data/runs/web_research_for_avi/20260503-github-web-research/codex-scorecards-draft.yml", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "git diff -- data/runs/web_research_for_avi/20260503-github-web-research/codex-scorecards-draft.yml", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t10", | |
| "support_probability": 0.999841570854187 | |
| }, | |
| { | |
| "event": { | |
| "command": "git diff --check", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "git diff --check", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t11", | |
| "support_probability": 0.9997040629386902 | |
| }, | |
| { | |
| "event": { | |
| "command": "git add data/runs/web_research_for_avi/20260503-github-web-research/codex-scorecards-draft.yml && git commit -m \"Mark web research eval corrected\"", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "git add data/runs/web_research_for_avi/20260503-github-web-research/codex-scorecards-draft.yml && git commit -m \"Mark web research eval corrected\"", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t16", | |
| "support_probability": 0.999555766582489 | |
| }, | |
| { | |
| "event": { | |
| "command": "rg -n \"review_status|aviary-chat-once|overall_usefulness|avi_feedback|decision|summary\" data/runs/web_research_for_avi/20260503-github-web-research/codex-scorecards-draft.yml", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "rg -n \"review_status|aviary-chat-once|overall_usefulness|avi_feedback|decision|summary\" data/runs/web_research_for_avi/20260503-github-web-research/codex-scorecards-draft.yml", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t3", | |
| "support_probability": 0.999480664730072 | |
| }, | |
| { | |
| "event": { | |
| "command": "git status --short", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "git status --short", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t18", | |
| "support_probability": 0.9994115829467773 | |
| } | |
| ] | |
| }, | |
| { | |
| "expected_event": { | |
| "command": "*** Begin Patch\n*** Update File: /home/avifenesh/projects/aviary/data/evals/personal_agent_for_avi/scorecards.yml\n@@\n - pr_workflow_reference\n - bookkeeping_recovery_reference\n - shell_quoting_negative_reference\n - safe_merge_policy_reference\n+- id: score-20260504T144215-qwen27-agnix-claude-code-triage\n+ created_at: 2026-05-04T14:42:15+03:00\n+ rater: avi\n+ lane: personal_agent_for_avi\n+ run_id: 20260504-live-agnix-claude-code-triage\n+ model_label: aviary-local-qwen\n+ case_id: agnix-claude-code-issue-850-triage\n+ result_path: /home/avifenesh/projects/aviary/data/runs/personal_agent_for_avi/20260504-live-agnix-claude-code-triage/aviary-local-qwen/agnix-claude-code-issue-850-triage/input.jsonl\n+ input_path: /home/avifenesh/projects/aviary/data/runs/personal_agent_for_avi/20260504-live-agnix-claude-code-triage/aviary-local-qwen/agnix-claude-code-issue-850-triage/input.jsonl\n+ decision: useful_success_with_branch_and_quoting_failures\n+ scores:\n+ agentic_behavior: 4\n+ bookkeeping: 2\n+ grounding_no_hallucination: 4\n+ instruction_following: 4\n+ judgment_prioritization: 3\n+ latency_acceptability: 4\n+ long_context_handling: 4\n+ output_quality: 3\n+ overall_usefulness: 4\n+ recovery: 4\n+ safety_restraint: 3\n+ terminal_usage: 3\n+ tone_intent_understanding: 4\n+ tool_use: 3\n+ notes: >-\n+ Avi initially judged this as another useful session: helpful, fast, good\n+ terminal/tool work, and good task understanding. After review, Avi corrected\n+ that the agent did make a mistake. The material mistake was branch hygiene:\n+ it opened the Claude Code PR while still based on the previous OpenCode PR\n+ branch state, so PR #863 initially included old OpenCode commits and\n+ triggered stale Copilot/Gemini review comments. The agent also repeated the\n+ shell quoting bug from the previous session: backticks in a double-quoted\n+ `gh pr create --body` command were interpreted by the shell, leaving a\n+ mangled PR body. The final result was still useful because it recovered after\n+ Avi pointed out the wrong branch, rebased onto main, produced a clean diff,\n+ waited for all checks, and merged PR #863 with issue #850 closed.\n+ evidence:\n+ - The agent correctly found 13 open issues, separated 5 auto issues from 8 manual issues, and selected #850 as the first auto issue to process.\n+ - It correctly classified Claude Code v2.1.126 as agnix-irrelevant, updated the baseline and Last Reviewed date, ran rule bookkeeping, opened PR #863, waited for CI/reviews, and eventually merged it.\n+ - The agent created the new branch before updating local main; `git log main..HEAD` showed the previous OpenCode commits in the Claude Code PR.\n+ - Avi corrected the branch mistake with \"because you need to move to main you are on a wrong branch\"; the agent switched to main, pulled, rebased, and reduced the PR to one clean Claude Code commit.\n+ - PR #863 body still shows shell quoting damage in the Changes bullets because Markdown backticks were interpreted by the shell during `gh pr create`.\n+ - All checks passed before merge, PR #863 merged, and issue #850 closed.\n+ failure_modes:\n+ - started_pr_from_wrong_base_branch\n+ - stale_prior_pr_commits_included_initially\n+ - stale_bot_review_noise_from_dirty_base\n+ - shell_quoting_backticks_mangled_pr_body\n+ - did_not_repair_mangled_pr_body\n+ - repeated_same_pr_body_quoting_failure\n+ - gh_pr_check_typo\n+ - attempted_review_dismissal_with_wrong_api\n+ tuning_uses:\n+ - preference_dataset\n+ - eval_regression\n+ - judge_training\n+ - supervised_finetune_example\n+ - positive_agentic_terminal_reference\n+ - branch_hygiene_negative_reference\n+ - pr_workflow_recovery_reference\n+ - shell_quoting_negative_reference\n+ - owner_correction_recovery_reference\n*** End Patch\n", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "apply_patch: *** Begin Patch\n*** Update File: /home/avifenesh/projects/aviary/data/evals/personal_agent_for_avi/scorecards.yml\n@@\n - pr_workflow_reference\n - bookkeeping_rec...", | |
| "tool_name": "apply_patch" | |
| }, | |
| "expected_probability": 0.48121002316474915, | |
| "expected_source_event_id": "t10", | |
| "field": "attempted_actions", | |
| "rank": 19, | |
| "reciprocal_rank": 0.05263157894736842, | |
| "row_id": "pta-codex-rollout-2026-04-30T19-43-19-019ddf45-fdda-7fb2-8764-8162ba11dc24::w45", | |
| "top_candidates": [ | |
| { | |
| "event": { | |
| "command": "cargo test eval_score --quiet", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "cargo test eval_score --quiet", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t13", | |
| "support_probability": 0.9996849298477173 | |
| }, | |
| { | |
| "event": { | |
| "command": "tail -n 20 data/evals/personal_agent_for_avi/scorecards.yml", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "tail -n 20 data/evals/personal_agent_for_avi/scorecards.yml", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t5", | |
| "support_probability": 0.9995827078819275 | |
| }, | |
| { | |
| "event": { | |
| "command": "date --iso-8601=seconds", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "date --iso-8601=seconds", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t4", | |
| "support_probability": 0.999555766582489 | |
| }, | |
| { | |
| "event": { | |
| "command": "python3 - <<'PY'\nfrom pathlib import Path\nimport yaml\np=Path('data/evals/personal_agent_for_avi/scorecards.yml')\ndata=yaml.safe_load(p.read_text())\nprint(data['lane'], len(data['scorecards']))\nfor c in data['scorecards']:\n print(c['id'], c['decision'], c['scores'].get('overall_usefulness'), c['scores'].get('bookkeeping'))\nPY", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "python3 - <<'PY'\nfrom pathlib import Path\nimport yaml\np=Path('data/evals/personal_agent_for_avi/scorecards.yml')\ndata=yaml.safe_load(p.read_text())\nprint(data['lane'], len(data[...", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t14", | |
| "support_probability": 0.9985449314117432 | |
| }, | |
| { | |
| "event": { | |
| "command": "python3 - <<'PY'\nfrom pathlib import Path\nimport yaml\nroot=Path('/home/avifenesh/projects/aviary')\ntotal=0\nfor p in sorted((root/'data/evals').glob('*/scorecards.yml')):\n data=yaml.safe_load(p.read_text()) or {}\n n=len(data.get('scorecards') or [])\n total+=n\n print(data.get('lane') or p.parent.name, n)\nprint('total', total)\nPY", | |
| "is_context_tool_echo": false, | |
| "role": "tool_call", | |
| "success": null, | |
| "text": "python3 - <<'PY'\nfrom pathlib import Path\nimport yaml\nroot=Path('/home/avifenesh/projects/aviary')\ntotal=0\nfor p in sorted((root/'data/evals').glob('*/scorecards.yml')):\n dat...", | |
| "tool_name": "exec_command" | |
| }, | |
| "label": 1, | |
| "source_event_id": "t15", | |
| "support_probability": 0.9928786158561707 | |
| } | |
| ] | |
| } | |
| ] | |
| }, | |
| "train": "data/tmp/mixed_mode_train.v2.field_decision.jsonl", | |
| "train_candidate_pairs": 63752, | |
| "train_groups": 5882, | |
| "train_rows": 1217 | |
| } | |