Buckets:
| [ | |
| { | |
| "folder": "01-codex-gpt-5.5", | |
| "mapping": { | |
| "leaderboard_row": { | |
| "release": "2026-07", | |
| "framework": "Codex", | |
| "model": "GPT-5.5", | |
| "openness": "closed" | |
| }, | |
| "note": "Trajectory-level only. This framework emits no per-step labels, so the step columns of the leaderboard are empty for this row.", | |
| "benchmark": "CUAStepBench (278 human-annotated trajectories)", | |
| "ground_truth_judge": "human-annotated-zzh", | |
| "runs": { | |
| "codex-gpt-5.5": { | |
| "source_judge_dir": "codex", | |
| "result_json": 278, | |
| "dump_jsonl": 0, | |
| "extra_files": 278, | |
| "bytes": 37505161 | |
| } | |
| }, | |
| "excluded_runs": [] | |
| }, | |
| "metrics": { | |
| "benchmark": "CUAStepBench", | |
| "ground_truth_judge": "human-annotated-zzh", | |
| "analysis_code": "seekjudge.analysis.analyse_judge (SeekJudge release)", | |
| "scorer_pickle_required": false, | |
| "leaderboard_row": { | |
| "release": "2026-07", | |
| "framework": "Codex", | |
| "model": "GPT-5.5", | |
| "openness": "closed" | |
| }, | |
| "n_runs": 1, | |
| "per_run": { | |
| "codex-gpt-5.5": { | |
| "cases": 278, | |
| "verdict_source": { | |
| "stored_verdict": 278, | |
| "scored_from_subcriteria": 0 | |
| }, | |
| "trajectory": { | |
| "accuracy": 79.14, | |
| "precision": 87.18, | |
| "recall": 70.34, | |
| "f1": 77.86 | |
| }, | |
| "step": null | |
| } | |
| }, | |
| "mean_over_runs": { | |
| "trajectory": { | |
| "accuracy": 79.14, | |
| "precision": 87.18, | |
| "recall": 70.34, | |
| "f1": 77.86 | |
| }, | |
| "step": null | |
| }, | |
| "published_leaderboard": { | |
| "trajectory": { | |
| "accuracy": 79.1, | |
| "precision": 87.1, | |
| "recall": 70.1, | |
| "f1": 77.7 | |
| }, | |
| "step": { | |
| "accuracy": null, | |
| "precision": null, | |
| "recall": null, | |
| "f1": null | |
| } | |
| } | |
| } | |
| }, | |
| { | |
| "folder": "02-seekjudge-deepseek-v4-pro", | |
| "mapping": { | |
| "leaderboard_row": { | |
| "release": "2026-07", | |
| "framework": "SeekJudge", | |
| "model": "DeepSeek-V4-Pro + Gemini-3.0-Flash", | |
| "openness": "closed" | |
| }, | |
| "note": "DeepSeek-V4-Pro drives the Seek agent, Gemini-3.0-Flash drives the condense/analyse VLM. Reported in the framework ablation of the paper, which lists F1 only.", | |
| "benchmark": "CUAStepBench (278 human-annotated trajectories)", | |
| "ground_truth_judge": "human-annotated-zzh", | |
| "runs": { | |
| "seekjudge-deepseek-v4-pro": { | |
| "source_judge_dir": "inequal_rewarder-datagen-28-dsv4pro-thinking-turn8", | |
| "result_json": 278, | |
| "dump_jsonl": 556, | |
| "extra_files": 0, | |
| "bytes": 112983455 | |
| }, | |
| "seekjudge-deepseek-v4-pro_rand1": { | |
| "source_judge_dir": "inequal_rewarder-datagen-28-dsv4pro-thinking-turn8_rand1", | |
| "result_json": 278, | |
| "dump_jsonl": 556, | |
| "extra_files": 0, | |
| "bytes": 122044767 | |
| } | |
| }, | |
| "excluded_runs": [] | |
| }, | |
| "metrics": { | |
| "benchmark": "CUAStepBench", | |
| "ground_truth_judge": "human-annotated-zzh", | |
| "analysis_code": "seekjudge.analysis.analyse_judge (SeekJudge release)", | |
| "scorer_pickle_required": false, | |
| "leaderboard_row": { | |
| "release": "2026-07", | |
| "framework": "SeekJudge", | |
| "model": "DeepSeek-V4-Pro + Gemini-3.0-Flash", | |
| "openness": "closed" | |
| }, | |
| "n_runs": 2, | |
| "per_run": { | |
| "seekjudge-deepseek-v4-pro": { | |
| "cases": 278, | |
| "verdict_source": { | |
| "stored_verdict": 278, | |
| "scored_from_subcriteria": 0 | |
| }, | |
| "trajectory": { | |
| "accuracy": 69.78, | |
| "precision": 67.23, | |
| "recall": 82.07, | |
| "f1": 73.91 | |
| }, | |
| "step": { | |
| "n_tasks": 277, | |
| "n_steps": 5550, | |
| "accuracy": 89.82, | |
| "precision": 46.32, | |
| "recall": 41.51, | |
| "f1": 43.78 | |
| } | |
| }, | |
| "seekjudge-deepseek-v4-pro_rand1": { | |
| "cases": 278, | |
| "verdict_source": { | |
| "stored_verdict": 278, | |
| "scored_from_subcriteria": 0 | |
| }, | |
| "trajectory": { | |
| "accuracy": 69.06, | |
| "precision": 66.67, | |
| "recall": 81.38, | |
| "f1": 73.29 | |
| }, | |
| "step": { | |
| "n_tasks": 277, | |
| "n_steps": 5550, | |
| "accuracy": 89.84, | |
| "precision": 46.47, | |
| "recall": 42.26, | |
| "f1": 44.27 | |
| } | |
| } | |
| }, | |
| "mean_over_runs": { | |
| "trajectory": { | |
| "accuracy": 69.42, | |
| "precision": 66.95, | |
| "recall": 81.72, | |
| "f1": 73.6 | |
| }, | |
| "step": { | |
| "accuracy": 89.83, | |
| "precision": 46.39, | |
| "recall": 41.88, | |
| "f1": 44.03 | |
| } | |
| }, | |
| "published_leaderboard": { | |
| "trajectory": { | |
| "accuracy": null, | |
| "precision": null, | |
| "recall": null, | |
| "f1": 73.6 | |
| }, | |
| "step": { | |
| "accuracy": null, | |
| "precision": null, | |
| "recall": null, | |
| "f1": 44.0 | |
| } | |
| } | |
| } | |
| }, | |
| { | |
| "folder": "03-cuajudge-gpt-5-mini", | |
| "mapping": { | |
| "leaderboard_row": { | |
| "release": "2026-07", | |
| "framework": "CUAJudge", | |
| "model": "GPT-5-mini", | |
| "openness": "closed" | |
| }, | |
| "note": "CUAJudge has no native step-level judgment. Its step labels come from the step extraction procedure of SeekJudge applied to the same base model.", | |
| "benchmark": "CUAStepBench (278 human-annotated trajectories)", | |
| "ground_truth_judge": "human-annotated-zzh", | |
| "runs": { | |
| "cuajudge-gpt-5-mini": { | |
| "source_judge_dir": "cuajudge-step-gpt5-mini", | |
| "result_json": 278, | |
| "dump_jsonl": 0, | |
| "extra_files": 0, | |
| "bytes": 2031433 | |
| } | |
| }, | |
| "excluded_runs": [] | |
| }, | |
| "metrics": { | |
| "benchmark": "CUAStepBench", | |
| "ground_truth_judge": "human-annotated-zzh", | |
| "analysis_code": "seekjudge.analysis.analyse_judge (SeekJudge release)", | |
| "scorer_pickle_required": false, | |
| "leaderboard_row": { | |
| "release": "2026-07", | |
| "framework": "CUAJudge", | |
| "model": "GPT-5-mini", | |
| "openness": "closed" | |
| }, | |
| "n_runs": 1, | |
| "per_run": { | |
| "cuajudge-gpt-5-mini": { | |
| "cases": 278, | |
| "verdict_source": { | |
| "stored_verdict": 278, | |
| "scored_from_subcriteria": 0 | |
| }, | |
| "trajectory": { | |
| "accuracy": 70.86, | |
| "precision": 78.57, | |
| "recall": 60.69, | |
| "f1": 68.48 | |
| }, | |
| "step": { | |
| "n_tasks": 266, | |
| "n_steps": 5281, | |
| "accuracy": 90.36, | |
| "precision": 49.52, | |
| "recall": 30.63, | |
| "f1": 37.85 | |
| } | |
| } | |
| }, | |
| "mean_over_runs": { | |
| "trajectory": { | |
| "accuracy": 70.86, | |
| "precision": 78.57, | |
| "recall": 60.69, | |
| "f1": 68.48 | |
| }, | |
| "step": { | |
| "accuracy": 90.36, | |
| "precision": 49.52, | |
| "recall": 30.63, | |
| "f1": 37.85 | |
| } | |
| }, | |
| "published_leaderboard": { | |
| "trajectory": { | |
| "accuracy": 71.1, | |
| "precision": 78.6, | |
| "recall": 61.1, | |
| "f1": 68.8 | |
| }, | |
| "step": { | |
| "accuracy": 90.4, | |
| "precision": 49.7, | |
| "recall": 30.8, | |
| "f1": 38.0 | |
| } | |
| } | |
| } | |
| }, | |
| { | |
| "folder": "04-seekjudge-9b", | |
| "mapping": { | |
| "leaderboard_row": { | |
| "release": "2026-07", | |
| "framework": "SeekJudge", | |
| "model": "SeekJudge-9B", | |
| "openness": "open" | |
| }, | |
| "note": "Four sampled runs. In the base run 4 of the 278 cases carry an empty subcriteria dict and are dropped by the analysis, leaving 274; the three _rand runs are complete at 278.", | |
| "benchmark": "CUAStepBench (278 human-annotated trajectories)", | |
| "ground_truth_judge": "human-annotated-zzh", | |
| "runs": { | |
| "seekjudge-9b": { | |
| "source_judge_dir": "inequal_rewarder-9b-trained-v28-forceconclusion-4.19-epoch2", | |
| "result_json": 278, | |
| "dump_jsonl": 0, | |
| "extra_files": 0, | |
| "bytes": 277024 | |
| }, | |
| "seekjudge-9b_rand1": { | |
| "source_judge_dir": "inequal_rewarder-9b-trained-v28-forceconclusion-4.19-epoch2_rand1", | |
| "result_json": 278, | |
| "dump_jsonl": 0, | |
| "extra_files": 0, | |
| "bytes": 280168 | |
| }, | |
| "seekjudge-9b_rand2": { | |
| "source_judge_dir": "inequal_rewarder-9b-trained-v28-forceconclusion-4.19-epoch2_rand2", | |
| "result_json": 278, | |
| "dump_jsonl": 0, | |
| "extra_files": 0, | |
| "bytes": 279899 | |
| }, | |
| "seekjudge-9b_rand3": { | |
| "source_judge_dir": "inequal_rewarder-9b-trained-v28-forceconclusion-4.19-epoch2_rand3", | |
| "result_json": 278, | |
| "dump_jsonl": 793, | |
| "extra_files": 0, | |
| "bytes": 99419510 | |
| } | |
| }, | |
| "excluded_runs": [ | |
| { | |
| "source_judge_dir": "inequal_rewarder-9b-trained-v28-forceconclusion-4.19-epoch2_rand4", | |
| "cases": 43, | |
| "reason": "partial run, 43/278 cases" | |
| } | |
| ] | |
| }, | |
| "metrics": { | |
| "benchmark": "CUAStepBench", | |
| "ground_truth_judge": "human-annotated-zzh", | |
| "analysis_code": "seekjudge.analysis.analyse_judge (SeekJudge release)", | |
| "scorer_pickle_required": false, | |
| "leaderboard_row": { | |
| "release": "2026-07", | |
| "framework": "SeekJudge", | |
| "model": "SeekJudge-9B", | |
| "openness": "open" | |
| }, | |
| "n_runs": 4, | |
| "per_run": { | |
| "seekjudge-9b": { | |
| "cases": 274, | |
| "verdict_source": { | |
| "stored_verdict": 274, | |
| "scored_from_subcriteria": 0 | |
| }, | |
| "trajectory": { | |
| "accuracy": 75.18, | |
| "precision": 75.0, | |
| "recall": 78.17, | |
| "f1": 76.55 | |
| }, | |
| "step": { | |
| "n_tasks": 273, | |
| "n_steps": 5449, | |
| "accuracy": 89.58, | |
| "precision": 44.83, | |
| "recall": 34.6, | |
| "f1": 39.06 | |
| } | |
| }, | |
| "seekjudge-9b_rand1": { | |
| "cases": 278, | |
| "verdict_source": { | |
| "stored_verdict": 278, | |
| "scored_from_subcriteria": 0 | |
| }, | |
| "trajectory": { | |
| "accuracy": 71.17, | |
| "precision": 72.34, | |
| "recall": 71.83, | |
| "f1": 72.08 | |
| }, | |
| "step": { | |
| "n_tasks": 273, | |
| "n_steps": 5449, | |
| "accuracy": 89.37, | |
| "precision": 43.22, | |
| "recall": 32.13, | |
| "f1": 36.86 | |
| } | |
| }, | |
| "seekjudge-9b_rand2": { | |
| "cases": 278, | |
| "verdict_source": { | |
| "stored_verdict": 278, | |
| "scored_from_subcriteria": 0 | |
| }, | |
| "trajectory": { | |
| "accuracy": 73.36, | |
| "precision": 72.26, | |
| "recall": 78.87, | |
| "f1": 75.42 | |
| }, | |
| "step": { | |
| "n_tasks": 273, | |
| "n_steps": 5449, | |
| "accuracy": 89.7, | |
| "precision": 45.38, | |
| "recall": 32.7, | |
| "f1": 38.01 | |
| } | |
| }, | |
| "seekjudge-9b_rand3": { | |
| "cases": 278, | |
| "verdict_source": { | |
| "stored_verdict": 278, | |
| "scored_from_subcriteria": 0 | |
| }, | |
| "trajectory": { | |
| "accuracy": 74.09, | |
| "precision": 74.15, | |
| "recall": 76.76, | |
| "f1": 75.43 | |
| }, | |
| "step": { | |
| "n_tasks": 273, | |
| "n_steps": 5449, | |
| "accuracy": 88.59, | |
| "precision": 38.24, | |
| "recall": 29.66, | |
| "f1": 33.4 | |
| } | |
| } | |
| }, | |
| "mean_over_runs": { | |
| "trajectory": { | |
| "accuracy": 73.45, | |
| "precision": 73.44, | |
| "recall": 76.41, | |
| "f1": 74.87 | |
| }, | |
| "step": { | |
| "accuracy": 89.31, | |
| "precision": 42.92, | |
| "recall": 32.27, | |
| "f1": 36.83 | |
| } | |
| }, | |
| "published_leaderboard": { | |
| "trajectory": { | |
| "accuracy": 73.1, | |
| "precision": 73.0, | |
| "recall": 76.1, | |
| "f1": 74.5 | |
| }, | |
| "step": { | |
| "accuracy": 89.6, | |
| "precision": 44.5, | |
| "recall": 33.3, | |
| "f1": 38.1 | |
| } | |
| } | |
| } | |
| }, | |
| { | |
| "folder": "05-seekjudge-qwen3vl-8b", | |
| "mapping": { | |
| "leaderboard_row": { | |
| "release": "2026-07", | |
| "framework": "SeekJudge", | |
| "model": "Qwen3VL-8B", | |
| "openness": "open" | |
| }, | |
| "note": "Six sampled runs. _rand1 is missing 1 of the 278 cases (277 present); the other five are complete.", | |
| "benchmark": "CUAStepBench (278 human-annotated trajectories)", | |
| "ground_truth_judge": "human-annotated-zzh", | |
| "runs": { | |
| "seekjudge-qwen3vl-8b": { | |
| "source_judge_dir": "inequal_rewarder-datagen-qwen3vlbase", | |
| "result_json": 278, | |
| "dump_jsonl": 0, | |
| "extra_files": 0, | |
| "bytes": 279639 | |
| }, | |
| "seekjudge-qwen3vl-8b_rand1": { | |
| "source_judge_dir": "inequal_rewarder-datagen-qwen3vlbase_rand1", | |
| "result_json": 277, | |
| "dump_jsonl": 0, | |
| "extra_files": 0, | |
| "bytes": 278992 | |
| }, | |
| "seekjudge-qwen3vl-8b_rand2": { | |
| "source_judge_dir": "inequal_rewarder-datagen-qwen3vlbase_rand2", | |
| "result_json": 278, | |
| "dump_jsonl": 0, | |
| "extra_files": 0, | |
| "bytes": 287975 | |
| }, | |
| "seekjudge-qwen3vl-8b_rand3": { | |
| "source_judge_dir": "inequal_rewarder-datagen-qwen3vlbase_rand3", | |
| "result_json": 278, | |
| "dump_jsonl": 553, | |
| "extra_files": 0, | |
| "bytes": 95267627 | |
| }, | |
| "seekjudge-qwen3vl-8b_rand4": { | |
| "source_judge_dir": "inequal_rewarder-datagen-qwen3vlbase_rand4", | |
| "result_json": 278, | |
| "dump_jsonl": 550, | |
| "extra_files": 0, | |
| "bytes": 95025284 | |
| }, | |
| "seekjudge-qwen3vl-8b_rand5": { | |
| "source_judge_dir": "inequal_rewarder-datagen-qwen3vlbase_rand5", | |
| "result_json": 278, | |
| "dump_jsonl": 553, | |
| "extra_files": 0, | |
| "bytes": 94042836 | |
| } | |
| }, | |
| "excluded_runs": [] | |
| }, | |
| "metrics": { | |
| "benchmark": "CUAStepBench", | |
| "ground_truth_judge": "human-annotated-zzh", | |
| "analysis_code": "seekjudge.analysis.analyse_judge (SeekJudge release)", | |
| "scorer_pickle_required": false, | |
| "leaderboard_row": { | |
| "release": "2026-07", | |
| "framework": "SeekJudge", | |
| "model": "Qwen3VL-8B", | |
| "openness": "open" | |
| }, | |
| "n_runs": 6, | |
| "per_run": { | |
| "seekjudge-qwen3vl-8b": { | |
| "cases": 278, | |
| "verdict_source": { | |
| "stored_verdict": 278, | |
| "scored_from_subcriteria": 0 | |
| }, | |
| "trajectory": { | |
| "accuracy": 63.54, | |
| "precision": 60.19, | |
| "recall": 88.19, | |
| "f1": 71.55 | |
| }, | |
| "step": { | |
| "n_tasks": 276, | |
| "n_steps": 4070, | |
| "accuracy": 89.43, | |
| "precision": 41.97, | |
| "recall": 20.3, | |
| "f1": 27.36 | |
| } | |
| }, | |
| "seekjudge-qwen3vl-8b_rand1": { | |
| "cases": 277, | |
| "verdict_source": { | |
| "stored_verdict": 277, | |
| "scored_from_subcriteria": 0 | |
| }, | |
| "trajectory": { | |
| "accuracy": 61.73, | |
| "precision": 58.96, | |
| "recall": 86.81, | |
| "f1": 70.22 | |
| }, | |
| "step": { | |
| "n_tasks": 276, | |
| "n_steps": 4104, | |
| "accuracy": 89.79, | |
| "precision": 45.21, | |
| "recall": 21.2, | |
| "f1": 28.86 | |
| } | |
| }, | |
| "seekjudge-qwen3vl-8b_rand2": { | |
| "cases": 278, | |
| "verdict_source": { | |
| "stored_verdict": 278, | |
| "scored_from_subcriteria": 0 | |
| }, | |
| "trajectory": { | |
| "accuracy": 61.01, | |
| "precision": 57.96, | |
| "recall": 90.97, | |
| "f1": 70.81 | |
| }, | |
| "step": { | |
| "n_tasks": 276, | |
| "n_steps": 5222, | |
| "accuracy": 89.41, | |
| "precision": 42.01, | |
| "recall": 22.16, | |
| "f1": 29.01 | |
| } | |
| }, | |
| "seekjudge-qwen3vl-8b_rand3": { | |
| "cases": 278, | |
| "verdict_source": { | |
| "stored_verdict": 278, | |
| "scored_from_subcriteria": 0 | |
| }, | |
| "trajectory": { | |
| "accuracy": 62.09, | |
| "precision": 58.99, | |
| "recall": 88.89, | |
| "f1": 70.91 | |
| }, | |
| "step": { | |
| "n_tasks": 276, | |
| "n_steps": 5227, | |
| "accuracy": 90.51, | |
| "precision": 53.11, | |
| "recall": 28.27, | |
| "f1": 36.9 | |
| } | |
| }, | |
| "seekjudge-qwen3vl-8b_rand4": { | |
| "cases": 278, | |
| "verdict_source": { | |
| "stored_verdict": 278, | |
| "scored_from_subcriteria": 0 | |
| }, | |
| "trajectory": { | |
| "accuracy": 62.82, | |
| "precision": 59.11, | |
| "recall": 92.36, | |
| "f1": 72.09 | |
| }, | |
| "step": { | |
| "n_tasks": 276, | |
| "n_steps": 5275, | |
| "accuracy": 88.93, | |
| "precision": 35.34, | |
| "recall": 15.89, | |
| "f1": 21.93 | |
| } | |
| }, | |
| "seekjudge-qwen3vl-8b_rand5": { | |
| "cases": 278, | |
| "verdict_source": { | |
| "stored_verdict": 278, | |
| "scored_from_subcriteria": 0 | |
| }, | |
| "trajectory": { | |
| "accuracy": 59.21, | |
| "precision": 56.89, | |
| "recall": 88.89, | |
| "f1": 69.38 | |
| }, | |
| "step": { | |
| "n_tasks": 276, | |
| "n_steps": 5228, | |
| "accuracy": 88.66, | |
| "precision": 33.05, | |
| "recall": 15.2, | |
| "f1": 20.83 | |
| } | |
| } | |
| }, | |
| "mean_over_runs": { | |
| "trajectory": { | |
| "accuracy": 61.73, | |
| "precision": 58.68, | |
| "recall": 89.35, | |
| "f1": 70.83 | |
| }, | |
| "step": { | |
| "accuracy": 89.45, | |
| "precision": 41.78, | |
| "recall": 20.5, | |
| "f1": 27.48 | |
| } | |
| }, | |
| "published_leaderboard": { | |
| "trajectory": { | |
| "accuracy": 61.7, | |
| "precision": 58.7, | |
| "recall": 89.4, | |
| "f1": 70.8 | |
| }, | |
| "step": { | |
| "accuracy": 89.4, | |
| "precision": 41.4, | |
| "recall": 20.4, | |
| "f1": 27.3 | |
| } | |
| } | |
| } | |
| }, | |
| { | |
| "folder": "06-cuajudge-qwen3vl-8b", | |
| "mapping": { | |
| "leaderboard_row": { | |
| "release": "2026-07", | |
| "framework": "CUAJudge", | |
| "model": "Qwen3VL-8B", | |
| "openness": "open" | |
| }, | |
| "note": "CUAJudge has no native step-level judgment. Its step labels come from the step extraction procedure of SeekJudge applied to the same base model, and are present for 185 of the 278 cases.", | |
| "benchmark": "CUAStepBench (278 human-annotated trajectories)", | |
| "ground_truth_judge": "human-annotated-zzh", | |
| "runs": { | |
| "cuajudge-qwen3vl-8b": { | |
| "source_judge_dir": "cuajudge_qwen3vl_img16_thres3_step", | |
| "result_json": 278, | |
| "dump_jsonl": 0, | |
| "extra_files": 0, | |
| "bytes": 11866461 | |
| } | |
| }, | |
| "excluded_runs": [] | |
| }, | |
| "metrics": { | |
| "benchmark": "CUAStepBench", | |
| "ground_truth_judge": "human-annotated-zzh", | |
| "analysis_code": "seekjudge.analysis.analyse_judge (SeekJudge release)", | |
| "scorer_pickle_required": false, | |
| "leaderboard_row": { | |
| "release": "2026-07", | |
| "framework": "CUAJudge", | |
| "model": "Qwen3VL-8B", | |
| "openness": "open" | |
| }, | |
| "n_runs": 1, | |
| "per_run": { | |
| "cuajudge-qwen3vl-8b": { | |
| "cases": 278, | |
| "verdict_source": { | |
| "stored_verdict": 278, | |
| "scored_from_subcriteria": 0 | |
| }, | |
| "trajectory": { | |
| "accuracy": 67.63, | |
| "precision": 72.0, | |
| "recall": 62.07, | |
| "f1": 66.67 | |
| }, | |
| "step": { | |
| "n_tasks": 185, | |
| "n_steps": 3411, | |
| "accuracy": 90.82, | |
| "precision": 66.67, | |
| "recall": 12.57, | |
| "f1": 21.16 | |
| } | |
| } | |
| }, | |
| "mean_over_runs": { | |
| "trajectory": { | |
| "accuracy": 67.63, | |
| "precision": 72.0, | |
| "recall": 62.07, | |
| "f1": 66.67 | |
| }, | |
| "step": { | |
| "accuracy": 90.82, | |
| "precision": 66.67, | |
| "recall": 12.57, | |
| "f1": 21.16 | |
| } | |
| }, | |
| "published_leaderboard": { | |
| "trajectory": { | |
| "accuracy": 67.9, | |
| "precision": 72.0, | |
| "recall": 62.5, | |
| "f1": 66.9 | |
| }, | |
| "step": { | |
| "accuracy": 90.9, | |
| "precision": 67.7, | |
| "recall": 12.7, | |
| "f1": 21.3 | |
| } | |
| } | |
| } | |
| }, | |
| { | |
| "folder": "07-osthemis-qwen3vl-8b", | |
| "mapping": { | |
| "leaderboard_row": { | |
| "release": "2026-07", | |
| "framework": "OSThemis", | |
| "model": "Qwen3VL-8B", | |
| "openness": "open" | |
| }, | |
| "note": "Three sampled runs, all complete at 278 trajectories. OSThemis has no native step-level judgment. Its step labels come from the step extraction procedure of SeekJudge applied to the same base model.", | |
| "benchmark": "CUAStepBench (278 human-annotated trajectories)", | |
| "ground_truth_judge": "human-annotated-zzh", | |
| "runs": { | |
| "osthemis-qwen3vl-8b": { | |
| "source_judge_dir": "os_themis_qwen3_8b", | |
| "result_json": 278, | |
| "dump_jsonl": 0, | |
| "extra_files": 0, | |
| "bytes": 766221 | |
| }, | |
| "osthemis-qwen3vl-8b_rand1": { | |
| "source_judge_dir": "os_themis_qwen3_8b_rand1", | |
| "result_json": 278, | |
| "dump_jsonl": 0, | |
| "extra_files": 0, | |
| "bytes": 772009 | |
| }, | |
| "osthemis-qwen3vl-8b_rand2": { | |
| "source_judge_dir": "os_themis_qwen3_8b_rand2", | |
| "result_json": 278, | |
| "dump_jsonl": 0, | |
| "extra_files": 0, | |
| "bytes": 771477 | |
| } | |
| }, | |
| "excluded_runs": [ | |
| { | |
| "source_judge_dir": "os_themis_qwen3_8b_rand3", | |
| "cases": 43, | |
| "reason": "partial run, 43/278 cases" | |
| } | |
| ] | |
| }, | |
| "metrics": { | |
| "benchmark": "CUAStepBench", | |
| "ground_truth_judge": "human-annotated-zzh", | |
| "analysis_code": "seekjudge.analysis.analyse_judge (SeekJudge release)", | |
| "scorer_pickle_required": false, | |
| "leaderboard_row": { | |
| "release": "2026-07", | |
| "framework": "OSThemis", | |
| "model": "Qwen3VL-8B", | |
| "openness": "open" | |
| }, | |
| "n_runs": 3, | |
| "per_run": { | |
| "osthemis-qwen3vl-8b": { | |
| "cases": 278, | |
| "verdict_source": { | |
| "stored_verdict": 278, | |
| "scored_from_subcriteria": 0 | |
| }, | |
| "trajectory": { | |
| "accuracy": 61.87, | |
| "precision": 69.31, | |
| "recall": 48.28, | |
| "f1": 56.91 | |
| }, | |
| "step": { | |
| "n_tasks": 272, | |
| "n_steps": 4223, | |
| "accuracy": 83.38, | |
| "precision": 16.67, | |
| "recall": 12.66, | |
| "f1": 14.39 | |
| } | |
| }, | |
| "osthemis-qwen3vl-8b_rand1": { | |
| "cases": 278, | |
| "verdict_source": { | |
| "stored_verdict": 278, | |
| "scored_from_subcriteria": 0 | |
| }, | |
| "trajectory": { | |
| "accuracy": 65.47, | |
| "precision": 72.9, | |
| "recall": 53.79, | |
| "f1": 61.9 | |
| }, | |
| "step": { | |
| "n_tasks": 270, | |
| "n_steps": 4157, | |
| "accuracy": 83.83, | |
| "precision": 18.23, | |
| "recall": 15.63, | |
| "f1": 16.83 | |
| } | |
| }, | |
| "osthemis-qwen3vl-8b_rand2": { | |
| "cases": 278, | |
| "verdict_source": { | |
| "stored_verdict": 278, | |
| "scored_from_subcriteria": 0 | |
| }, | |
| "trajectory": { | |
| "accuracy": 64.39, | |
| "precision": 71.7, | |
| "recall": 52.41, | |
| "f1": 60.56 | |
| }, | |
| "step": { | |
| "n_tasks": 273, | |
| "n_steps": 4312, | |
| "accuracy": 83.6, | |
| "precision": 16.16, | |
| "recall": 14.58, | |
| "f1": 15.33 | |
| } | |
| } | |
| }, | |
| "mean_over_runs": { | |
| "trajectory": { | |
| "accuracy": 63.91, | |
| "precision": 71.3, | |
| "recall": 51.49, | |
| "f1": 59.79 | |
| }, | |
| "step": { | |
| "accuracy": 83.6, | |
| "precision": 17.02, | |
| "recall": 14.29, | |
| "f1": 15.52 | |
| } | |
| }, | |
| "published_leaderboard": { | |
| "trajectory": { | |
| "accuracy": 63.8, | |
| "precision": 71.1, | |
| "recall": 51.2, | |
| "f1": 59.5 | |
| }, | |
| "step": { | |
| "accuracy": 83.6, | |
| "precision": 17.0, | |
| "recall": 14.3, | |
| "f1": 15.6 | |
| } | |
| } | |
| } | |
| } | |
| ] |
Xet Storage Details
- Size:
- 25.6 kB
- Xet hash:
- 47b91c8e0ccf435a8b3890d0cbb048c42303e714ecf6c8217d5d78b35f7c1a9a
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.